piclets-backend / app.py
Fraser's picture
Switch caption stage to joy-caption-beta-one
c850f47
Raw
History Blame Contribute Delete
35.6 kB
"""
app.py — Piclets Discovery Server (backend).
A free CPU Space that is the *only* writer to a public HF Dataset (the shared
monster database). It does exactly two jobs:
1. Orchestrate the AI pipeline for a scan, forwarding the *player's* HF token to
three ZeroGPU Spaces so GPU time is billed to the player, not to us:
photo -> identify object (VLM) -> design monster (LLM, JSON) -> art (T2I)
2. Persist a genuinely-new monster to the dataset in a single commit, and keep
the small aggregate index files (dex / feed / leaderboard / stats) in sync.
Everything the frontend *reads* (dex, feed, leaderboard, a user's collection)
is fetched by the browser directly from the dataset CDN — it never touches this
server. See ARCHITECTURE.md for the full design and the platform limits that
shape it.
Two tokens, never confused:
- HF_TOKEN (env secret) : OUR write token for the dataset. Never leaves here.
- hf_token (per call) : the PLAYER's OAuth token. Used only to (a) prove
identity and (b) call the AI Spaces on their quota.
"""
from __future__ import annotations
import io
import os
import re
import json
import time
import tempfile
import threading
from collections import defaultdict
from datetime import datetime, timezone
import requests
from PIL import Image
import gradio as gr
from gradio_client import Client, handle_file
from huggingface_hub import HfApi, hf_hub_download, CommitOperationAdd, CommitOperationDelete
from auth import verify_hf_token
# ============================================================================
# Configuration (all overridable via Space Variables / Secrets)
# ============================================================================
HF_TOKEN = os.getenv("HF_API_KEY") or os.getenv("HF_TOKEN") # SECRET: our dataset write token
DATASET_REPO = os.getenv("DATASET_REPO", "Fraser/Pictuary") # the public "database"
ADMIN_TOKEN = os.getenv("ADMIN_TOKEN") # SECRET: guards /admin_* endpoints
# --- The three ZeroGPU Spaces we orchestrate -------------------------------
# Called with the *player's* token so their quota is spent, not ours. To swap in
# newer models, change only these IDs and (if their signature differs) the three
# functions in the AI layer below — nothing else in the app depends on them.
# >>> VERIFY EACH SIGNATURE with view_api() before trusting it. See ARCHITECTURE.md
# "Swapping the AI Spaces". Quick check:
# from gradio_client import Client
# Client("krea/Krea-2").view_api()
CAPTION_SPACE = os.getenv("CAPTION_SPACE", "fancyfeast/joy-caption-beta-one")
CONCEPT_SPACE = os.getenv("CONCEPT_SPACE", "huggingface-projects/gemma-4-12b-it")
IMAGE_SPACE = os.getenv("IMAGE_SPACE", "krea/Krea-2")
# --- Safety / limits --------------------------------------------------------
MAX_IMAGE_BYTES = int(os.getenv("MAX_IMAGE_BYTES", str(16 * 1024 * 1024))) # reject uploads > 16 MB; downscaled before the AI call
OUTPUT_IMAGE_MAX = int(os.getenv("OUTPUT_IMAGE_MAX", "768")) # px, longest side of stored art
SCAN_WINDOW_S = int(os.getenv("SCAN_WINDOW_S", "600")) # rate-limit window (10 min)
SCAN_MAX_IN_WINDOW = int(os.getenv("SCAN_MAX_IN_WINDOW", "30")) # scans per window per user
CONCURRENCY = int(os.getenv("CONCURRENCY", "8")) # simultaneous scans in flight
FEED_SIZE = 50
LEADERBOARD_SIZE = 100
# The 10 monster categories (reuse the existing type logos on the frontend).
MONSTER_TYPES = [
"beast", "bug", "aquatic", "flora", "mineral",
"space", "machina", "structure", "culture", "cuisine",
]
api = HfApi(token=HF_TOKEN)
# ============================================================================
# In-memory state — safe because a free CPU Space runs as a single replica.
# The dataset is the source of truth; this is a hot cache + the dedup index.
# ============================================================================
_write_lock = threading.Lock() # serializes commits (one writer)
_state_ready = False
_dedup_keys : set[str] = set() # normalized object keys known to exist
_dex : list[dict] = [] # index/monsters.json (all monster summaries)
_feed : list[dict] = [] # index/feed.json (recent discoveries)
_user_scores : dict[str, dict] = {} # sub -> leaderboard entry
_stats : dict = {
"total_monsters": 0, "total_users": 0, "total_rarity_all": 0, "last_updated": None,
}
_scan_times : dict[str, list[float]] = defaultdict(list) # sub -> recent scan timestamps
_token_cache : dict[str, tuple[dict, float]] = {} # token -> (userinfo, expiry)
# ============================================================================
# Small helpers
# ============================================================================
def _now_iso() -> str:
return datetime.now(timezone.utc).isoformat()
def _dataset_url(path: str) -> str:
"""Public CDN (resolver) URL for a file in the dataset."""
return f"https://huggingface.co/datasets/{DATASET_REPO}/resolve/main/{path}"
_ARTICLE_RE = re.compile(r"^(the|a|an)\s+")
def normalize_object_name(name: str) -> str:
"""Canonical dedup key: lowercase, drop articles, strip punctuation, light
singularization, spaces -> underscores. 'The Blue Pillows' -> 'blue_pillow'."""
name = (name or "").strip().lower()
name = _ARTICLE_RE.sub("", name)
name = re.sub(r"[^a-z0-9\s]", "", name)
out = []
for w in name.split():
if len(w) > 4 and w.endswith("ies"):
w = w[:-3] + "y"
elif len(w) > 4 and w.endswith("ves"):
w = w[:-3] + "f"
elif len(w) > 3 and w.endswith("es") and not w.endswith(("ses", "xes", "zes", "ches", "shes")):
w = w[:-2]
elif len(w) > 3 and w.endswith("s") and not w.endswith("ss"):
w = w[:-1]
out.append(w)
return "_".join(out)
def _num(v, default: float, lo: float, hi: float) -> float:
try:
f = float(v)
except (TypeError, ValueError):
return default
if f != f: # NaN
return default
return max(lo, min(hi, f))
def _download_json(path: str, default):
"""Read a JSON file from the dataset (with our token). Returns default if absent."""
try:
local = hf_hub_download(DATASET_REPO, path, repo_type="dataset", token=HF_TOKEN)
with open(local, encoding="utf-8") as f:
return json.load(f)
except Exception:
return default
# ============================================================================
# Startup: load the aggregate indices into memory (build the dedup set)
# ============================================================================
def load_state() -> None:
global _dex, _feed, _user_scores, _stats, _dedup_keys, _state_ready
_dex = _download_json("index/monsters.json", [])
_feed = _download_json("index/feed.json", [])
lb = _download_json("index/leaderboard.json", [])
_user_scores = {e["sub"]: e for e in lb if isinstance(e, dict) and e.get("sub")}
_stats = _download_json("index/stats.json", _stats)
_dedup_keys = {m["key"] for m in _dex if isinstance(m, dict) and m.get("key")}
_state_ready = True
print(f"[state] loaded {len(_dedup_keys)} monsters, {len(_user_scores)} users")
# ============================================================================
# Identity, rate limiting, input validation
# ============================================================================
def _identify(token: str | None) -> dict | None:
"""Verify token -> userinfo, cached for 5 minutes to avoid re-hitting userinfo."""
if not token:
return None
now = time.time()
cached = _token_cache.get(token)
if cached and cached[1] > now:
return cached[0]
info = verify_hf_token(token)
if info:
_token_cache[token] = (info, now + 300)
return info
def _rate_ok(sub: str) -> bool:
now = time.time()
cutoff = now - SCAN_WINDOW_S
times = _scan_times[sub]
times[:] = [t for t in times if t > cutoff]
if len(times) >= SCAN_MAX_IN_WINDOW:
return False
times.append(now)
return True
def _validate_input_image(path: str | None) -> str | None:
"""Return an error string if the upload is unacceptable, else None."""
if not path or not os.path.exists(path):
return "No image was received."
size = os.path.getsize(path)
if size > MAX_IMAGE_BYTES:
return (f"That image is {size // 1024} KB, over the {MAX_IMAGE_BYTES // 1024 // 1024} MB limit. "
"The app should downscale photos before uploading.")
try:
with Image.open(path) as im:
im.verify()
except Exception:
return "That file isn't a readable image."
return None
def _prepare_image(path: str) -> str:
"""Downscale an uploaded photo to a max dimension (1024px) before it goes to
the AI Spaces, so a big phone photo doesn't slow or break the caption call.
Returns a fresh temp filepath (PNG). The original upload is left untouched.
This is the practical "don't send huge images" guarantee; true client-side
downscaling would need a JS snippet, but this keeps the pipeline robust
regardless of what the browser uploads.
"""
with Image.open(path) as im:
im = im.convert("RGB")
im.thumbnail((1024, 1024))
tmp = tempfile.NamedTemporaryFile(suffix=".png", delete=False)
im.save(tmp, format="PNG")
tmp.close()
return tmp.name
def _friendly_ai_error(exc: Exception, stage: str) -> str:
msg = str(exc).lower()
quota = ("quota" in msg) or ("429" in msg) or ("gpu" in msg and "exceed" in msg)
if quota:
return ("You've used up your free Hugging Face GPU time for now — it resets "
"daily, or Hugging Face Pro gives you far more. "
f"(Ran out while {stage}.)")
return f"The AI step failed while {stage}. Please try again in a moment."
# ============================================================================
# AI layer — the ONLY place that talks to the model Spaces.
# Each function takes the player's token and forwards it so their quota is used.
# >>> If you swap Spaces, verify these three signatures with view_api(). <<<
# ============================================================================
def caption_object(image_path: str, token: str) -> str:
"""Stage 1 (cheap): identify the object as a short noun phrase. This is the
dedup key, so we run it BEFORE the expensive design/art stages — a repeat
scan of a known object then costs almost no GPU.
fancyfeast/joy-caption-beta-one /chat_joycaption positional args (verified
via view_api()):
(input_image, prompt, temperature, top_p, max_new_tokens, log_prompt)
Returns the caption as a single string. We pass our identify instruction as
the prompt with temperature=0 for deterministic, terse object names (the
'very short' intent is encoded in the prompt's '1 to 4 words' — beta-one
has no separate caption_length knob on this endpoint).
"""
client = Client(CAPTION_SPACE, hf_token=token)
instruction = (
"Identify the single main physical object in this image as a short, generic "
"noun phrase of 1 to 4 words (for example 'ceramic coffee mug' or 'wooden "
"chair'). Ignore the background. Reply with ONLY the object name."
)
result = client.predict(
handle_file(image_path), # input_image
instruction, # prompt
0, # temperature (deterministic)
api_name="/chat_joycaption",
)
text = result if isinstance(result, str) else (
result[1] if isinstance(result, (list, tuple)) and len(result) > 1 else str(result)
)
first_line = (text.strip().splitlines() or [""])[0]
first_line = re.sub(r"^[\"'\s]+|[\"'.\s]+$", "", first_line)
words = first_line.split()
return " ".join(words[:5]) or "unknown object"
CONCEPT_SYSTEM = (
"You are a creature designer for a monster-collection game called Piclets. Given a "
"real-world object, you invent ONE original collectible creature inspired by it. You "
"always reply with exactly one JSON object and nothing else — no prose, no markdown, "
"no code fences."
)
def _concept_prompt(descriptor: str) -> str:
return (
f'Design a Piclet inspired by this object: "{descriptor}".\n\n'
"Return a JSON object with EXACTLY these keys and nothing else:\n"
'- "name": an original creature name, 1-2 words, max 20 characters. Must not '
"contain the object's name.\n"
f'- "type": exactly one of {MONSTER_TYPES}. Pick the best thematic fit.\n'
'- "appearance": 1-3 sentences describing the creature\'s body, colours, features '
"and pose, written for an image generator. Do NOT name the source object and do "
"NOT mention any art style.\n"
'- "description": 1-2 sentences of flavour about its personality or behaviour.\n'
'- "weight_kg": a number (kilograms) that feels right for such a creature.\n'
'- "height_m": a number (metres).\n'
'- "rarity": an integer 1-100 (1 = extremely common, 100 = legendary), based on '
"how unusual or striking the object is.\n\n"
"Reply with only the JSON object."
)
def _extract_json(text: str) -> dict:
text = text.replace("**💬 Response:**", "")
text = re.sub(r"^\s*assistant(final)?\s*", "", text, flags=re.IGNORECASE) # gpt-oss framing
text = re.sub(r"```(?:json)?", "", text) # code fences
start, end = text.find("{"), text.rfind("}")
if start != -1 and end > start:
text = text[start:end + 1]
return json.loads(text)
def generate_concept(descriptor: str, token: str) -> dict:
"""Stage 2: expand the short object description into a full monster spec (JSON).
huggingface-projects/gemma-4-12b-it /chat positional args (verified via
view_api()):
(text, files, history, thinking, max_new_tokens, image_token_budget,
system_prompt, temperature, top_p, top_k, repetition_penalty)
Only `text` is required; we pass defaults for the slots before system_prompt
so we can set the system prompt + temperature. `thinking=False` keeps the
reply a clean JSON object instead of interleaved reasoning.
"""
client = Client(CONCEPT_SPACE, hf_token=token)
result = client.predict(
_concept_prompt(descriptor), # text
None, # files
None, # history
False, # thinking (off -> direct JSON, no reasoning trace)
2000, # max_new_tokens
280, # image_token_budget
CONCEPT_SYSTEM, # system_prompt
0.7, # temperature
api_name="/chat",
)
# gemma returns {"reasoning": "", "content": "<reply>"}; older Spaces returned
# a bare string or a tuple. Normalize to the reply string before JSON parsing.
if isinstance(result, dict):
raw = result.get("content") or result.get("text") or result.get("response") or ""
elif isinstance(result, (list, tuple)) and result:
raw = result[0]
else:
raw = result
raw = raw if isinstance(raw, str) else str(raw)
try:
data = _extract_json(raw)
except Exception:
data = {}
def s(key: str) -> str:
v = data.get(key, "")
return v if isinstance(v, str) else str(v)
mtype = s("type").strip().lower()
if mtype not in MONSTER_TYPES:
mtype = _guess_type(descriptor)
return {
"name": (re.sub(r'[*"`]', "", s("name")).strip()[:40] or "Piclet"),
"type": mtype,
"appearance": (s("appearance").strip()[:600] or f"a small creature inspired by {descriptor}"),
"description": s("description").strip()[:300],
"weight_kg": round(_num(data.get("weight_kg"), 1.0, 0.01, 100000.0), 2),
"height_m": round(_num(data.get("height_m"), 0.3, 0.01, 1000.0), 2),
"rarity": int(_num(data.get("rarity"), 20, 1, 100)),
}
_TYPE_HINTS = {
"bug": ["insect", "bug", "spider", "beetle", "ant", "moth"],
"aquatic": ["fish", "water", "ocean", "sea", "shell", "coral", "boat"],
"flora": ["plant", "flower", "tree", "leaf", "fruit", "vegetable", "wood"],
"mineral": ["rock", "stone", "crystal", "metal", "gem", "gold"],
"space": ["star", "planet", "cosmic", "galaxy", "moon", "rocket"],
"machina": ["machine", "device", "electronic", "engine", "robot", "tool", "computer", "phone", "car"],
"structure": ["building", "bridge", "tower", "house", "furniture", "chair", "table", "lamp"],
"culture": ["book", "art", "music", "toy", "game", "instrument", "paper"],
"cuisine": ["food", "drink", "meal", "snack", "mug", "cup", "bottle", "plate", "coffee"],
"beast": ["animal", "dog", "cat", "bird", "fur", "bear"],
}
def _guess_type(descriptor: str) -> str:
d = (descriptor or "").lower()
for t, hints in _TYPE_HINTS.items():
if any(h in d for h in hints):
return t
return "beast"
def generate_image(appearance: str, token: str) -> bytes:
"""Stage 3: render the creature, then re-encode compactly as WebP.
krea/Krea-2 exposes `/generate` (not `/infer`). Verified positional args:
(prompt, negative_prompt, model, steps, guidance, width, height, seed, randomize)
`prompt` and `negative_prompt` are both required; the rest have sensible
defaults (Turbo, 8 steps, 1024x1024), so we pass only the two prompts and rely
on those defaults. Returns (result, seed); result is an Image filepath/dict
that `_read_image_result` normalizes.
"""
client = Client(IMAGE_SPACE, hf_token=token)
prompt = (
f"{appearance}. Full-body original creature, centered, with a simple "
"thematic background reflecting its habitat, Pokémon-style anime "
"creature design, soft cel shading, clean lines."
)
negative_prompt = (
"text, watermark, signature, logo, blurry, low quality, deformed, "
"extra limbs, extra faces, cluttered background"
)
result = client.predict(prompt, negative_prompt, api_name="/generate")
return _reencode_webp(_read_image_result(result))
def _read_image_result(result) -> bytes:
"""gradio_client image outputs come back as a local temp path, a URL, or a dict.
Normalize to raw bytes."""
ref = result[0] if isinstance(result, (list, tuple)) and result else result
if isinstance(ref, dict):
ref = ref.get("url") or ref.get("path") or ref.get("image") or ref.get("name")
if not isinstance(ref, str):
raise ValueError(f"Unexpected image result type: {type(result)}")
if ref.startswith("http"):
return requests.get(ref, timeout=60).content
with open(ref, "rb") as f:
return f.read()
def _reencode_webp(raw: bytes) -> bytes:
im = Image.open(io.BytesIO(raw)).convert("RGB")
im.thumbnail((OUTPUT_IMAGE_MAX, OUTPUT_IMAGE_MAX))
buf = io.BytesIO()
im.save(buf, format="WEBP", quality=85, method=6)
return buf.getvalue()
# ============================================================================
# Persistence — the single-writer commit path.
# All in-memory indices are only mutated AFTER the commit succeeds, so a failed
# commit never leaves memory ahead of the dataset (no rollback needed).
# ============================================================================
def _persist_new_monster(key: str, monster: dict, user_info: dict, image_bytes: bytes) -> None:
global _dex, _feed, _stats
sub = user_info["sub"]
username = user_info.get("preferred_username") or user_info.get("name") or sub
# -- update the discoverer's user record --
user = _download_json(f"users/{sub}.json", {
"sub": sub, "discoveries": [], "total_rarity": 0,
"discovery_count": 0, "joined_at": _now_iso(),
})
user["username"] = username
user["name"] = user_info.get("name", "")
user["picture"] = user_info.get("picture", "")
user["last_seen"] = _now_iso()
if key not in user["discoveries"]:
user["discoveries"].append(key)
user["total_rarity"] = int(user.get("total_rarity", 0)) + monster["rarity"]
user["discovery_count"] = len(user["discoveries"])
# -- compute prospective index snapshots (don't mutate globals yet) --
summary = {
"key": key, "name": monster["name"], "type": monster["type"],
"rarity": monster["rarity"], "image_url": monster["image_url"],
"discoverer_username": username, "discovered_at": monster["discovered_at"],
}
new_dex = _dex + [summary]
new_feed = ([summary] + _feed)[:FEED_SIZE]
scores = dict(_user_scores)
scores[sub] = {
"sub": sub, "username": username, "picture": user["picture"],
"total_rarity": user["total_rarity"], "discovery_count": user["discovery_count"],
}
leaderboard = sorted(scores.values(), key=lambda e: e["total_rarity"], reverse=True)[:LEADERBOARD_SIZE]
new_stats = {
"total_monsters": len(new_dex),
"total_users": len(scores),
"total_rarity_all": _stats.get("total_rarity_all", 0) + monster["rarity"],
"last_updated": _now_iso(),
}
def json_add(path: str, obj) -> CommitOperationAdd:
blob = json.dumps(obj, ensure_ascii=False, indent=2).encode("utf-8")
return CommitOperationAdd(path_in_repo=path, path_or_fileobj=io.BytesIO(blob))
operations = [
json_add(f"monsters/{key}.json", monster),
CommitOperationAdd(path_in_repo=monster["image_path"], path_or_fileobj=io.BytesIO(image_bytes)),
json_add(f"users/{sub}.json", user),
json_add("index/monsters.json", new_dex),
json_add("index/feed.json", new_feed),
json_add("index/leaderboard.json", leaderboard),
json_add("index/stats.json", new_stats),
]
# ONE commit for the whole discovery. huggingface_hub >= 1.2.0 retries on 429.
api.create_commit(
repo_id=DATASET_REPO, repo_type="dataset",
operations=operations,
commit_message=f"Discover {monster['name']} ({key})",
)
# commit OK -> publish to memory
_dex = new_dex
_feed = new_feed
_user_scores[sub] = scores[sub]
_stats = new_stats
# ============================================================================
# Public endpoint: scan
# ============================================================================
def scan(image, hf_token):
"""Scan a photo. Returns a JSON-able dict:
{success, status: "new"|"existing", descriptor, monster, message}
or {success: False, error}
In production `hf_token` is the player's OAuth token (used for identity + AI
quota, never for dataset writes). In local single-user mode you may leave it
blank and the server's own key (HF_API_KEY) is used for the AI calls too.
"""
if not _state_ready:
return {"success": False, "error": "The server is still waking up — try again in a few seconds."}
# Local-mode fallback: no player token -> use the server's key for identity
# and AI calls. (Production always sends the player's OAuth token, so the
# two-token invariant holds there; this only affects local testing.)
player_token = hf_token or HF_TOKEN
user = _identify(player_token)
if not user:
return {"success": False, "error": "Please sign in with Hugging Face to scan and save discoveries."}
sub = user["sub"]
if not _rate_ok(sub):
return {"success": False, "error": "You're scanning very fast — give it a minute and try again."}
err = _validate_input_image(image)
if err:
return {"success": False, "error": err}
# Shrink the upload before the AI call (big phone photos -> 1024px PNG).
image = _prepare_image(image)
# Stage 1: identify (cheap) -> dedup BEFORE spending GPU on design/art.
try:
descriptor = caption_object(image, player_token)
except Exception as exc:
return {"success": False, "error": _friendly_ai_error(exc, "identifying the object")}
key = normalize_object_name(descriptor)
if not key:
return {"success": False, "error": "Couldn't make out a clear object — try another photo."}
if key in _dedup_keys:
existing = _download_json(f"monsters/{key}.json", None)
if existing:
return {
"success": True, "status": "existing", "descriptor": descriptor,
"monster": existing,
"message": f"{existing.get('name', 'This Piclet')} has already been discovered!",
}
# index/file out of sync (rare) — fall through and (re)create.
# Stages 2 & 3: only for genuinely new objects.
try:
spec = generate_concept(descriptor, player_token)
except Exception as exc:
return {"success": False, "error": _friendly_ai_error(exc, "designing the creature")}
try:
image_bytes = generate_image(spec["appearance"], player_token)
except Exception as exc:
return {"success": False, "error": _friendly_ai_error(exc, "painting the creature")}
monster = {
"key": key,
"descriptor": descriptor,
"name": spec["name"],
"type": spec["type"],
"appearance": spec["appearance"],
"description": spec["description"],
"weight_kg": spec["weight_kg"],
"height_m": spec["height_m"],
"rarity": spec["rarity"],
"image_path": f"images/{key}.webp",
"image_url": _dataset_url(f"images/{key}.webp"),
"discoverer": {
"sub": sub,
"username": user.get("preferred_username") or user.get("name") or sub,
"name": user.get("name", ""),
"picture": user.get("picture", ""),
},
"discovered_at": _now_iso(),
}
with _write_lock:
# Re-check under the lock in case someone discovered the same object
# while we were generating.
if key in _dedup_keys:
existing = _download_json(f"monsters/{key}.json", monster)
return {
"success": True, "status": "existing", "descriptor": descriptor,
"monster": existing,
"message": f"{existing.get('name', 'This Piclet')} was just discovered by someone else!",
}
try:
_persist_new_monster(key, monster, user, image_bytes)
except Exception as exc:
return {"success": False, "error": f"Couldn't save the discovery ({exc}). Your GPU time was not wasted — try again."}
_dedup_keys.add(key)
return {
"success": True, "status": "new", "descriptor": descriptor,
"monster": monster, "message": f"You discovered {monster['name']}!",
}
# ============================================================================
# Admin endpoints (guarded by ADMIN_TOKEN, not user OAuth) — moderation tools.
# ============================================================================
def admin_delete(monster_key: str, admin_token: str):
global _dex, _feed, _stats
if not ADMIN_TOKEN or admin_token != ADMIN_TOKEN:
return {"success": False, "error": "Unauthorized."}
key = normalize_object_name(monster_key)
monster = _download_json(f"monsters/{key}.json", None)
if not monster:
return {"success": False, "error": f"No monster '{key}'."}
with _write_lock:
# remove from the discoverer's record + score
sub = (monster.get("discoverer") or {}).get("sub")
ops = [
CommitOperationDelete(path_in_repo=f"monsters/{key}.json"),
CommitOperationDelete(path_in_repo=monster.get("image_path", f"images/{key}.webp")),
]
if sub:
user = _download_json(f"users/{sub}.json", None)
if user and key in user.get("discoveries", []):
user["discoveries"].remove(key)
user["total_rarity"] = max(0, int(user.get("total_rarity", 0)) - int(monster.get("rarity", 0)))
user["discovery_count"] = len(user["discoveries"])
blob = json.dumps(user, ensure_ascii=False, indent=2).encode("utf-8")
ops.append(CommitOperationAdd(path_in_repo=f"users/{sub}.json", path_or_fileobj=io.BytesIO(blob)))
if sub in _user_scores:
_user_scores[sub]["total_rarity"] = user["total_rarity"]
_user_scores[sub]["discovery_count"] = user["discovery_count"]
# rebuild the in-memory indices without this monster
_dex = [m for m in _dex if m.get("key") != key]
_feed = [m for m in _feed if m.get("key") != key]
leaderboard = sorted(_user_scores.values(), key=lambda e: e["total_rarity"], reverse=True)[:LEADERBOARD_SIZE]
_stats = {
"total_monsters": len(_dex),
"total_users": len(_user_scores),
"total_rarity_all": max(0, _stats.get("total_rarity_all", 0) - int(monster.get("rarity", 0))),
"last_updated": _now_iso(),
}
for path, obj in [("index/monsters.json", _dex), ("index/feed.json", _feed),
("index/leaderboard.json", leaderboard), ("index/stats.json", _stats)]:
blob = json.dumps(obj, ensure_ascii=False, indent=2).encode("utf-8")
ops.append(CommitOperationAdd(path_in_repo=path, path_or_fileobj=io.BytesIO(blob)))
api.create_commit(repo_id=DATASET_REPO, repo_type="dataset",
operations=ops, commit_message=f"Admin delete {key}")
_dedup_keys.discard(key)
return {"success": True, "deleted": key}
def admin_rebuild(admin_token: str):
"""Recompute all index/* files from monsters/ + users/. Use if an index drifts."""
if not ADMIN_TOKEN or admin_token != ADMIN_TOKEN:
return {"success": False, "error": "Unauthorized."}
files = api.list_repo_files(DATASET_REPO, repo_type="dataset")
dex = []
for f in files:
if f.startswith("monsters/") and f.endswith(".json"):
m = _download_json(f, None)
if not m:
continue
dex.append({
"key": m["key"], "name": m["name"], "type": m["type"], "rarity": m["rarity"],
"image_url": m.get("image_url", _dataset_url(m.get("image_path", ""))),
"discoverer_username": (m.get("discoverer") or {}).get("username", ""),
"discovered_at": m.get("discovered_at", ""),
})
dex.sort(key=lambda e: e.get("discovered_at", ""))
feed = list(reversed(dex))[:FEED_SIZE]
scores = {}
for f in files:
if f.startswith("users/") and f.endswith(".json"):
u = _download_json(f, None)
if not u or not u.get("sub"):
continue
scores[u["sub"]] = {
"sub": u["sub"], "username": u.get("username", u["sub"]),
"picture": u.get("picture", ""),
"total_rarity": int(u.get("total_rarity", 0)),
"discovery_count": int(u.get("discovery_count", len(u.get("discoveries", [])))),
}
leaderboard = sorted(scores.values(), key=lambda e: e["total_rarity"], reverse=True)[:LEADERBOARD_SIZE]
stats = {
"total_monsters": len(dex), "total_users": len(scores),
"total_rarity_all": sum(m["rarity"] for m in dex), "last_updated": _now_iso(),
}
def json_add(path, obj):
blob = json.dumps(obj, ensure_ascii=False, indent=2).encode("utf-8")
return CommitOperationAdd(path_in_repo=path, path_or_fileobj=io.BytesIO(blob))
with _write_lock:
api.create_commit(
repo_id=DATASET_REPO, repo_type="dataset",
operations=[json_add("index/monsters.json", dex), json_add("index/feed.json", feed),
json_add("index/leaderboard.json", leaderboard), json_add("index/stats.json", stats)],
commit_message="Admin rebuild indices",
)
global _dex, _feed, _user_scores, _stats, _dedup_keys
_dex, _feed, _user_scores, _stats = dex, feed, scores, stats
_dedup_keys = {m["key"] for m in dex}
return {"success": True, "monsters": len(dex), "users": len(scores)}
# ============================================================================
# Gradio app — an API first; the small UI is handy for manual testing.
# Reads (dex/feed/leaderboard/collection) are NOT served here; the frontend
# fetches them straight from the dataset CDN. See ARCHITECTURE.md "Read paths".
# ============================================================================
READ_PATHS_HELP = f"""
### Read paths (frontend fetches these directly from the dataset CDN — no server load)
- Dex (all monsters): `{_dataset_url('index/monsters.json')}`
- Feed (recent): `{_dataset_url('index/feed.json')}`
- Leaderboard: `{_dataset_url('index/leaderboard.json')}`
- Global stats: `{_dataset_url('index/stats.json')}`
- A monster: `{_dataset_url('monsters/<key>.json')}`
- A user: `{_dataset_url('users/<sub>.json')}` (a user's `discoveries` list -> monster keys)
- Art: `{_dataset_url('images/<key>.webp')}`
### Write path (this server, one endpoint)
`POST /scan` — inputs: `image` (file), `hf_token` (the player's OAuth access token).
"""
with gr.Blocks(title="Piclets Discovery Server") as demo:
gr.Markdown(
"# 🔮 Piclets Discovery Server\n"
"Backend for the Piclets monster-discovery game. The heavy AI runs on ZeroGPU "
"Spaces using **your** token, and new monsters are saved to a public dataset. "
"This page is mainly for manual testing — the game talks to the API."
)
with gr.Tab("Scan"):
img_in = gr.Image(type="filepath", label="Photo of an object")
tok_in = gr.Textbox(
label="Your HF access token (optional)",
type="password",
placeholder="Leave blank to use the server's key (local mode); paste an OAuth token for production",
)
scan_btn = gr.Button("Scan", variant="primary")
scan_out = gr.JSON(label="Result")
scan_btn.click(scan, [img_in, tok_in], scan_out, api_name="scan", concurrency_limit=CONCURRENCY)
with gr.Tab("Admin"):
gr.Markdown("Moderation tools. Requires the `ADMIN_TOKEN` secret, not a user login.")
admin_tok = gr.Textbox(label="Admin token", type="password")
with gr.Row():
del_key = gr.Textbox(label="Monster key to delete (e.g. 'coffee_mug')")
del_btn = gr.Button("Delete monster", variant="stop")
del_out = gr.JSON(label="Result")
del_btn.click(admin_delete, [del_key, admin_tok], del_out, api_name="admin_delete")
rebuild_btn = gr.Button("Rebuild indices from source")
rebuild_out = gr.JSON(label="Result")
rebuild_btn.click(admin_rebuild, [admin_tok], rebuild_out, api_name="admin_rebuild")
gr.Markdown(READ_PATHS_HELP)
demo.queue(default_concurrency_limit=CONCURRENCY)
if __name__ == "__main__":
if not HF_TOKEN:
print("[warn] HF_API_KEY (or HF_TOKEN) is not set — the server cannot write to the "
"dataset. Set it in Space Settings -> Secrets.")
load_state()
demo.launch()