"""ASE-GLM: a minimal Hugging Face code harness. FastAPI serves the built character-grid frontend and proxies chat completions to the Hugging Face inference router. There is no Gradio here on purpose: the whole UI is one canvas, so a framework that renders its own DOM would only be something to fight. """ from __future__ import annotations import json import os import secrets from pathlib import Path import httpx from fastapi import FastAPI, Header, Request from fastapi.responses import FileResponse, JSONResponse, StreamingResponse from fastapi.staticfiles import StaticFiles from starlette.middleware.sessions import SessionMiddleware from . import catalog, chats, credentials, insight, oauth, ratelimit from .stream import Accumulated, build_payload, sse, translate ROUTER_URL = "https://router.huggingface.co/v1/chat/completions" STATIC = Path(__file__).resolve().parent.parent / "static" BUILD = os.environ.get("ASE_BUILD", "dev") app = FastAPI(title="ASE-GLM", docs_url=None, redoc_url=None) # Signed cookie sessions hold the viewer's OAuth token, so it never reaches the # browser as a value a script could read. # # `same_site` is the subtle one. A Space is normally viewed embedded in an # iframe on huggingface.co, which makes this cookie third-party: under `lax` the # browser will not send it on the cross-site navigation back from the OAuth # provider, the session arrives empty, and the callback rejects a state it did # in fact issue. `none` is required for the embedded case, and `none` requires # `Secure`, which a Space has and local http does not -- hence the split. # # Falling back to a random key means a restart signs everyone out, and a # sign-in that spans a restart fails the same way. Set ASE_SESSION_SECRET. app.add_middleware( SessionMiddleware, secret_key=os.environ.get("ASE_SESSION_SECRET") or secrets.token_urlsafe(32), session_cookie="ase_session", https_only=credentials.on_space(), same_site="none" if credentials.on_space() else "lax", max_age=60 * 60 * 8, ) app.include_router(oauth.router) # One per process. See `ratelimit.py`: this guards the Space's own CPU, not # anybody's credits, and is off entirely when ASE_RATE_LIMIT is 0. _limiter = ratelimit.RateLimiter() def _owner(request: Request) -> str | None: """Whose chats these are. Only a signed-in viewer has a shared store: a pasted key identifies nobody, and the Space's own token identifies the owner rather than the reader. """ sess = request.scope.get("session") or {} return sess.get("oauth_user") def _credential(request: Request, user_key: str | None) -> credentials.Credential: """Resolve the credential for this request. OAuth lands here in Phase 3: a Space with ``hf_oauth`` gives the signed-in viewer's token, which is what makes inference bill them rather than us. Until then the session carries no OAuth token and the ladder falls through. """ # `request.session` is a property that asserts rather than raising # AttributeError, so hasattr() triggers the very failure it would guard. sess = request.scope.get("session") or {} return credentials.resolve( oauth_token=sess.get("oauth_token"), oauth_user=sess.get("oauth_user"), user_key=user_key, ) @app.get("/api/config") async def config(request: Request, x_ase_key: str | None = Header(default=None)): cred = _credential(request, x_ase_key) sess = request.scope.get("session") or {} return { "models": [m.to_json() for m in catalog.MODELS], "defaultModel": catalog.DEFAULT_MODEL, "auth": { **cred.to_json(), # Whether sign-in is even offered: off a Space there is no OAuth # app, so the button would only ever produce an error. "canSignIn": oauth.configured(), "profile": sess.get("oauth_profile"), }, "spaceId": os.environ.get("SPACE_ID"), "build": BUILD, "chatStore": chats.state(_owner(request)).to_json(), "rateLimit": { "enabled": ratelimit.enabled(), "perWindow": ratelimit.limit(), "windowSeconds": int(ratelimit.window_s()), }, } @app.get("/api/model") async def model(id: str, request: Request, x_ase_key: str | None = Header(default=None)): if not catalog.is_allowed(id): return JSONResponse({"error": "unknown model"}, status_code=400) cred = _credential(request, x_ase_key) return await insight.model_info(id, cred.token) @app.get("/api/chats") async def list_chats(request: Request): return chats.load_all(_owner(request) or "") @app.put("/api/chats/{chat_id}") async def put_chat(chat_id: str, request: Request): body = await request.json() if not isinstance(body, dict): return JSONResponse({"error": "expected a chat object"}, status_code=400) return chats.save(_owner(request) or "", chat_id, body).to_json() @app.delete("/api/chats/{chat_id}") async def delete_chat(chat_id: str, request: Request): return chats.delete(_owner(request) or "", chat_id).to_json() @app.get("/api/health") async def health(): return {"ok": True, "build": BUILD, "onSpace": credentials.on_space()} @app.post("/api/chat") async def chat(request: Request, x_ase_key: str | None = Header(default=None)): body = await request.json() model_id = body.get("model") or catalog.DEFAULT_MODEL # The model id is interpolated into an upstream request. Allow-list it # rather than sanitising: the set of models we offer is small and known. if not catalog.is_allowed(model_id): return _error_stream(f"Unknown model: {model_id!r}") messages = body.get("messages") if not isinstance(messages, list) or not messages: return _error_stream("No messages in the request.") cred = _credential(request, x_ase_key) if not cred.token: return _error_stream(cred.detail) # Checked after the credential, so somebody who cannot make a request at all # is told that rather than being told to slow down. verdict = _limiter.check( ratelimit.caller_key(_owner(request), request.client.host if request.client else None)) if not verdict.ok: return _error_stream(verdict.detail) spec = catalog.get(model_id) payload = build_payload(body, model_id, bool(spec and spec.reasoning)) async def gen(): acc = Accumulated() headers = { "Authorization": f"Bearer {cred.token}", "Content-Type": "application/json", } try: # No total timeout: a max-effort reasoning request legitimately # thinks for minutes. The read timeout still catches a dead route. timeout = httpx.Timeout(connect=20.0, read=180.0, write=20.0, pool=20.0) async with httpx.AsyncClient(timeout=timeout) as client: async with client.stream("POST", ROUTER_URL, headers=headers, json=payload) as upstream: if upstream.status_code >= 400: detail = (await upstream.aread()).decode("utf-8", "replace") yield sse("error", _clean_error(upstream.status_code, detail)) return async for line in upstream.aiter_lines(): for frame in translate(line, acc): yield frame if not acc.finished: yield sse("stats", json.dumps(acc.stats)) except httpx.HTTPError as e: yield sse("error", f"Upstream connection failed: {type(e).__name__}: {e}") return StreamingResponse( gen(), media_type="text/event-stream", headers={ "Cache-Control": "no-cache, no-transform", "Connection": "keep-alive", # Spaces sit behind a reverse proxy that will otherwise buffer the # whole response and deliver it as one block, which looks exactly # like a model that streams nothing and then answers all at once. "X-Accel-Buffering": "no", }, ) def _clean_error(status: int, detail: str) -> str: """Upstream errors, minus the noise, and never echoing a credential.""" try: parsed = json.loads(detail) msg = parsed.get("error") if isinstance(msg, dict): msg = msg.get("message") detail = msg or detail except json.JSONDecodeError: pass detail = str(detail)[:600] if status in (401, 403): return (f"The provider rejected the credential ({status}). " f"Check the token has inference permission. {detail}") if status == 402: return f"Payment required ({status}): the account has no inference credit. {detail}" if status == 404: return f"No provider is serving this model right now ({status}). {detail}" if status == 429: return f"Rate limited ({status}). {detail}" return f"Upstream error {status}: {detail}" def _error_stream(message: str) -> StreamingResponse: """Report a rejected request through the same channel as a live one. A 4xx here would make the client show a transport failure for what is really the model's turn failing, and the message would never reach the transcript where the reader is looking. """ async def gen(): yield sse("error", message) return StreamingResponse(gen(), media_type="text/event-stream", headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"}) # -- static frontend -------------------------------------------------------- # Registered last, so a declared route always wins. Absent in a bare checkout: # the bundle is built by CI, and the API is still testable without it. if STATIC.is_dir(): app.mount("/assets", StaticFiles(directory=STATIC / "assets"), name="assets") # The bundle's filename carries a content hash, so `index.html` is the only # thing that has to be re-fetched for a deploy to be visible. Cached, it # keeps pointing at the previous bundle for as long as the browser feels # like it -- which reads as "the deploy did nothing", and cost an afternoon # of testing against a build that was no longer there. NO_CACHE = {"Cache-Control": "no-cache, must-revalidate"} @app.get("/") async def index(): return FileResponse(STATIC / "index.html", headers=NO_CACHE) @app.get("/{path:path}") async def spa(path: str): # Being registered last is not enough: this pattern still matches an # /api path that no route above claimed -- a typo'd endpoint, or the # wrong method on a real one -- and answering it with the index page # turns a 404 into a 200 full of HTML, which surfaces in the client as # a JSON parse error a long way from the cause. if path.startswith("api/"): return JSONResponse({"error": "not found"}, status_code=404) target = STATIC / path if target.is_file(): return FileResponse(target) return FileResponse(STATIC / "index.html", headers=NO_CACHE)