Spaces:
Paused
Paused
Download server/main.py from Bit-Trading-Company/ASE-GLM: direct link, hf CLI and curl.
- Browser
- Download file 11.2 kB
-
https://huggingface.co/spaces/Bit-Trading-Company/ASE-GLM/resolve/main/server/main.py
- Command line
-
hf download hf://spaces/Bit-Trading-Company/ASE-GLM/server/main.py
-
curl -L -o main.py https://huggingface.co/spaces/Bit-Trading-Company/ASE-GLM/resolve/main/server/main.py
11.2 kB
| """ASE-GLM: a minimal Hugging Face code harness. | |
| FastAPI serves the built character-grid frontend and proxies chat completions | |
| to the Hugging Face inference router. There is no Gradio here on purpose: the | |
| whole UI is one canvas, so a framework that renders its own DOM would only be | |
| something to fight. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import os | |
| import secrets | |
| from pathlib import Path | |
| import httpx | |
| from fastapi import FastAPI, Header, Request | |
| from fastapi.responses import FileResponse, JSONResponse, StreamingResponse | |
| from fastapi.staticfiles import StaticFiles | |
| from starlette.middleware.sessions import SessionMiddleware | |
| from . import catalog, chats, credentials, insight, oauth, ratelimit | |
| from .stream import Accumulated, build_payload, sse, translate | |
| ROUTER_URL = "https://router.huggingface.co/v1/chat/completions" | |
| STATIC = Path(__file__).resolve().parent.parent / "static" | |
| BUILD = os.environ.get("ASE_BUILD", "dev") | |
| app = FastAPI(title="ASE-GLM", docs_url=None, redoc_url=None) | |
| # Signed cookie sessions hold the viewer's OAuth token, so it never reaches the | |
| # browser as a value a script could read. | |
| # | |
| # `same_site` is the subtle one. A Space is normally viewed embedded in an | |
| # iframe on huggingface.co, which makes this cookie third-party: under `lax` the | |
| # browser will not send it on the cross-site navigation back from the OAuth | |
| # provider, the session arrives empty, and the callback rejects a state it did | |
| # in fact issue. `none` is required for the embedded case, and `none` requires | |
| # `Secure`, which a Space has and local http does not -- hence the split. | |
| # | |
| # Falling back to a random key means a restart signs everyone out, and a | |
| # sign-in that spans a restart fails the same way. Set ASE_SESSION_SECRET. | |
| app.add_middleware( | |
| SessionMiddleware, | |
| secret_key=os.environ.get("ASE_SESSION_SECRET") or secrets.token_urlsafe(32), | |
| session_cookie="ase_session", | |
| https_only=credentials.on_space(), | |
| same_site="none" if credentials.on_space() else "lax", | |
| max_age=60 * 60 * 8, | |
| ) | |
| app.include_router(oauth.router) | |
| # One per process. See `ratelimit.py`: this guards the Space's own CPU, not | |
| # anybody's credits, and is off entirely when ASE_RATE_LIMIT is 0. | |
| _limiter = ratelimit.RateLimiter() | |
| def _owner(request: Request) -> str | None: | |
| """Whose chats these are. | |
| Only a signed-in viewer has a shared store: a pasted key identifies nobody, | |
| and the Space's own token identifies the owner rather than the reader. | |
| """ | |
| sess = request.scope.get("session") or {} | |
| return sess.get("oauth_user") | |
| def _credential(request: Request, user_key: str | None) -> credentials.Credential: | |
| """Resolve the credential for this request. | |
| OAuth lands here in Phase 3: a Space with ``hf_oauth`` gives the signed-in | |
| viewer's token, which is what makes inference bill them rather than us. | |
| Until then the session carries no OAuth token and the ladder falls through. | |
| """ | |
| # `request.session` is a property that asserts rather than raising | |
| # AttributeError, so hasattr() triggers the very failure it would guard. | |
| sess = request.scope.get("session") or {} | |
| return credentials.resolve( | |
| oauth_token=sess.get("oauth_token"), | |
| oauth_user=sess.get("oauth_user"), | |
| user_key=user_key, | |
| ) | |
| async def config(request: Request, x_ase_key: str | None = Header(default=None)): | |
| cred = _credential(request, x_ase_key) | |
| sess = request.scope.get("session") or {} | |
| return { | |
| "models": [m.to_json() for m in catalog.MODELS], | |
| "defaultModel": catalog.DEFAULT_MODEL, | |
| "auth": { | |
| **cred.to_json(), | |
| # Whether sign-in is even offered: off a Space there is no OAuth | |
| # app, so the button would only ever produce an error. | |
| "canSignIn": oauth.configured(), | |
| "profile": sess.get("oauth_profile"), | |
| }, | |
| "spaceId": os.environ.get("SPACE_ID"), | |
| "build": BUILD, | |
| "chatStore": chats.state(_owner(request)).to_json(), | |
| "rateLimit": { | |
| "enabled": ratelimit.enabled(), | |
| "perWindow": ratelimit.limit(), | |
| "windowSeconds": int(ratelimit.window_s()), | |
| }, | |
| } | |
| async def model(id: str, request: Request, x_ase_key: str | None = Header(default=None)): | |
| if not catalog.is_allowed(id): | |
| return JSONResponse({"error": "unknown model"}, status_code=400) | |
| cred = _credential(request, x_ase_key) | |
| return await insight.model_info(id, cred.token) | |
| async def list_chats(request: Request): | |
| return chats.load_all(_owner(request) or "") | |
| async def put_chat(chat_id: str, request: Request): | |
| body = await request.json() | |
| if not isinstance(body, dict): | |
| return JSONResponse({"error": "expected a chat object"}, status_code=400) | |
| return chats.save(_owner(request) or "", chat_id, body).to_json() | |
| async def delete_chat(chat_id: str, request: Request): | |
| return chats.delete(_owner(request) or "", chat_id).to_json() | |
| async def health(): | |
| return {"ok": True, "build": BUILD, "onSpace": credentials.on_space()} | |
| async def chat(request: Request, x_ase_key: str | None = Header(default=None)): | |
| body = await request.json() | |
| model_id = body.get("model") or catalog.DEFAULT_MODEL | |
| # The model id is interpolated into an upstream request. Allow-list it | |
| # rather than sanitising: the set of models we offer is small and known. | |
| if not catalog.is_allowed(model_id): | |
| return _error_stream(f"Unknown model: {model_id!r}") | |
| messages = body.get("messages") | |
| if not isinstance(messages, list) or not messages: | |
| return _error_stream("No messages in the request.") | |
| cred = _credential(request, x_ase_key) | |
| if not cred.token: | |
| return _error_stream(cred.detail) | |
| # Checked after the credential, so somebody who cannot make a request at all | |
| # is told that rather than being told to slow down. | |
| verdict = _limiter.check( | |
| ratelimit.caller_key(_owner(request), request.client.host if request.client else None)) | |
| if not verdict.ok: | |
| return _error_stream(verdict.detail) | |
| spec = catalog.get(model_id) | |
| payload = build_payload(body, model_id, bool(spec and spec.reasoning)) | |
| async def gen(): | |
| acc = Accumulated() | |
| headers = { | |
| "Authorization": f"Bearer {cred.token}", | |
| "Content-Type": "application/json", | |
| } | |
| try: | |
| # No total timeout: a max-effort reasoning request legitimately | |
| # thinks for minutes. The read timeout still catches a dead route. | |
| timeout = httpx.Timeout(connect=20.0, read=180.0, write=20.0, pool=20.0) | |
| async with httpx.AsyncClient(timeout=timeout) as client: | |
| async with client.stream("POST", ROUTER_URL, headers=headers, | |
| json=payload) as upstream: | |
| if upstream.status_code >= 400: | |
| detail = (await upstream.aread()).decode("utf-8", "replace") | |
| yield sse("error", _clean_error(upstream.status_code, detail)) | |
| return | |
| async for line in upstream.aiter_lines(): | |
| for frame in translate(line, acc): | |
| yield frame | |
| if not acc.finished: | |
| yield sse("stats", json.dumps(acc.stats)) | |
| except httpx.HTTPError as e: | |
| yield sse("error", f"Upstream connection failed: {type(e).__name__}: {e}") | |
| return StreamingResponse( | |
| gen(), | |
| media_type="text/event-stream", | |
| headers={ | |
| "Cache-Control": "no-cache, no-transform", | |
| "Connection": "keep-alive", | |
| # Spaces sit behind a reverse proxy that will otherwise buffer the | |
| # whole response and deliver it as one block, which looks exactly | |
| # like a model that streams nothing and then answers all at once. | |
| "X-Accel-Buffering": "no", | |
| }, | |
| ) | |
| def _clean_error(status: int, detail: str) -> str: | |
| """Upstream errors, minus the noise, and never echoing a credential.""" | |
| try: | |
| parsed = json.loads(detail) | |
| msg = parsed.get("error") | |
| if isinstance(msg, dict): | |
| msg = msg.get("message") | |
| detail = msg or detail | |
| except json.JSONDecodeError: | |
| pass | |
| detail = str(detail)[:600] | |
| if status in (401, 403): | |
| return (f"The provider rejected the credential ({status}). " | |
| f"Check the token has inference permission. {detail}") | |
| if status == 402: | |
| return f"Payment required ({status}): the account has no inference credit. {detail}" | |
| if status == 404: | |
| return f"No provider is serving this model right now ({status}). {detail}" | |
| if status == 429: | |
| return f"Rate limited ({status}). {detail}" | |
| return f"Upstream error {status}: {detail}" | |
| def _error_stream(message: str) -> StreamingResponse: | |
| """Report a rejected request through the same channel as a live one. | |
| A 4xx here would make the client show a transport failure for what is | |
| really the model's turn failing, and the message would never reach the | |
| transcript where the reader is looking. | |
| """ | |
| async def gen(): | |
| yield sse("error", message) | |
| return StreamingResponse(gen(), media_type="text/event-stream", | |
| headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"}) | |
| # -- static frontend -------------------------------------------------------- | |
| # Registered last, so a declared route always wins. Absent in a bare checkout: | |
| # the bundle is built by CI, and the API is still testable without it. | |
| if STATIC.is_dir(): | |
| app.mount("/assets", StaticFiles(directory=STATIC / "assets"), name="assets") | |
| # The bundle's filename carries a content hash, so `index.html` is the only | |
| # thing that has to be re-fetched for a deploy to be visible. Cached, it | |
| # keeps pointing at the previous bundle for as long as the browser feels | |
| # like it -- which reads as "the deploy did nothing", and cost an afternoon | |
| # of testing against a build that was no longer there. | |
| NO_CACHE = {"Cache-Control": "no-cache, must-revalidate"} | |
| async def index(): | |
| return FileResponse(STATIC / "index.html", headers=NO_CACHE) | |
| async def spa(path: str): | |
| # Being registered last is not enough: this pattern still matches an | |
| # /api path that no route above claimed -- a typo'd endpoint, or the | |
| # wrong method on a real one -- and answering it with the index page | |
| # turns a 404 into a 200 full of HTML, which surfaces in the client as | |
| # a JSON parse error a long way from the cause. | |
| if path.startswith("api/"): | |
| return JSONResponse({"error": "not found"}, status_code=404) | |
| target = STATIC / path | |
| if target.is_file(): | |
| return FileResponse(target) | |
| return FileResponse(STATIC / "index.html", headers=NO_CACHE) | |