| """OpenAI-compatible proxy that forces terminus-2's action schema via constrained decoding. |
| |
| Why: mining the benchmark's own parse labels showed only 7.9% of soyuz-4B turns produce a |
| valid terminus action; 92% are malformed, so most turns are wasted before the task is even |
| attempted. Additive steering did not fix this (p=0.60). Constrained decoding does: |
| 0% -> 86% clean on 100 replayed bench prefixes (Fisher p=1.35e-11). |
| |
| terminus-2 does not send `response_format`, so this proxy injects it on the way to vLLM. |
| Run it on PROXY_PORT and point harbor's api_base at it instead of vLLM directly. |
| |
| Env: |
| GUIDED_UPSTREAM (default http://localhost:30007/v1) |
| GUIDED_PORT (default 30010) |
| GUIDED_ENABLE (1/0, default 1 — set 0 for an A/B control arm through the same proxy) |
| """ |
| from __future__ import annotations |
| import json, os |
|
|
| import httpx |
| from fastapi import FastAPI, Request |
| from fastapi.responses import JSONResponse |
| import uvicorn |
|
|
| UPSTREAM = os.environ.get("GUIDED_UPSTREAM", "http://localhost:30007/v1") |
| PORT = int(os.environ.get("GUIDED_PORT", "30010")) |
| ENABLE = os.environ.get("GUIDED_ENABLE", "1") == "1" |
|
|
| TERMINUS_SCHEMA = { |
| "type": "object", |
| "properties": { |
| "analysis": {"type": "string"}, |
| "plan": {"type": "string"}, |
| "commands": { |
| "type": "array", |
| "items": { |
| "type": "object", |
| "properties": { |
| "keystrokes": {"type": "string"}, |
| "duration": {"type": "number"}, |
| }, |
| "required": ["keystrokes", "duration"], |
| }, |
| }, |
| "task_complete": {"type": "boolean"}, |
| }, |
| "required": ["analysis", "plan", "commands"], |
| } |
|
|
| app = FastAPI() |
| client = httpx.AsyncClient(timeout=1800.0) |
|
|
|
|
| @app.get("/v1/models") |
| async def models(): |
| r = await client.get(f"{UPSTREAM}/models") |
| return JSONResponse(r.json()) |
|
|
|
|
| @app.post("/v1/chat/completions") |
| async def chat(req: Request): |
| body = await req.json() |
| if ENABLE and "response_format" not in body: |
| body["response_format"] = { |
| "type": "json_schema", |
| "json_schema": {"name": "terminus_action", "schema": TERMINUS_SCHEMA}, |
| } |
| |
| body["max_tokens"] = max(int(body.get("max_tokens") or 0), 1500) |
| r = await client.post(f"{UPSTREAM}/chat/completions", json=body) |
| return JSONResponse(r.json(), status_code=r.status_code) |
|
|
|
|
| if __name__ == "__main__": |
| print(f"[guided-proxy] :{PORT} -> {UPSTREAM} enforce_schema={ENABLE}", flush=True) |
| uvicorn.run(app, host="0.0.0.0", port=PORT, log_level="warning") |
|
|