Spaces:
Running
Running
File size: 9,611 Bytes
69e310f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 | """Runtime configuration.
Every secret and tunable is read from the environment — nothing is hard-coded and
nothing is committed. See `.env.example` at the repository root for the full list.
"""
from __future__ import annotations
import secrets
from functools import lru_cache
from pathlib import Path
from typing import Literal
from pydantic import field_validator
from pydantic_settings import BaseSettings, SettingsConfigDict
DEFAULT_WATCHLIST: tuple[str, ...] = ("AAPL", "MSFT", "NVDA", "TSLA", "AMZN")
#: Path of the shared local-development approval token, relative to the checkout
#: root. Mirrored by `apps/web/lib/server.ts`; change both together.
LOCAL_TOKEN_RELPATH = ("var", "approval_token")
#: Model routing by task weight (spec §3 "Models"). Haiku plans, Sonnet works.
SUPERVISOR_MODEL = "claude-haiku-4-5"
WORKER_MODEL = "claude-sonnet-4-6"
#: USD per 1M tokens, (input, output). Source: Anthropic published pricing.
MODEL_PRICING: dict[str, tuple[float, float]] = {
"claude-haiku-4-5": (1.00, 5.00),
"claude-sonnet-4-6": (3.00, 15.00),
"claude-sonnet-5": (3.00, 15.00),
"claude-opus-4-8": (5.00, 25.00),
"claude-opus-5": (5.00, 25.00),
}
EngineName = Literal["auto", "anthropic", "deterministic"]
McpTransport = Literal["stdio", "inmemory"]
class Settings(BaseSettings):
"""Immutable, env-driven application settings."""
model_config = SettingsConfigDict(
env_file=(".env", "../../.env"),
env_file_encoding="utf-8",
extra="ignore",
case_sensitive=False,
# `model_` is a Pydantic-protected namespace; our model_* fields are
# deliberate configuration, so the protection is disabled explicitly.
protected_namespaces=(),
)
# ---------------------------------------------------------------- app ---
environment: Literal["local", "production"] = "local"
log_level: Literal["DEBUG", "INFO", "WARNING", "ERROR"] = "INFO"
api_title: str = "AlphaBrief API"
# -------------------------------------------------------------- claude ---
anthropic_api_key: str | None = None
model_supervisor: str = SUPERVISOR_MODEL
model_worker: str = WORKER_MODEL
model_writer: str = WORKER_MODEL
#: "auto" uses Anthropic when a key is present, else the deterministic engine.
llm_engine: EngineName = "auto"
llm_max_tokens: int = 4096
llm_timeout_seconds: float = 120.0
llm_max_retries: int = 2
#: Effort hint for Sonnet workers. Haiku 4.5 does not accept `effort`.
worker_effort: Literal["low", "medium", "high"] = "medium"
# ------------------------------------------------------------ guardrails ---
max_iterations: int = 15
token_budget_usd: float = 0.50
token_budget_tokens: int = 400_000
max_watchlist_size: int = 12
max_regenerations: int = 1
# ------------------------------------------------------------------ mcp ---
mcp_transport: McpTransport = "stdio"
#: Share one in-memory MCP server (and therefore one provider cache) across
#: runs. Used by the eval harness so 20 runs are polite to free providers.
mcp_shared_server: bool = False
mcp_startup_timeout_seconds: float = 30.0
mcp_call_timeout_seconds: float = 60.0
price_history_days: int = 120
news_limit: int = 6
#: Polite delay between outbound provider calls, per MCP server process.
provider_min_interval_seconds: float = 0.20
#: Attempts per provider call, including the first. Yahoo answers the first
#: call from a cold session on a shared egress IP with a rate-limit error and
#: then serves everything after it, so one attempt loses the first ticker of
#: every run on a hosted deployment.
provider_max_attempts: int = 3
provider_retry_backoff_seconds: float = 0.75
# ------------------------------------------------------------ persistence ---
database_url: str = "sqlite:///./alphabrief.db"
db_echo: bool = False
# ---------------------------------------------------------- observability ---
langfuse_public_key: str | None = None
langfuse_secret_key: str | None = None
langfuse_host: str = "https://cloud.langfuse.com"
# ------------------------------------------------------------------ smtp ---
smtp_host: str | None = None
smtp_port: int = 587
smtp_username: str | None = None
smtp_password: str | None = None
smtp_from: str | None = None
smtp_to: str | None = None
smtp_timeout_seconds: float = 20.0
# ------------------------------------------------------------- security ---
#: Bearer token required by the approval endpoints. Auto-generated when unset
#: so that a deployment can never accidentally ship a well-known default.
approval_token: str | None = None
#: The console's own origin. Never "*": these endpoints take a bearer token.
cors_allow_origins: str = "http://localhost:3001,http://127.0.0.1:3001"
#: Hard ceiling on any single inbound request body (bytes).
max_request_bytes: int = 64 * 1024
# ---------------------------------------------------------------- runs ---
default_watchlist: str = ",".join(DEFAULT_WATCHLIST)
max_events_per_run: int = 4000
run_timeout_seconds: float = 900.0
@field_validator("cors_allow_origins", "default_watchlist")
@classmethod
def _strip(cls, value: str) -> str:
return value.strip()
@property
def watchlist(self) -> list[str]:
return [t.strip().upper() for t in self.default_watchlist.split(",") if t.strip()]
@property
def cors_origins(self) -> list[str]:
return [o.strip() for o in self.cors_allow_origins.split(",") if o.strip()]
@property
def langfuse_enabled(self) -> bool:
return bool(self.langfuse_public_key and self.langfuse_secret_key)
@property
def smtp_enabled(self) -> bool:
return bool(self.smtp_host and self.smtp_from and self.smtp_to)
@property
def resolved_engine(self) -> Literal["anthropic", "deterministic"]:
"""Which LLM engine this process will actually use."""
if self.llm_engine == "anthropic":
return "anthropic"
if self.llm_engine == "deterministic":
return "deterministic"
return "anthropic" if self.anthropic_api_key else "deterministic"
def require_approval_token(self) -> str:
"""Return the approval bearer token, minting one on first use.
There is deliberately no default value: a committed constant would be a
well-known credential on every deployment that forgot to configure one.
Local development still needs *two* processes — this API and the Next
console — to agree on a token nobody configured, or the approve button
is dead on a fresh clone. So on `local` the token is minted once into a
gitignored per-checkout file that both read. `production` never gets
that handshake: there an unset token stays random per process, which
fails closed rather than trusting a file an attacker might plant.
"""
token = self.approval_token
if not token and self.environment == "local":
token = read_or_mint_local_token()
if not token:
token = secrets.token_urlsafe(32)
self.approval_token = token
return token
def checkout_root() -> Path | None:
"""The repository root, or ``None`` when running from a container image.
Identified by the two application directories rather than `.git`, so it is
also correct for a source tarball. A container copies only `apps/api`, so
this returns ``None`` there — which is the intended answer, since the token
handshake is a local-development affair.
"""
for parent in Path(__file__).resolve().parents:
if (parent / "apps" / "api").is_dir() and (parent / "apps" / "web").is_dir():
return parent
return None
def read_or_mint_local_token() -> str | None:
"""Read the shared local approval token, creating it if this is the first run.
Created with ``exist_ok=False`` so that two processes racing on first boot
cannot each mint a different token and then disagree about every approval:
the loser of the race falls through to reading what the winner wrote.
Returns ``None`` if there is nowhere to write, leaving the caller to fall
back to a per-process token.
"""
root = checkout_root()
if root is None:
return None
path = root.joinpath(*LOCAL_TOKEN_RELPATH)
try:
existing = path.read_text(encoding="utf-8").strip()
if existing:
return existing
except OSError:
pass
token = secrets.token_urlsafe(32)
try:
path.parent.mkdir(parents=True, exist_ok=True)
path.touch(mode=0o600, exist_ok=False)
path.write_text(f"{token}\n", encoding="utf-8")
except FileExistsError:
try:
return path.read_text(encoding="utf-8").strip() or None
except OSError:
return None
except OSError:
return None
return token
@lru_cache(maxsize=1)
def get_settings() -> Settings:
"""Process-wide settings singleton."""
return Settings()
def price_of(model: str) -> tuple[float, float]:
"""USD per 1M (input, output) tokens for `model`.
Unknown models fall back to the most expensive known tier so that a
mis-configured model can never silently under-report spend to the budget guard.
"""
if model in MODEL_PRICING:
return MODEL_PRICING[model]
return max(MODEL_PRICING.values(), key=lambda p: p[1])
|