File size: 6,672 Bytes
5828b5b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 | """Static configuration for SpriteBench.
Holds repo paths, OpenRouter / budget constants, scoring weights, the judge
model, the verified candidate roster, the sampling plan, and the mock model
identifiers. config.py is the single source of truth for the candidate list;
discover_models.py only verifies against it.
"""
from __future__ import annotations
import os
from dataclasses import dataclass
from pathlib import Path
# --- Paths -----------------------------------------------------------------
ROOT: Path = Path(__file__).resolve().parent
RUNS: Path = ROOT / "runs"
RESULTS: Path = ROOT / "results"
LOGS: Path = ROOT / "logs"
# --- OpenRouter / budget ---------------------------------------------------
OPENROUTER_BASE = "https://openrouter.ai/api/v1"
BUDGET_USD = 25.0 # hard pause threshold; override via env SPRITEBENCH_BUDGET
MAX_PALETTE = 12
GEN_MAX_TOKENS = 8192
# Hidden thinking is disabled for ALL candidate generation calls (fairness:
# same setting every model, every round). Smoke testing showed hybrid
# reasoners (qwen3.6-27b) burn the entire max_tokens budget on reasoning and
# return empty text; the target app is interactive, so thinking-off is also
# the production configuration. None = provider default (judge keeps None).
CANDIDATE_REASONING: bool | None = False
# --- Judge -----------------------------------------------------------------
JUDGE_MODEL = "google/gemini-3.1-pro-preview" # frontier vision, NOT a candidate
JUDGE_TEMP = 0.0
# Mean abs per-dim delta on duplicate judge pairs; above this -> run a 2nd
# judge pass and average.
CONSISTENCY_THRESHOLD = 1.5
# --- Scoring weights -------------------------------------------------------
WEIGHTS = {"validity": 0.30, "visual": 0.50, "repair": 0.10, "costlat": 0.10}
@dataclass(frozen=True)
class Candidate:
"""A model under test."""
id: str # exact OpenRouter id, or a local alias like "local/gemma-4-12b"
params_b: float # total params, billions (must be <= 32)
vision: bool # can accept images (tier 3 eligible)
enabled: bool
notes: str = ""
# Custom OpenAI-compatible endpoint. None -> OpenRouter. When set, the
# client sends requests there (no real API key, cost recorded as $0, one
# request at a time) and omits OpenRouter-only payload fields.
base_url: str | None = None
api_model: str | None = None # model name for the payload; defaults to id
# Verified against live OpenRouter API 2026-06-09. All three confirmed
# available, modality "text+image+video->text", ctx 262144:
#
# | id | params_b | vision | pricing in/out per M | notes |
# |---------------------------------|----------|--------|----------------------|----------------------------------------|
# | qwen/qwen3.6-27b | 27 | True | $0.289 / $2.40 | dense, prime suspect |
# | google/gemma-4-31b-it:free | 31 | True | $0 / $0 | free tier - expect 429s, backoff handles |
# | google/gemma-4-26b-a4b-it:free | 26 | True | $0 / $0 | MoE A4B, free tier |
#
# These three ONLY -- per user instruction (2026-06-09) no other candidates
# are added, even ones named in the mission.
#
# MiniCPM: NOT available on OpenRouter at all (verified live 2026-06-09). The
# mission's sponsor-prize MiniCPM candidate could therefore not be included;
# the final report must note this substitution/exclusion. The roster above is
# user-locked: do not add Nemotron, Mistral Small, tiny <=4B models, or any
# other candidate named in the mission -- the user explicitly froze the list
# to these three on 2026-06-09.
CANDIDATES: list[Candidate] = [
Candidate(
id="qwen/qwen3.6-27b",
params_b=27,
vision=True,
enabled=True,
notes="dense, prime suspect",
),
Candidate(
id="google/gemma-4-31b-it:free",
params_b=31,
vision=True,
enabled=True,
notes="free tier - expect 429s, backoff handles",
),
Candidate(
id="google/gemma-4-26b-a4b-it:free",
params_b=26,
vision=True,
enabled=True,
notes="MoE A4B, free tier",
),
# User's own local server (added 2026-06-09 at user request): llama.cpp-
# style OpenAI-compatible endpoint, 64K ctx, ~49 tok/s, all layers on GPU.
# Text-only as served (no images) -> vision=False, sits out Tier-3 repair.
# The model emits hidden reasoning that cannot be disabled via the API;
# the answer arrives in message.content so parsing is unaffected, and the
# latency metric honestly reflects the thinking time.
Candidate(
id="local/gemma-4-12b",
params_b=12,
vision=False,
enabled=True,
notes="user's local server; reasoning baked in; cost $0",
base_url="http://localhost:8080/v1",
api_model="gemma-4-12b",
),
]
# Sampling plan: 3 samples at temp 0.7, 1 at temp 0.2; fixed seeds for
# reproducibility where the API honors them.
SAMPLES: list[dict] = [
{"sample": 0, "temp": 0.7, "seed": 11},
{"sample": 1, "temp": 0.7, "seed": 22},
{"sample": 2, "temp": 0.7, "seed": 33},
{"sample": 3, "temp": 0.2, "seed": 44},
]
# Mock model identifiers used by MockClient for offline testing; all vision=True.
MOCK_MODELS = ["mock/good-model", "mock/flaky-model", "mock/bad-model"]
def enabled_candidates(mock: bool = False) -> list[Candidate]:
"""Return the candidate roster to run.
When mock=True, returns the three MOCK_MODELS wrapped as Candidate objects
(params_b=1, vision=True, enabled=True). Otherwise returns the enabled
real candidates from CANDIDATES.
"""
if mock:
return [
Candidate(id=model_id, params_b=1, vision=True, enabled=True)
for model_id in MOCK_MODELS
]
return [c for c in CANDIDATES if c.enabled]
def endpoint_for(model_id: str) -> tuple[str, str, bool]:
"""Resolve (base_url, payload_model_name, is_openrouter) for a model id.
Candidates with a custom base_url route there; everything else (including
the judge model) goes to OpenRouter.
"""
for c in CANDIDATES:
if c.id == model_id and c.base_url:
return c.base_url, c.api_model or c.id, False
return OPENROUTER_BASE, model_id, True
def budget_usd() -> float:
"""Effective budget: env SPRITEBENCH_BUDGET if set and numeric, else BUDGET_USD."""
raw = os.environ.get("SPRITEBENCH_BUDGET")
if raw is not None:
try:
return float(raw)
except ValueError:
pass
return BUDGET_USD
|