contimp-app / tests /test_backends.py
lefft's picture
Per-task LLM backends (OpenAI + Anthropic wire); config-copilot pinned to the oumi backend
1bcb9d8 verified
Raw
History Blame Contribute Delete
5.11 kB
"""Per-task backend profile resolution and provider dispatch."""
import importlib
import pytest
@pytest.fixture()
def config(monkeypatch):
# Clear any default LLM_* so the default backend is deterministic.
for k in ("LLM_PROVIDER", "LLM_BASE_URL", "LLM_MODEL", "LLM_API_KEY", "ANTHROPIC_API_KEY"):
monkeypatch.delenv(k, raising=False)
import app.config as config
return importlib.reload(config)
def test_default_backend(config):
be = config.backend(None)
assert be.provider == "openai"
assert be.base_url == "https://api.anthropic.com/v1/"
assert be.model == "claude-haiku-4-5"
assert config.backend("") is config.DEFAULT_BACKEND
def test_named_backend_resolves(config, monkeypatch):
monkeypatch.setenv("LLM_OUMI_PROVIDER", "anthropic")
monkeypatch.setenv("LLM_OUMI_BASE_URL", "https://api.oumi.ai/inference")
monkeypatch.setenv("LLM_OUMI_MODEL", "projects/X/deployments/22")
monkeypatch.setenv("LLM_OUMI_API_KEY", "oumi_test")
be = config.backend("oumi")
assert be.provider == "anthropic"
assert be.base_url == "https://api.oumi.ai/inference"
assert be.model == "projects/X/deployments/22"
assert be.api_key == "oumi_test"
assert be.name == "oumi"
def test_named_provider_defaults_to_openai(config, monkeypatch):
monkeypatch.setenv("LLM_QWEN_BASE_URL", "https://api.oumi.ai/inference/v1/")
monkeypatch.setenv("LLM_QWEN_MODEL", "projects/X/deployments/99")
monkeypatch.setenv("LLM_QWEN_API_KEY", "oumi_test")
assert config.backend("qwen").provider == "openai"
def test_unconfigured_named_backend_falls_back_to_default(config, monkeypatch):
monkeypatch.setenv("LLM_OUMI_BASE_URL", "https://api.oumi.ai/inference")
# MODEL and API_KEY missing -> not fully configured -> default
assert config.backend("oumi") is config.DEFAULT_BACKEND
# ...and the trace says "default", not "oumi", so the fallback isn't silent.
assert config.backend("oumi").name == "default"
def test_task_backend_defaults_none():
from app.tasks.base import Task
t = Task(
id="x", title="X", tagline="", system_prompt="", ui={},
sample=lambda: None, lookup_truth=lambda i: None,
parse_output=lambda s: s, score=lambda tr, p: {}, present=lambda p, tr: {},
)
assert t.backend is None
def test_runner_dispatch_by_provider():
from app import engine
from app.config import Backend
oa = Backend("openai", "u", "m", "k")
an = Backend("anthropic", "u", "m", "k")
assert engine._runner_for(oa) is engine._run_openai
assert engine._runner_for(an) is engine._run_anthropic
class _Block:
"""Stands in for an anthropic content block."""
def __init__(self, **kw):
self.__dict__.update(kw)
def model_dump(self):
return dict(self.__dict__)
def _resp(stop_reason, content):
return _Block(
stop_reason=stop_reason,
content=content,
usage=_Block(input_tokens=1, output_tokens=2),
)
def test_anthropic_run_emits_openai_format_trajectory(monkeypatch):
"""The anthropic path must return the same standard chat record as the
openai path — it's the training-data artifact the UI downloads."""
monkeypatch.setenv("LANGFUSE_TRACING_ENABLED", "false")
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-lf-test")
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-lf-test")
from app import engine
from app.config import Backend
from app.tasks.base import Task
task = Task(
id="t", title="T", tagline="", system_prompt="SYS", ui={},
sample=lambda: None, lookup_truth=lambda i: None,
parse_output=lambda s: s, score=lambda tr, p: {}, present=lambda p, tr: {},
tools=[{
"type": "function",
"function": {
"name": "lookup",
"description": "d",
"parameters": {"type": "object", "properties": {"q": {"type": "string"}}},
},
}],
execute_tool=lambda name, args: "TOOL_RESULT",
)
turns = [
_resp("tool_use", [_Block(type="tool_use", id="tu_1", name="lookup", input={"q": "x"})]),
_resp("end_turn", [_Block(type="text", text="DONE")]),
]
client = _Block(messages=_Block(create=lambda **kw: turns.pop(0)))
monkeypatch.setattr(engine, "_anthropic", lambda be: client)
final, transcript, messages = engine._run_anthropic(
task, "hello", Backend("anthropic", "u", "m", "k")
)
assert final == "DONE"
assert transcript == [{"tool": "lookup", "args": {"q": "x"}, "result": "TOOL_RESULT"}]
assert [m["role"] for m in messages] == ["system", "user", "assistant", "tool", "assistant"]
assert messages[0]["content"] == "SYS" and messages[1]["content"] == "hello"
assert messages[2]["tool_calls"] == [{
"id": "tu_1",
"type": "function",
"function": {"name": "lookup", "arguments": '{"q": "x"}'},
}]
assert messages[3] == {"role": "tool", "tool_call_id": "tu_1", "content": "TOOL_RESULT"}
assert messages[-1] == {"role": "assistant", "content": "DONE"}