Spaces:
Sleeping
Sleeping
| """Unified multi-provider LLM layer (requests-based — no provider SDKs). | |
| PROVIDERS drives the selection cards; every provider goes through one | |
| llm_call() signature. Model lists verified current as of July 2026; provider | |
| model catalogs change often, so each card links its console for the live list. | |
| Vision support varies per model and is declared per provider. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import requests | |
| TIMEOUT = 300 | |
| PROVIDERS = { | |
| "Claude (Anthropic)": { | |
| "icon": "🟠", "note": "Best plan-reading vision", "free_tier": False, | |
| # Anthropic API model strings (July 2026). See console for the live list. | |
| "models": ["claude-opus-4-8", "claude-sonnet-5", "claude-haiku-4-5-20251001", | |
| "claude-sonnet-4-6"], | |
| "vision_models": "all", | |
| "key_url": "https://console.anthropic.com/settings/keys", | |
| "key_hint": "sk-ant-...", | |
| "env": "ANTHROPIC_API_KEY", | |
| "kind": "anthropic", | |
| }, | |
| "ChatGPT (OpenAI)": { | |
| "icon": "🟢", "note": "GPT-5 family + legacy 4o", "free_tier": False, | |
| # GPT-4o retired from ChatGPT Feb 2026 but still callable via API for now; | |
| # kept as legacy. GPT-5 family is the current generation. | |
| "models": ["gpt-5.4", "gpt-5.4-mini", "gpt-5.3", "gpt-4.1", "gpt-4.1-mini", | |
| "gpt-4o", "gpt-4o-mini"], | |
| "vision_models": "all", | |
| "key_url": "https://platform.openai.com/api-keys", | |
| "key_hint": "sk-...", | |
| "env": "OPENAI_API_KEY", | |
| "kind": "openai", | |
| }, | |
| "Gemini (Google)": { | |
| "icon": "🔵", "note": "Free tier · Gemini 3.x", "free_tier": True, | |
| # 2.5 line superseded; 3.x is current GA (July 2026). | |
| "models": ["gemini-3.6-flash", "gemini-flash-latest", "gemini-3.5-flash-lite", | |
| "gemini-3.1-pro-preview"], | |
| "vision_models": "all", | |
| "key_url": "https://aistudio.google.com/apikey", | |
| "key_hint": "AIza...", | |
| "env": "GEMINI_API_KEY", | |
| "kind": "gemini", | |
| }, | |
| "Groq (fast)": { | |
| "icon": "🟣", "note": "Free · fast · text-only", "free_tier": True, | |
| # Llama 3.3 / Llama 4 chat models decommissioned 2026; current = gpt-oss + qwen. | |
| # These are TEXT-ONLY on Groq — review falls back to embedded page text. | |
| "models": ["openai/gpt-oss-120b", "openai/gpt-oss-20b", "qwen/qwen3.6-27b"], | |
| "vision_models": [], | |
| "key_url": "https://console.groq.com/keys", | |
| "key_hint": "gsk_...", | |
| "env": "GROQ_API_KEY", | |
| "kind": "openai_compat", | |
| "base_url": "https://api.groq.com/openai/v1", | |
| }, | |
| "Custom / Local (OpenAI-compatible)": { | |
| "icon": "⚙️", "note": "Ollama · vLLM · LM Studio · any OpenAI-compatible URL", | |
| "free_tier": True, | |
| "models": ["__custom__"], # user types the model id in the box below | |
| "vision_models": "all", # depends on the local model; user's call | |
| "key_url": "https://github.com/ollama/ollama/blob/main/docs/openai.md", | |
| "key_hint": "leave blank for local, or a token if your server needs one", | |
| "env": "CUSTOM_LLM_API_KEY", | |
| "kind": "openai_compat", | |
| "base_url": "", # user supplies, e.g. http://host:11434/v1 | |
| "custom": True, | |
| }, | |
| } | |
| def supports_vision(provider: str, model: str) -> bool: | |
| vm = PROVIDERS[provider]["vision_models"] | |
| return vm == "all" or model in vm | |
| def _raise_for_api(resp: requests.Response, provider: str): | |
| if resp.status_code >= 400: | |
| try: | |
| detail = json.dumps(resp.json())[:500] | |
| except Exception: # noqa: BLE001 | |
| detail = resp.text[:500] | |
| raise RuntimeError(f"{provider} API error {resp.status_code}: {detail}") | |
| def _uses_completion_tokens(model: str) -> bool: | |
| """GPT-5 family and o-series reasoning models renamed max_tokens -> | |
| max_completion_tokens and reject the old field. Detect by model id.""" | |
| m = model.lower() | |
| return (m.startswith("gpt-5") or m.startswith("o1") or m.startswith("o3") | |
| or m.startswith("o4") or m.startswith("gpt-6")) | |
| def _openai_compatible(url: str, api_key: str, model: str, system: str, user_text: str, | |
| images_b64: list[str], history: list[dict], max_tokens: int, | |
| provider: str) -> str: | |
| content = [{"type": "text", "text": user_text}] | |
| content += [{"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{b}"}} | |
| for b in images_b64] | |
| msgs = [{"role": "system", "content": system}] | |
| msgs += [{"role": h["role"], "content": h["content"]} for h in history] | |
| msgs.append({"role": "user", "content": content if images_b64 else user_text}) | |
| headers = {"Content-Type": "application/json"} | |
| if api_key: | |
| headers["Authorization"] = f"Bearer {api_key}" | |
| body = {"model": model, "messages": msgs} | |
| tok_field = "max_completion_tokens" if _uses_completion_tokens(model) else "max_tokens" | |
| body[tok_field] = max_tokens | |
| r = requests.post(f"{url.rstrip('/')}/chat/completions", headers=headers, | |
| json=body, timeout=TIMEOUT) | |
| # Fallback: if a provider rejects the token field name, retry with the other. | |
| if r.status_code == 400 and "max_tokens" in r.text and "max_completion_tokens" in r.text: | |
| body.pop(tok_field, None) | |
| alt = "max_tokens" if tok_field == "max_completion_tokens" else "max_completion_tokens" | |
| body[alt] = max_tokens | |
| r = requests.post(f"{url.rstrip('/')}/chat/completions", headers=headers, | |
| json=body, timeout=TIMEOUT) | |
| _raise_for_api(r, provider) | |
| return r.json()["choices"][0]["message"]["content"] or "" | |
| def llm_call(provider: str, model: str, api_key: str, system: str, | |
| user_text: str, images_b64: list[str] | None = None, | |
| history: list[dict] | None = None, max_tokens: int = 6000, | |
| base_url: str | None = None) -> str: | |
| """One call, any provider. base_url overrides the endpoint for custom/local.""" | |
| images_b64 = images_b64 or [] | |
| history = history or [] | |
| pinfo = PROVIDERS.get(provider, {}) | |
| kind = pinfo.get("kind", "openai") | |
| if kind == "anthropic": | |
| content: list[dict] = [{"type": "text", "text": user_text}] | |
| content += [{"type": "image", | |
| "source": {"type": "base64", "media_type": "image/jpeg", "data": b}} | |
| for b in images_b64] | |
| msgs = [{"role": h["role"], "content": h["content"]} for h in history] | |
| msgs.append({"role": "user", "content": content}) | |
| r = requests.post( | |
| "https://api.anthropic.com/v1/messages", | |
| headers={"x-api-key": api_key, "anthropic-version": "2023-06-01", | |
| "content-type": "application/json"}, | |
| json={"model": model, "max_tokens": max_tokens, "system": system, "messages": msgs}, | |
| timeout=TIMEOUT) | |
| _raise_for_api(r, provider) | |
| return "".join(b.get("text", "") for b in r.json()["content"] if b.get("type") == "text") | |
| if kind == "openai": | |
| return _openai_compatible("https://api.openai.com/v1", api_key, model, system, | |
| user_text, images_b64, history, max_tokens, provider) | |
| if kind == "openai_compat": | |
| url = base_url or pinfo.get("base_url") or "" | |
| if not url: | |
| raise RuntimeError(f"{provider}: no endpoint URL provided. Enter your " | |
| f"server's base URL (e.g. http://localhost:11434/v1).") | |
| return _openai_compatible(url, api_key, model, system, user_text, | |
| images_b64, history, max_tokens, provider) | |
| if kind == "gemini": | |
| parts: list[dict] = [{"text": user_text}] | |
| parts += [{"inline_data": {"mime_type": "image/jpeg", "data": b}} for b in images_b64] | |
| contents = [{"role": "user" if h["role"] == "user" else "model", | |
| "parts": [{"text": h["content"]}]} for h in history] | |
| contents.append({"role": "user", "parts": parts}) | |
| r = requests.post( | |
| f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent", | |
| params={"key": api_key}, | |
| json={"system_instruction": {"parts": [{"text": system}]}, | |
| "contents": contents, | |
| "generationConfig": {"maxOutputTokens": max_tokens}}, | |
| timeout=TIMEOUT) | |
| _raise_for_api(r, provider) | |
| cands = r.json().get("candidates", []) | |
| if not cands: | |
| raise RuntimeError(f"Gemini returned no candidates: {json.dumps(r.json())[:300]}") | |
| return "".join(p.get("text", "") for p in cands[0]["content"]["parts"]) | |
| raise ValueError(f"Unknown provider: {provider}") | |