Spaces:
Runtime error
Runtime error
| import { BaseProvider, providerHttpError } from './base.js'; | |
| /** | |
| * Generic provider for platforms that use an OpenAI-compatible API. | |
| * Covers: Groq, Cerebras, NVIDIA NIM, Mistral, OpenRouter, | |
| * GitHub Models, Fireworks AI. | |
| */ | |
| export class OpenAICompatProvider extends BaseProvider { | |
| platform; | |
| name; | |
| baseUrl; | |
| extraHeaders; | |
| validateUrl; | |
| /** Per-provider HTTP timeout override. Cloud APIs finish in ~15s; locally-hosted | |
| * inference (llama.cpp / vLLM on CPU) can take 30-120s for long prompts. Default 15000. */ | |
| timeoutMs; | |
| /** NVIDIA NIM models reject any request that permits parallel tool calls with | |
| * `400 This model only supports single tool-calls at once!`. When set, pin | |
| * parallel_tool_calls to false whenever tools are in play. See issue #255. */ | |
| forceSingleToolCall; | |
| constructor(opts) { | |
| super(); | |
| this.platform = opts.platform; | |
| this.name = opts.name; | |
| this.baseUrl = opts.baseUrl; | |
| this.extraHeaders = opts.extraHeaders ?? {}; | |
| this.validateUrl = opts.validateUrl; | |
| this.timeoutMs = opts.timeoutMs ?? 15000; | |
| this.keyless = opts.keyless ?? false; | |
| this.forceSingleToolCall = opts.forceSingleToolCall ?? false; | |
| } | |
| /** Resolve the parallel_tool_calls flag to send upstream. For providers that | |
| * only accept single tool calls (NVIDIA NIM), force `false` whenever tools are | |
| * present so the model never tries to emit two at once and 400s; otherwise pass | |
| * the caller's value through unchanged. See issue #255. */ | |
| resolveParallelToolCalls(options) { | |
| if (this.forceSingleToolCall && options?.tools && options.tools.length > 0) | |
| return false; | |
| return options?.parallel_tool_calls; | |
| } | |
| /** Keyless providers (Kilo's anonymous free tier) must send NO Authorization | |
| * header β a stored sentinel like `Bearer no-key` could be treated as an | |
| * invalid key. Everyone else sends the bearer as usual. */ | |
| authHeader(apiKey) { | |
| return this.keyless ? {} : { 'Authorization': `Bearer ${apiKey}` }; | |
| } | |
| async chatCompletion(apiKey, messages, modelId, options) { | |
| const res = await this.fetchWithTimeout(`${this.baseUrl}/chat/completions`, { | |
| method: 'POST', | |
| headers: { | |
| ...this.authHeader(apiKey), | |
| 'Content-Type': 'application/json', | |
| ...this.extraHeaders, | |
| }, | |
| body: JSON.stringify({ | |
| model: modelId, | |
| messages, | |
| temperature: options?.temperature, | |
| max_tokens: options?.max_tokens, | |
| top_p: options?.top_p, | |
| tools: options?.tools, | |
| tool_choice: options?.tool_choice, | |
| parallel_tool_calls: this.resolveParallelToolCalls(options), | |
| }), | |
| }, options?.timeoutMs ?? this.timeoutMs); | |
| if (!res.ok) { | |
| const err = await res.json().catch(() => ({})); | |
| throw providerHttpError(res, `${this.name} API error ${res.status}: ${err.error?.message ?? res.statusText}`); | |
| } | |
| let data; | |
| try { | |
| data = await res.json(); | |
| } | |
| catch { | |
| // A 200 whose body isn't a single JSON document β typically a base URL | |
| // pointing at a non-OpenAI-compatible API (e.g. Ollama's native NDJSON | |
| // /api endpoints instead of /v1, #189). Surface what's wrong instead of | |
| // the raw JSON.parse position error. | |
| throw new Error(`${this.name} returned 200 with a non-JSON body β the endpoint is not OpenAI-compatible. ` + | |
| `Check the base URL (for Ollama use http://host:11434/v1, for llama.cpp/vLLM/LM Studio the /v1 path).`); | |
| } | |
| normalizeChoices(data); | |
| data._routed_via = { platform: this.platform, model: modelId }; | |
| return data; | |
| } | |
| async *streamChatCompletion(apiKey, messages, modelId, options) { | |
| const res = await this.fetchWithTimeout(`${this.baseUrl}/chat/completions`, { | |
| method: 'POST', | |
| headers: { | |
| ...this.authHeader(apiKey), | |
| 'Content-Type': 'application/json', | |
| ...this.extraHeaders, | |
| }, | |
| body: JSON.stringify({ | |
| model: modelId, | |
| messages, | |
| temperature: options?.temperature, | |
| max_tokens: options?.max_tokens, | |
| top_p: options?.top_p, | |
| tools: options?.tools, | |
| tool_choice: options?.tool_choice, | |
| parallel_tool_calls: this.resolveParallelToolCalls(options), | |
| stream: true, | |
| }), | |
| }, this.timeoutMs); | |
| if (!res.ok) { | |
| const err = await res.json().catch(() => ({})); | |
| throw providerHttpError(res, `${this.name} API error ${res.status}: ${err.error?.message ?? res.statusText}`); | |
| } | |
| yield* this.readSseStream(res); | |
| } | |
| async validateKey(apiKey) { | |
| // Note: transport errors (DNS / timeout / TLS) propagate to the caller. | |
| // health.ts catches them and marks status='error' WITHOUT incrementing | |
| // the consecutive-failure counter β only confirmed 401/403 disables a key. | |
| const url = this.validateUrl ?? `${this.baseUrl}/models`; | |
| // 30s (not 10s): some upstreams return a large /v1/models catalog that | |
| // takes >10s from high-latency regions (e.g. NVIDIA NIM measured ~11.2s | |
| // from India). A 10s cap aborted those calls and health.ts marked a | |
| // perfectly good key status='error'. 30s aligns with chatCompletion's | |
| // own slow-upstream allowance and costs nothing for fast providers. | |
| const res = await this.fetchWithTimeout(url, { | |
| method: 'GET', | |
| headers: { | |
| ...this.authHeader(apiKey), | |
| ...this.extraHeaders, | |
| }, | |
| }, 30000); | |
| return res.status !== 401 && res.status !== 403; | |
| } | |
| } | |
| /** | |
| * Some providers (Z.ai glm-4.5-flash, Cloudflare DeepSeek-R1-distill, others) | |
| * return reasoning models' actual answer in `message.reasoning_content` with | |
| * `message.content === ""`. Fold reasoning_content into content so OpenAI- | |
| * compatible clients see a non-empty assistant message. | |
| * | |
| * Other providers (Mistral magistral-medium) return `message.content` as an | |
| * array of text segments instead of a string. Flatten to string. | |
| */ | |
| function normalizeChoices(data) { | |
| for (const choice of data.choices ?? []) { | |
| const msg = choice.message; | |
| // Flatten array content (Mistral magistral) β join text segments. | |
| if (Array.isArray(msg.content)) { | |
| msg.content = msg.content | |
| .map(seg => (typeof seg === 'string' ? seg : (seg.text ?? ''))) | |
| .join(''); | |
| } | |
| // Fold reasoning into content if content is empty AND there are no | |
| // tool_calls. With tool_calls present, content=null is the correct OpenAI | |
| // shape; folding reasoning would confuse clients that branch on content. | |
| // Field naming varies by provider: Z.ai uses `reasoning_content`, Ollama | |
| // uses `reasoning`. Prefer `reasoning_content` when both are set. | |
| const hasToolCalls = Array.isArray(msg.tool_calls) && msg.tool_calls.length > 0; | |
| if (!hasToolCalls && (msg.content === '' || msg.content == null)) { | |
| const fold = (typeof msg.reasoning_content === 'string' && msg.reasoning_content.length > 0) | |
| ? msg.reasoning_content | |
| : (typeof msg.reasoning === 'string' && msg.reasoning.length > 0 ? msg.reasoning : null); | |
| if (fold !== null) | |
| msg.content = fold; | |
| } | |
| } | |
| } | |
| //# sourceMappingURL=openai-compat.js.map |