{ "fetched_at": "2026-08-18T03:46:15.442239+00:00", "source_url": "https://openrouter.ai/api/v1/models", "models": [ { "id": "qwen/qwen3.8-27b", "canonical_slug": "qwen/qwen3.8-27b-20260814", "hugging_face_id": "Qwen/Qwen3.8-27B", "name": "Qwen: Qwen3.8 27B", "created": 1786722910, "description": "Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.00000045", "completion": "0.0000032", "input_cache_read": "0.00000005" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 20 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.8-27b-20260814/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "xhigh", "medium", "low" ], "default_effort": "xhigh" } }, { "id": "dots-studio/dots-3-note-preview:free", "canonical_slug": "dots-studio/dots-3-note-preview-20260813", "hugging_face_id": null, "name": "Dots Studio: Dots3-Note Preview (free)", "created": 1786680361, "description": "Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...", "context_length": 512000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 512000, "max_completion_tokens": 512000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/dots-studio/dots-3-note-preview-20260813/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "google/gemini-3.7-flash", "canonical_slug": "google/gemini-3.7-flash-20260813", "hugging_face_id": null, "name": "Google: Gemini 3.7 Flash", "created": 1786640581, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000208333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.7-flash-20260813/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1253, "win_rate": 58, "rank": 6 }, { "arena": "models", "category": "3d", "elo": 1374, "win_rate": 67.6, "rank": 4 }, { "arena": "models", "category": "codecategories", "elo": 1331, "win_rate": 58.7, "rank": 5 }, { "arena": "models", "category": "dataviz", "elo": 1358, "win_rate": 63, "rank": 5 }, { "arena": "models", "category": "gamedev", "elo": 1362, "win_rate": 61.6, "rank": 4 }, { "arena": "models", "category": "uicomponent", "elo": 1312, "win_rate": 54.2, "rank": 12 }, { "arena": "models", "category": "website", "elo": 1318, "win_rate": 56.7, "rank": 5 } ], "artificial_analysis": { "intelligence_index": 56, "coding_index": 76.1, "agentic_index": 45.1 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "google/gemini-3.7-flash:batch", "canonical_slug": "google/gemini-3.7-flash-20260813", "hugging_face_id": null, "name": "Google: Gemini 3.7 Flash (batch)", "created": 1786640581, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000001875", "completion": "0.0000009375", "image": "0.0000001875", "audio": "0.0000001875", "input_audio_cache": "0.00000001875", "web_search": "0.014", "internal_reasoning": "0.0000009375", "input_cache_read": "0.00000001875", "input_cache_write": "0.0000000208333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.7-flash-20260813/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1253, "win_rate": 58, "rank": 6 }, { "arena": "models", "category": "3d", "elo": 1374, "win_rate": 67.6, "rank": 4 }, { "arena": "models", "category": "codecategories", "elo": 1331, "win_rate": 58.7, "rank": 5 }, { "arena": "models", "category": "dataviz", "elo": 1358, "win_rate": 63, "rank": 5 }, { "arena": "models", "category": "gamedev", "elo": 1362, "win_rate": 61.6, "rank": 4 }, { "arena": "models", "category": "uicomponent", "elo": 1312, "win_rate": 54.2, "rank": 12 }, { "arena": "models", "category": "website", "elo": 1318, "win_rate": 56.7, "rank": 5 } ], "artificial_analysis": { "intelligence_index": 56, "coding_index": 76.1, "agentic_index": 45.1 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "bytedance-seed/seed-2-1-turbo", "canonical_slug": "bytedance-seed/seed-2-1-turbo-20260810", "hugging_face_id": null, "name": "ByteDance Seed: Seed 2.1 Turbo", "created": 1786552176, "description": "Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.0000025" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/bytedance-seed/seed-2-1-turbo-20260810/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3.8-2.4t-a95b", "canonical_slug": "qwen/qwen3.8-2.4t-a95b-20260812", "hugging_face_id": "Qwen/Qwen3.8-2.4T-A95B", "name": "Qwen: Qwen3.8 2.4T A95B", "created": 1786551702, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "context_length": 1048576, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.00000025" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 20 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.8-2.4t-a95b-20260812/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 57.7, "coding_index": 71.9, "agentic_index": 57.1 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "xhigh", "medium", "low" ], "default_effort": "xhigh" } }, { "id": "bytedance-seed/seed-2.0-code", "canonical_slug": "bytedance-seed/seed-2.0-code-20260730", "hugging_face_id": null, "name": "ByteDance Seed: Seed-2.0-Code", "created": 1786550701, "description": "Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.000001", "completion": "0.000006" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/bytedance-seed/seed-2.0-code-20260730/endpoints" }, "reasoning": { "mandatory": false, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "deepseek/deepseek-v4-pro-0813", "canonical_slug": "deepseek/deepseek-v4-pro-20260813", "hugging_face_id": "deepseek-ai/DeepSeek-V4-Pro-0813", "name": "DeepSeek: DeepSeek V4 Pro 0813", "created": 1786549364, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "context_length": 1048576, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044", "overrides": [ { "utc_start": 1000, "utc_end": 100, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 100, "utc_end": 400, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" }, { "utc_start": 400, "utc_end": 600, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 600, "utc_end": 1000, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 384000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 1 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-v4-pro-20260813/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 53.2, "coding_index": 68.8, "agentic_index": 49.6 } }, "reasoning": { "mandatory": false, "supported_efforts": [ "max", "high", "low" ], "default_effort": "high" } }, { "id": "x-ai/grok-4.6", "canonical_slug": "x-ai/grok-4.6-20260810", "hugging_face_id": null, "name": "SpaceXAI: Grok 4.6", "created": 1786548957, "description": "Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.", "context_length": 500000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "web_search": "0.005", "input_cache_read": "0.0000005", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000012", "input_cache_read": "0.000001" } ] }, "top_provider": { "context_length": 500000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/x-ai/grok-4.6-20260810/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1182, "win_rate": 46.8, "rank": 22 }, { "arena": "agents", "category": "fullstack", "elo": 1285, "win_rate": 55.5, "rank": 5 }, { "arena": "agents", "category": "mobileapps", "elo": 1276, "win_rate": 57.6, "rank": 2 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1211, "win_rate": 55.5, "rank": 10 }, { "arena": "agents", "category": "webapps", "elo": 1279, "win_rate": 57.2, "rank": 5 }, { "arena": "models", "category": "3d", "elo": 1332, "win_rate": 52.8, "rank": 9 }, { "arena": "models", "category": "asciiart", "elo": 1344, "win_rate": 64.7, "rank": 2 }, { "arena": "models", "category": "codecategories", "elo": 1322, "win_rate": 53.5, "rank": 7 }, { "arena": "models", "category": "dataviz", "elo": 1313, "win_rate": 51.8, "rank": 9 }, { "arena": "models", "category": "gamedev", "elo": 1346, "win_rate": 55.7, "rank": 6 }, { "arena": "models", "category": "uicomponent", "elo": 1347, "win_rate": 59.5, "rank": 3 }, { "arena": "models", "category": "website", "elo": 1317, "win_rate": 54, "rank": 7 } ], "artificial_analysis": { "intelligence_index": 60.9, "coding_index": 76.8, "agentic_index": 58.7 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "nvidia/nemotron-3.5-lightning", "canonical_slug": "nvidia/nemotron-3.5-lightning-20260807", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16", "name": "NVIDIA: Nemotron 3.5 Lightning", "created": 1786452751, "description": "NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000008", "completion": "0.0000002", "input_cache_read": "0.00000004" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3.5-lightning-20260807/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 23.6, "coding_index": 26.8, "agentic_index": 13.8 } }, "reasoning": { "mandatory": false } }, { "id": "nvidia/nemotron-3.5-lightning:free", "canonical_slug": "nvidia/nemotron-3.5-lightning-20260807", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16", "name": "NVIDIA: Nemotron 3.5 Lightning (free)", "created": 1786452751, "description": "NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "seed", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3.5-lightning-20260807/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 23.6, "coding_index": 26.8, "agentic_index": 13.8 } }, "reasoning": { "mandatory": false } }, { "id": "sakana/sakana-namazu", "canonical_slug": "sakana/namazu-20260811", "hugging_face_id": null, "name": "Sakana: Sakana Namazu", "created": 1786410129, "description": "Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...", "context_length": 262144, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "web_search": "0.007", "input_cache_read": "0.00000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "reasoning", "reasoning_effort", "structured_outputs", "tool_choice", "tools", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/sakana/namazu-20260811/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "none" ], "default_effort": "high" } }, { "id": "upstage/solar-pro4", "canonical_slug": "upstage/solar-pro4-20260810", "hugging_face_id": null, "name": "Upstage: Solar Pro 4", "created": 1786371636, "description": "Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...", "context_length": 524288, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000003", "completion": "0.00000012", "input_cache_read": "0.000000006" }, "top_provider": { "context_length": 524288, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "response_format", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/upstage/solar-pro4-20260810/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 1163, "win_rate": 33.8, "rank": 68 }, { "arena": "models", "category": "dataviz", "elo": 1141, "win_rate": 31, "rank": 79 }, { "arena": "models", "category": "gamedev", "elo": 1143, "win_rate": 30.2, "rank": 72 }, { "arena": "models", "category": "website", "elo": 1167, "win_rate": 34.2, "rank": 70 } ], "artificial_analysis": { "intelligence_index": 41.6, "coding_index": 52.7, "agentic_index": 33.6 } }, "reasoning": { "mandatory": false } }, { "id": "meta/muse-glimmer-30b", "canonical_slug": "meta/muse-glimmer-30b-20260810", "hugging_face_id": "meta-models/Muse-Glimmer-30B", "name": "Meta: Muse Glimmer 30B", "created": 1786302394, "description": "Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...", "context_length": 131072, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000035", "completion": "0.0000015", "input_cache_read": "0.00000004" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 64 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/meta/muse-glimmer-30b-20260810/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "meta/muse-spark-1.2", "canonical_slug": "meta/muse-spark-1.2-20260805", "hugging_face_id": null, "name": "Meta: Muse Spark 1.2", "created": 1785959287, "description": "Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00000425", "web_search": "0.0025", "input_cache_read": "0.00000015" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": null, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/meta/muse-spark-1.2-20260805/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1237, "win_rate": 54, "rank": 8 }, { "arena": "agents", "category": "fullstack", "elo": 1242, "win_rate": 50.1, "rank": 12 }, { "arena": "agents", "category": "htmlslides", "elo": 1082, "win_rate": 33.3, "rank": 20 }, { "arena": "agents", "category": "mobileapps", "elo": 1207, "win_rate": 47.5, "rank": 16 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1121, "win_rate": 37.2, "rank": 19 }, { "arena": "agents", "category": "webapps", "elo": 1262, "win_rate": 52.4, "rank": 8 }, { "arena": "models", "category": "3d", "elo": 1352, "win_rate": 62.4, "rank": 7 }, { "arena": "models", "category": "codecategories", "elo": 1340, "win_rate": 59.3, "rank": 3 }, { "arena": "models", "category": "dataviz", "elo": 1365, "win_rate": 63.5, "rank": 3 }, { "arena": "models", "category": "gamedev", "elo": 1357, "win_rate": 60.6, "rank": 5 }, { "arena": "models", "category": "uicomponent", "elo": 1343, "win_rate": 58.4, "rank": 4 }, { "arena": "models", "category": "website", "elo": 1329, "win_rate": 57.8, "rank": 2 } ], "artificial_analysis": { "intelligence_index": 56.8, "coding_index": 72.2, "agentic_index": 49.3 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "qwen/qwen3.8-max", "canonical_slug": "qwen/qwen3.8-max-20260803", "hugging_face_id": null, "name": "Qwen: Qwen3.8 Max", "created": 1785731612, "description": "Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...", "context_length": 1000000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.00000025", "input_cache_write": "0.0000025" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.8-max-20260803/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1384, "win_rate": 59.6, "rank": 2 }, { "arena": "models", "category": "codecategories", "elo": 1321, "win_rate": 55.9, "rank": 8 }, { "arena": "models", "category": "dataviz", "elo": 1286, "win_rate": 53.7, "rank": 17 }, { "arena": "models", "category": "gamedev", "elo": 1340, "win_rate": 57.8, "rank": 9 }, { "arena": "models", "category": "uicomponent", "elo": 1342, "win_rate": 58.3, "rank": 5 }, { "arena": "models", "category": "website", "elo": 1300, "win_rate": 54.3, "rank": 11 } ], "artificial_analysis": { "intelligence_index": 58.1, "coding_index": 71.8, "agentic_index": 58.4 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "xhigh", "high", "medium", "low", "minimal" ], "default_effort": "xhigh" } }, { "id": "~deepseek/deepseek-v4-flash-latest", "canonical_slug": "~deepseek/deepseek-v4-flash-latest", "alias_target": { "name": "DeepSeek: DeepSeek V4 Flash 0731", "slug": "deepseek/deepseek-v4-flash-0731" }, "hugging_face_id": null, "name": "DeepSeek V4 Flash Latest", "created": 1785606009, "description": "This model always redirects to the latest model in the DeepSeek V4 Flash family.", "context_length": 1310720, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.000000078596", "completion": "0.000000157192", "input_cache_read": "0.0000000157192" }, "top_provider": { "context_length": 1024000, "max_completion_tokens": 384000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "parallel_tool_calls", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": [], "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~deepseek/deepseek-v4-flash-latest/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "high", "low" ], "default_effort": "high" } }, { "id": "deepseek/deepseek-v4-flash-0731", "canonical_slug": "deepseek/deepseek-v4-flash-20260731", "hugging_face_id": "deepseek-ai/DeepSeek-V4-Flash-0731", "name": "DeepSeek: DeepSeek V4 Flash 0731", "created": 1785478908, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "context_length": 1310720, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 393216, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "parallel_tool_calls", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": [], "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-v4-flash-20260731/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1262, "win_rate": 51.1, "rank": 32 }, { "arena": "models", "category": "codecategories", "elo": 1253, "win_rate": 47.6, "rank": 35 }, { "arena": "models", "category": "dataviz", "elo": 1184, "win_rate": 38.8, "rank": 58 }, { "arena": "models", "category": "gamedev", "elo": 1256, "win_rate": 46.8, "rank": 31 }, { "arena": "models", "category": "uicomponent", "elo": 1269, "win_rate": 48, "rank": 29 }, { "arena": "models", "category": "website", "elo": 1252, "win_rate": 47.5, "rank": 33 } ], "artificial_analysis": { "intelligence_index": 51.8, "coding_index": 69.1, "agentic_index": 48.4 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "high", "low" ], "default_effort": "high" } }, { "id": "thinkingmachines/inkling-small", "canonical_slug": "thinkingmachines/inkling-small-20260730", "hugging_face_id": "thinkingmachines/Inkling-Small", "name": "Thinking Machines: Inkling Small", "created": 1785443117, "description": "Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...", "context_length": 524288, "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000045", "completion": "0.0000012", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 524288, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/thinkingmachines/inkling-small-20260730/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 41.2, "coding_index": 52.9, "agentic_index": 31.9 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "high", "medium", "low", "minimal", "none" ], "default_effort": "high" } }, { "id": "qwen/qwen3.7-flash", "canonical_slug": "qwen/qwen3.7-flash-20260727", "hugging_face_id": null, "name": "Qwen: Qwen3.7 Flash", "created": 1785190561, "description": "Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...", "context_length": 1000000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.00000003", "completion": "0.00000013", "input_cache_read": "0.000000006", "input_cache_write": "0.000000038", "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.0000001", "completion": "0.0000004", "input_cache_read": "0.00000002", "input_cache_write": "0.000000125" }, { "min_prompt_tokens": 256000, "prompt": "0.0000002", "completion": "0.0000008", "input_cache_read": "0.00000004", "input_cache_write": "0.00000025" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.7-flash-20260727/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true } }, { "id": "anthropic/claude-opus-5-fast", "canonical_slug": "anthropic/claude-opus-5-fast-20260723", "hugging_face_id": null, "name": "Claude Opus 5 (Fast)", "created": 1784912546, "description": "Fast-mode variant of [Opus 5](/anthropic/claude-opus-5) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 5.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-opus-5-fast-20260723/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-opus-5", "canonical_slug": "anthropic/claude-opus-5-20260723", "hugging_face_id": null, "name": "Claude Opus 5", "created": 1784912544, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-opus-5-20260723/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "fullstack", "elo": 1347, "win_rate": 69.2, "rank": 2 }, { "arena": "agents", "category": "webapps", "elo": 1278, "win_rate": 56.3, "rank": 6 }, { "arena": "models", "category": "3d", "elo": 1383, "win_rate": 62.9, "rank": 3 }, { "arena": "models", "category": "codecategories", "elo": 1347, "win_rate": 59.2, "rank": 2 }, { "arena": "models", "category": "dataviz", "elo": 1364, "win_rate": 60.5, "rank": 4 }, { "arena": "models", "category": "gamedev", "elo": 1380, "win_rate": 61.7, "rank": 3 }, { "arena": "models", "category": "svg", "elo": 1359, "win_rate": 59.8, "rank": 1 }, { "arena": "models", "category": "uicomponent", "elo": 1368, "win_rate": 59.5, "rank": 2 }, { "arena": "models", "category": "website", "elo": 1327, "win_rate": 58.1, "rank": 3 } ], "artificial_analysis": { "intelligence_index": 63.1, "coding_index": 78, "agentic_index": 59.2 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-opus-5:batch", "canonical_slug": "anthropic/claude-opus-5-20260723", "hugging_face_id": null, "name": "Claude Opus 5 (batch)", "created": 1784912544, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-opus-5-20260723/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "fullstack", "elo": 1347, "win_rate": 69.2, "rank": 2 }, { "arena": "agents", "category": "webapps", "elo": 1278, "win_rate": 56.3, "rank": 6 }, { "arena": "models", "category": "3d", "elo": 1383, "win_rate": 62.9, "rank": 3 }, { "arena": "models", "category": "codecategories", "elo": 1347, "win_rate": 59.2, "rank": 2 }, { "arena": "models", "category": "dataviz", "elo": 1364, "win_rate": 60.5, "rank": 4 }, { "arena": "models", "category": "gamedev", "elo": 1380, "win_rate": 61.7, "rank": 3 }, { "arena": "models", "category": "svg", "elo": 1359, "win_rate": 59.8, "rank": 1 }, { "arena": "models", "category": "uicomponent", "elo": 1368, "win_rate": 59.5, "rank": 2 }, { "arena": "models", "category": "website", "elo": 1327, "win_rate": 58.1, "rank": 3 } ], "artificial_analysis": { "intelligence_index": 63.1, "coding_index": 78, "agentic_index": 59.2 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "inclusionai/ling-3.0-flash", "canonical_slug": "inclusionai/ling-3.0-flash-20260723", "hugging_face_id": "inclusionAI/Ling-3.0-flash", "name": "Ling-3.0-flash", "created": 1784818580, "description": "*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000021", "completion": "0.000000063", "input_cache_read": "0.0000000042" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/inclusionai/ling-3.0-flash-20260723/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 37.8, "coding_index": 50.6, "agentic_index": 29.3 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "poolside/laguna-s-2.1", "canonical_slug": "poolside/laguna-s-2.1-20260720", "hugging_face_id": "poolside/Laguna-S-2.1", "name": "Poolside: Laguna S 2.1", "created": 1784652683, "description": "Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...", "context_length": 1048576, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000009", "completion": "0.00000018", "input_cache_read": "0.000000009" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "temperature", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/poolside/laguna-s-2.1-20260720/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "poolside/laguna-s-2.1:free", "canonical_slug": "poolside/laguna-s-2.1-20260720", "hugging_face_id": "poolside/Laguna-S-2.1", "name": "Poolside: Laguna S 2.1 (free)", "created": 1784652683, "description": "Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "temperature", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/poolside/laguna-s-2.1-20260720/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "google/gemini-3.6-flash", "canonical_slug": "google/gemini-3.6-flash-20260721", "hugging_face_id": null, "name": "Google: Gemini 3.6 Flash", "created": 1784646733, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000075", "completion": "0.00000375", "image": "0.00000075", "audio": "0.00000075", "input_audio_cache": "0.000000075", "web_search": "0.014", "internal_reasoning": "0.00000375", "input_cache_read": "0.000000075", "input_cache_write": "0.0000000416666666666667" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.6-flash-20260721/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1200, "win_rate": 53.9, "rank": 6 }, { "arena": "agents", "category": "androidnative", "elo": 1223, "win_rate": 53.7, "rank": 11 }, { "arena": "agents", "category": "fullstack", "elo": 1202, "win_rate": 46.4, "rank": 18 }, { "arena": "agents", "category": "htmlslides", "elo": 1159, "win_rate": 40.8, "rank": 17 }, { "arena": "agents", "category": "mobileapps", "elo": 1253, "win_rate": 54.3, "rank": 6 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1153, "win_rate": 39.6, "rank": 17 }, { "arena": "agents", "category": "webapps", "elo": 1245, "win_rate": 50, "rank": 13 }, { "arena": "models", "category": "3d", "elo": 1330, "win_rate": 53.5, "rank": 10 }, { "arena": "models", "category": "codecategories", "elo": 1315, "win_rate": 54.2, "rank": 9 }, { "arena": "models", "category": "dataviz", "elo": 1325, "win_rate": 53.5, "rank": 7 }, { "arena": "models", "category": "gamedev", "elo": 1300, "win_rate": 51.9, "rank": 18 }, { "arena": "models", "category": "uicomponent", "elo": 1331, "win_rate": 55, "rank": 9 }, { "arena": "models", "category": "website", "elo": 1317, "win_rate": 56.5, "rank": 6 } ], "artificial_analysis": { "intelligence_index": 51.6, "coding_index": 69.2, "agentic_index": 40.5 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "google/gemini-3.6-flash:batch", "canonical_slug": "google/gemini-3.6-flash-20260721", "hugging_face_id": null, "name": "Google: Gemini 3.6 Flash (batch)", "created": 1784646733, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000416666666666667" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.6-flash-20260721/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1200, "win_rate": 53.9, "rank": 6 }, { "arena": "agents", "category": "androidnative", "elo": 1223, "win_rate": 53.7, "rank": 11 }, { "arena": "agents", "category": "fullstack", "elo": 1202, "win_rate": 46.4, "rank": 18 }, { "arena": "agents", "category": "htmlslides", "elo": 1159, "win_rate": 40.8, "rank": 17 }, { "arena": "agents", "category": "mobileapps", "elo": 1253, "win_rate": 54.3, "rank": 6 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1153, "win_rate": 39.6, "rank": 17 }, { "arena": "agents", "category": "webapps", "elo": 1245, "win_rate": 50, "rank": 13 }, { "arena": "models", "category": "3d", "elo": 1330, "win_rate": 53.5, "rank": 10 }, { "arena": "models", "category": "codecategories", "elo": 1315, "win_rate": 54.2, "rank": 9 }, { "arena": "models", "category": "dataviz", "elo": 1325, "win_rate": 53.5, "rank": 7 }, { "arena": "models", "category": "gamedev", "elo": 1300, "win_rate": 51.9, "rank": 18 }, { "arena": "models", "category": "uicomponent", "elo": 1331, "win_rate": 55, "rank": 9 }, { "arena": "models", "category": "website", "elo": 1317, "win_rate": 56.5, "rank": 6 } ], "artificial_analysis": { "intelligence_index": 51.6, "coding_index": 69.2, "agentic_index": 40.5 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "google/gemini-3.5-flash-lite", "canonical_slug": "google/gemini-3.5-flash-lite-20260721", "hugging_face_id": null, "name": "Google: Gemini 3.5 Flash Lite", "created": 1784646726, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.0000003", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.5-flash-lite-20260721/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 37.4, "coding_index": 49.3, "agentic_index": 27.2 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "minimal" } }, { "id": "google/gemini-3.5-flash-lite:batch", "canonical_slug": "google/gemini-3.5-flash-lite-20260721", "hugging_face_id": null, "name": "Google: Gemini 3.5 Flash Lite (batch)", "created": 1784646726, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "audio": "0.00000015", "input_audio_cache": "0.000000015", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.000000015" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.5-flash-lite-20260721/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 37.4, "coding_index": 49.3, "agentic_index": 27.2 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "minimal" } }, { "id": "meituan/longcat-2.0", "canonical_slug": "meituan/longcat-2.0-20260720", "hugging_face_id": "meituan-longcat/LongCat-2.0", "name": "Meituan: LongCat 2.0", "created": 1784554658, "description": "LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...", "context_length": 1048756, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.000000006" }, "top_provider": { "context_length": 1048756, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/meituan/longcat-2.0-20260720/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 45.3, "agentic_index": null } }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true } }, { "id": "thinkingmachines/inkling", "canonical_slug": "thinkingmachines/inkling-20260715", "hugging_face_id": "thinkingmachines/Inkling", "name": "Thinking Machines: Inkling", "created": 1784325956, "description": "Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...", "context_length": 1048576, "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000095", "completion": "0.00000405", "input_cache_read": "0.00000016" }, "top_provider": { "context_length": 524288, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/thinkingmachines/inkling-20260715/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1212, "win_rate": 43.4, "rank": 46 }, { "arena": "models", "category": "codecategories", "elo": 1216, "win_rate": 42.4, "rank": 44 }, { "arena": "models", "category": "dataviz", "elo": 1199, "win_rate": 40.5, "rank": 51 }, { "arena": "models", "category": "gamedev", "elo": 1198, "win_rate": 38.9, "rank": 53 }, { "arena": "models", "category": "uicomponent", "elo": 1202, "win_rate": 39.8, "rank": 50 }, { "arena": "models", "category": "website", "elo": 1222, "win_rate": 43.3, "rank": 43 } ], "artificial_analysis": { "intelligence_index": 42.3, "coding_index": 52.1, "agentic_index": 34.1 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "high", "medium", "low", "minimal", "none" ], "default_effort": "high" } }, { "id": "thinkingmachines/inkling:batch", "canonical_slug": "thinkingmachines/inkling-20260715", "hugging_face_id": "thinkingmachines/Inkling", "name": "Thinking Machines: Inkling (batch)", "created": 1784325956, "description": "Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...", "context_length": 524288, "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.000002025", "input_cache_read": "0.000000085" }, "top_provider": { "context_length": 524288, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/thinkingmachines/inkling-20260715/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1212, "win_rate": 43.4, "rank": 46 }, { "arena": "models", "category": "codecategories", "elo": 1216, "win_rate": 42.4, "rank": 44 }, { "arena": "models", "category": "dataviz", "elo": 1199, "win_rate": 40.5, "rank": 51 }, { "arena": "models", "category": "gamedev", "elo": 1198, "win_rate": 38.9, "rank": 53 }, { "arena": "models", "category": "uicomponent", "elo": 1202, "win_rate": 39.8, "rank": 50 }, { "arena": "models", "category": "website", "elo": 1222, "win_rate": 43.3, "rank": 43 } ], "artificial_analysis": { "intelligence_index": 42.3, "coding_index": 52.1, "agentic_index": 34.1 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "high", "medium", "low", "minimal", "none" ], "default_effort": "high" } }, { "id": "openrouter/auto-beta", "canonical_slug": "openrouter/auto-beta", "hugging_face_id": null, "name": "Auto Router (Beta)", "created": 1784311165, "description": "Auto Router (Beta) is a task-aware router from OpenRouter. It classifies each request, then routes it the [most popular model](/rankings#task-spend) for that task based on aggregate spend, filtered by your...", "context_length": 2000000, "architecture": { "modality": "text+image+file+audio+video->text+image", "input_modalities": [ "text", "image", "audio", "file", "video" ], "output_modalities": [ "text", "image" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "-1", "completion": "-1" }, "top_provider": { "context_length": null, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "prediction", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openrouter/auto-beta/endpoints" } }, { "id": "moonshotai/kimi-k3", "canonical_slug": "moonshotai/kimi-k3-20260715", "hugging_face_id": "moonshotai/Kimi-K3", "name": "MoonshotAI: Kimi K3", "created": 1784215858, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "context_length": 1048576, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k3-20260715/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1202, "win_rate": 49.3, "rank": 17 }, { "arena": "agents", "category": "fullstack", "elo": 1347, "win_rate": 66.4, "rank": 3 }, { "arena": "agents", "category": "godotgamedev", "elo": 1199, "win_rate": 48.5, "rank": 11 }, { "arena": "agents", "category": "mobileapps", "elo": 1292, "win_rate": 58.1, "rank": 1 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1383, "win_rate": 70.2, "rank": 1 }, { "arena": "agents", "category": "webapps", "elo": 1319, "win_rate": 62, "rank": 1 }, { "arena": "models", "category": "3d", "elo": 1449, "win_rate": 69.1, "rank": 1 }, { "arena": "models", "category": "codecategories", "elo": 1403, "win_rate": 65.8, "rank": 1 }, { "arena": "models", "category": "dataviz", "elo": 1371, "win_rate": 63.4, "rank": 1 }, { "arena": "models", "category": "gamedev", "elo": 1443, "win_rate": 66.4, "rank": 1 }, { "arena": "models", "category": "svg", "elo": 1341, "win_rate": 64.2, "rank": 2 }, { "arena": "models", "category": "uicomponent", "elo": 1384, "win_rate": 63.1, "rank": 1 }, { "arena": "models", "category": "website", "elo": 1368, "win_rate": 62.9, "rank": 1 } ], "artificial_analysis": { "intelligence_index": 59.7, "coding_index": 76.2, "agentic_index": 54.3 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "high", "low" ], "default_effort": "max" } }, { "id": "meta/muse-spark-1.1", "canonical_slug": "meta/muse-spark-1.1-20260709", "hugging_face_id": null, "name": "Meta: Muse Spark 1.1", "created": 1784215741, "description": "Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00000425", "web_search": "0.0025", "input_cache_read": "0.00000015" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": null, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/meta/muse-spark-1.1-20260709/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1188, "win_rate": 48.2, "rank": 8 }, { "arena": "agents", "category": "androidnative", "elo": 1222, "win_rate": 51.4, "rank": 12 }, { "arena": "agents", "category": "fullstack", "elo": 1240, "win_rate": 49.4, "rank": 13 }, { "arena": "agents", "category": "godotgamedev", "elo": 1142, "win_rate": 39.4, "rank": 17 }, { "arena": "agents", "category": "htmlslides", "elo": 1218, "win_rate": 51.7, "rank": 8 }, { "arena": "agents", "category": "mobileapps", "elo": 1200, "win_rate": 44.8, "rank": 19 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1185, "win_rate": 44.2, "rank": 14 }, { "arena": "agents", "category": "webapps", "elo": 1239, "win_rate": 50.3, "rank": 14 }, { "arena": "models", "category": "3d", "elo": 1304, "win_rate": 53.4, "rank": 17 }, { "arena": "models", "category": "asciiart", "elo": 1326, "win_rate": 61.7, "rank": 3 }, { "arena": "models", "category": "codecategories", "elo": 1298, "win_rate": 53.5, "rank": 13 }, { "arena": "models", "category": "dataviz", "elo": 1303, "win_rate": 52.5, "rank": 12 }, { "arena": "models", "category": "gamedev", "elo": 1320, "win_rate": 52.2, "rank": 12 }, { "arena": "models", "category": "svg", "elo": 1258, "win_rate": 48.2, "rank": 10 }, { "arena": "models", "category": "uicomponent", "elo": 1321, "win_rate": 53.4, "rank": 10 }, { "arena": "models", "category": "website", "elo": 1284, "win_rate": 53.2, "rank": 18 } ], "artificial_analysis": { "intelligence_index": 53.2, "coding_index": 71.3, "agentic_index": 39.7 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "kwaipilot/kat-coder-air-v2.5", "canonical_slug": "kwaipilot/kat-coder-air-v2.5-20260710", "hugging_face_id": null, "name": "Kwaipilot: KAT-Coder-Air V2.5", "created": 1783714590, "description": "KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 80000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "logprobs", "max_tokens", "presence_penalty", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/kwaipilot/kat-coder-air-v2.5-20260710/endpoints" } }, { "id": "kwaipilot/kat-coder-pro-v2.5", "canonical_slug": "kwaipilot/kat-coder-pro-v2.5-20260710", "hugging_face_id": null, "name": "Kwaipilot: KAT-Coder-Pro V2.5", "created": 1783714589, "description": "KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000074", "completion": "0.00000296", "input_cache_read": "0.00000015" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 80000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "logprobs", "max_tokens", "presence_penalty", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/kwaipilot/kat-coder-pro-v2.5-20260710/endpoints" } }, { "id": "openai/gpt-5.6-luna-pro", "canonical_slug": "openai/gpt-5.6-luna-pro-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Luna Pro", "created": 1783590867, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000004", "completion": "0.0000018", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-luna-pro-20260709/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-luna-pro:batch", "canonical_slug": "openai/gpt-5.6-luna-pro-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Luna Pro (batch)", "created": 1783590867, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000006", "web_search": "0.01", "input_cache_read": "0.00000001", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000002", "completion": "0.0000009", "input_cache_read": "0.00000002" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-luna-pro-20260709/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-luna", "canonical_slug": "openai/gpt-5.6-luna-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Luna", "created": 1783590864, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000004", "completion": "0.0000018", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-luna-20260709/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 52.3, "coding_index": 71.4, "agentic_index": 46.9 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-luna:batch", "canonical_slug": "openai/gpt-5.6-luna-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Luna (batch)", "created": 1783590864, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000006", "web_search": "0.01", "input_cache_read": "0.00000001", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000002", "completion": "0.0000009", "input_cache_read": "0.00000002" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-luna-20260709/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 52.3, "coding_index": 71.4, "agentic_index": 46.9 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-terra-pro", "canonical_slug": "openai/gpt-5.6-terra-pro-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Terra Pro", "created": 1783590861, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000004", "completion": "0.000018", "input_cache_read": "0.0000004", "input_cache_write": "0.000005" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-terra-pro-20260709/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-terra-pro:batch", "canonical_slug": "openai/gpt-5.6-terra-pro-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Terra Pro (batch)", "created": 1783590861, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000006", "web_search": "0.01", "input_cache_read": "0.0000001", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000002", "completion": "0.000009", "input_cache_read": "0.0000002" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-terra-pro-20260709/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-terra", "canonical_slug": "openai/gpt-5.6-terra-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Terra", "created": 1783590857, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000004", "completion": "0.000018", "input_cache_read": "0.0000004", "input_cache_write": "0.000005" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-terra-20260709/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 56.6, "coding_index": 76.7, "agentic_index": 50.2 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-terra:batch", "canonical_slug": "openai/gpt-5.6-terra-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Terra (batch)", "created": 1783590857, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000006", "web_search": "0.01", "input_cache_read": "0.0000001", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000002", "completion": "0.000009", "input_cache_read": "0.0000002" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-terra-20260709/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 56.6, "coding_index": 76.7, "agentic_index": 50.2 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-sol-pro", "canonical_slug": "openai/gpt-5.6-sol-pro-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Sol Pro", "created": 1783590854, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-sol-pro-20260709/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-sol-pro:batch", "canonical_slug": "openai/gpt-5.6-sol-pro-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Sol Pro (batch)", "created": 1783590854, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-sol-pro-20260709/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-sol", "canonical_slug": "openai/gpt-5.6-sol-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Sol", "created": 1783590850, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-sol-20260709/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 60.9, "coding_index": 77.4, "agentic_index": 57.8 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.6-sol:batch", "canonical_slug": "openai/gpt-5.6-sol-20260709", "hugging_face_id": null, "name": "OpenAI: GPT-5.6 Sol (batch)", "created": 1783590850, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.6-sol-20260709/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 60.9, "coding_index": 77.4, "agentic_index": 57.8 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "x-ai/grok-4.5", "canonical_slug": "x-ai/grok-4.5-20260708", "hugging_face_id": null, "name": "SpaceXAI: Grok 4.5", "created": 1783523154, "description": "Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.", "context_length": 500000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "web_search": "0.005", "input_cache_read": "0.0000003", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000012", "input_cache_read": "0.0000006" } ] }, "top_provider": { "context_length": 500000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/x-ai/grok-4.5-20260708/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1220, "win_rate": 56.6, "rank": 5 }, { "arena": "agents", "category": "androidnative", "elo": 1273, "win_rate": 67.1, "rank": 2 }, { "arena": "agents", "category": "fullstack", "elo": 1275, "win_rate": 63.1, "rank": 7 }, { "arena": "agents", "category": "godotgamedev", "elo": 1271, "win_rate": 62.2, "rank": 2 }, { "arena": "agents", "category": "htmlslides", "elo": 1223, "win_rate": 53.5, "rank": 6 }, { "arena": "agents", "category": "mobileapps", "elo": 1252, "win_rate": 54.5, "rank": 7 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1245, "win_rate": 55.3, "rank": 7 }, { "arena": "agents", "category": "webapps", "elo": 1249, "win_rate": 53.9, "rank": 11 }, { "arena": "models", "category": "3d", "elo": 1313, "win_rate": 50.3, "rank": 14 }, { "arena": "models", "category": "asciiart", "elo": 1299, "win_rate": 58.5, "rank": 7 }, { "arena": "models", "category": "codecategories", "elo": 1301, "win_rate": 51.3, "rank": 12 }, { "arena": "models", "category": "dataviz", "elo": 1296, "win_rate": 49.6, "rank": 15 }, { "arena": "models", "category": "gamedev", "elo": 1310, "win_rate": 49.1, "rank": 15 }, { "arena": "models", "category": "svg", "elo": 1259, "win_rate": 50.7, "rank": 9 }, { "arena": "models", "category": "uicomponent", "elo": 1310, "win_rate": 52.1, "rank": 13 }, { "arena": "models", "category": "website", "elo": 1299, "win_rate": 53.9, "rank": 12 } ], "artificial_analysis": { "intelligence_index": 55.8, "coding_index": 72.4, "agentic_index": 48.9 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "high" } }, { "id": "~x-ai/grok-latest", "canonical_slug": "~x-ai/grok-latest", "alias_target": { "name": "SpaceXAI: Grok 4.6", "slug": "x-ai/grok-4.6" }, "hugging_face_id": null, "name": "xAI: Grok Latest", "created": 1783519360, "description": "This model always redirects to the latest Grok model from xAI.", "context_length": 500000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "web_search": "0.005", "input_cache_read": "0.0000005", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000012", "input_cache_read": "0.000001" } ] }, "top_provider": { "context_length": 500000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~x-ai/grok-latest/endpoints" }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "aion-labs/aion-3.0-mini", "canonical_slug": "aion-labs/aion-3.0-mini-20260707", "hugging_face_id": null, "name": "AionLabs: Aion-3.0-Mini", "created": 1783443096, "description": "Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000007", "completion": "0.0000014", "input_cache_read": "0.00000018" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/aion-labs/aion-3.0-mini-20260707/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "aion-labs/aion-3.0", "canonical_slug": "aion-labs/aion-3.0-20260707", "hugging_face_id": null, "name": "AionLabs: Aion-3.0", "created": 1783443095, "description": "Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000006", "input_cache_read": "0.00000075" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/aion-labs/aion-3.0-20260707/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "tencent/hy3", "canonical_slug": "tencent/hy3-20260706", "hugging_face_id": "tencent/Hy3", "name": "Tencent: Hy3", "created": 1783344048, "description": "Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000132", "completion": "0.000000528", "input_cache_read": "0.000000033" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_completion_tokens", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 0.9, "top_p": 1, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/tencent/hy3-20260706/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1227, "win_rate": 43.8, "rank": 42 }, { "arena": "models", "category": "codecategories", "elo": 1197, "win_rate": 41.1, "rank": 52 }, { "arena": "models", "category": "dataviz", "elo": 1147, "win_rate": 36.1, "rank": 74 }, { "arena": "models", "category": "gamedev", "elo": 1175, "win_rate": 38.6, "rank": 62 }, { "arena": "models", "category": "uicomponent", "elo": 1190, "win_rate": 40.2, "rank": 60 }, { "arena": "models", "category": "website", "elo": 1194, "win_rate": 41.4, "rank": 58 } ] }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "low", "none" ], "default_effort": "high" } }, { "id": "poolside/laguna-xs-2.1", "canonical_slug": "poolside/laguna-xs-2.1-20260625", "hugging_face_id": "poolside/Laguna-XS-2.1", "name": "Poolside: Laguna XS 2.1", "created": 1783002429, "description": "Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000006", "completion": "0.00000012", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "temperature", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/poolside/laguna-xs-2.1-20260625/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "poolside/laguna-xs-2.1:free", "canonical_slug": "poolside/laguna-xs-2.1-20260625", "hugging_face_id": "poolside/Laguna-XS-2.1", "name": "Poolside: Laguna XS 2.1 (free)", "created": 1783002429, "description": "Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "temperature", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/poolside/laguna-xs-2.1-20260625/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "anthropic/claude-sonnet-5", "canonical_slug": "anthropic/claude-sonnet-5-20260630", "hugging_face_id": null, "name": "Anthropic: Claude Sonnet 5", "created": 1782843083, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-sonnet-5-20260630/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1234, "win_rate": 55.7, "rank": 4 }, { "arena": "agents", "category": "androidnative", "elo": 1232, "win_rate": 53.9, "rank": 10 }, { "arena": "agents", "category": "fullstack", "elo": 1274, "win_rate": 57.5, "rank": 8 }, { "arena": "agents", "category": "godotgamedev", "elo": 1270, "win_rate": 60.4, "rank": 4 }, { "arena": "agents", "category": "htmlslides", "elo": 1229, "win_rate": 53.9, "rank": 4 }, { "arena": "agents", "category": "mobileapps", "elo": 1234, "win_rate": 51.8, "rank": 10 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1244, "win_rate": 54.7, "rank": 8 }, { "arena": "agents", "category": "webapps", "elo": 1286, "win_rate": 57.5, "rank": 4 }, { "arena": "models", "category": "3d", "elo": 1309, "win_rate": 55, "rank": 15 }, { "arena": "models", "category": "asciiart", "elo": 1238, "win_rate": 52.2, "rank": 15 }, { "arena": "models", "category": "codecategories", "elo": 1295, "win_rate": 54.1, "rank": 16 }, { "arena": "models", "category": "dataviz", "elo": 1257, "win_rate": 52.3, "rank": 26 }, { "arena": "models", "category": "gamedev", "elo": 1343, "win_rate": 55.5, "rank": 7 }, { "arena": "models", "category": "svg", "elo": 1225, "win_rate": 53, "rank": 19 }, { "arena": "models", "category": "uicomponent", "elo": 1307, "win_rate": 54.5, "rank": 14 }, { "arena": "models", "category": "website", "elo": 1286, "win_rate": 54.1, "rank": 16 } ], "artificial_analysis": { "intelligence_index": 55.3, "coding_index": 71.5, "agentic_index": 49.7 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-sonnet-5:batch", "canonical_slug": "anthropic/claude-sonnet-5-20260630", "hugging_face_id": null, "name": "Anthropic: Claude Sonnet 5 (batch)", "created": 1782843083, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-sonnet-5-20260630/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1234, "win_rate": 55.7, "rank": 4 }, { "arena": "agents", "category": "androidnative", "elo": 1232, "win_rate": 53.9, "rank": 10 }, { "arena": "agents", "category": "fullstack", "elo": 1274, "win_rate": 57.5, "rank": 8 }, { "arena": "agents", "category": "godotgamedev", "elo": 1270, "win_rate": 60.4, "rank": 4 }, { "arena": "agents", "category": "htmlslides", "elo": 1229, "win_rate": 53.9, "rank": 4 }, { "arena": "agents", "category": "mobileapps", "elo": 1234, "win_rate": 51.8, "rank": 10 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1244, "win_rate": 54.7, "rank": 8 }, { "arena": "agents", "category": "webapps", "elo": 1286, "win_rate": 57.5, "rank": 4 }, { "arena": "models", "category": "3d", "elo": 1309, "win_rate": 55, "rank": 15 }, { "arena": "models", "category": "asciiart", "elo": 1238, "win_rate": 52.2, "rank": 15 }, { "arena": "models", "category": "codecategories", "elo": 1295, "win_rate": 54.1, "rank": 16 }, { "arena": "models", "category": "dataviz", "elo": 1257, "win_rate": 52.3, "rank": 26 }, { "arena": "models", "category": "gamedev", "elo": 1343, "win_rate": 55.5, "rank": 7 }, { "arena": "models", "category": "svg", "elo": 1225, "win_rate": 53, "rank": 19 }, { "arena": "models", "category": "uicomponent", "elo": 1307, "win_rate": 54.5, "rank": 14 }, { "arena": "models", "category": "website", "elo": 1286, "win_rate": 54.1, "rank": 16 } ], "artificial_analysis": { "intelligence_index": 55.3, "coding_index": 71.5, "agentic_index": 49.7 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "google/gemini-3.1-flash-lite-image", "canonical_slug": "google/gemini-3.1-flash-lite-image-20260630", "hugging_face_id": null, "name": "Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)", "created": 1782837225, "description": "Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...", "context_length": 65536, "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image_output": "0.00003", "web_search": "0.014" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 66000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "temperature", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-01-01", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-flash-lite-image-20260630/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "minimal" ], "default_effort": "minimal" } }, { "id": "nex-agi/nex-n2-mini", "canonical_slug": "nex-agi/nex-n2-mini", "hugging_face_id": "nex-agi/Nex-N2-Mini", "name": "Nex AGI: Nex-N2-Mini", "created": 1782312964, "description": "Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.000000025", "completion": "0.0000001", "input_cache_read": "0.0000000025" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "response_format", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.7, "top_p": 0.95, "top_k": 40, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nex-agi/nex-n2-mini/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "sakana/fugu-ultra", "canonical_slug": "sakana/fugu-ultra-20260615", "hugging_face_id": null, "name": "Sakana: Fugu Ultra", "created": 1782276303, "description": "Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...", "context_length": 1000000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00001", "completion": "0.000045", "input_cache_read": "0.000001" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "reasoning", "reasoning_effort", "structured_outputs", "tool_choice", "tools", "web_search_options" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/sakana/fugu-ultra-20260615/endpoints" }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high" ], "default_effort": "xhigh" } }, { "id": "google/gemini-3.1-flash-image", "canonical_slug": "google/gemini-3.1-flash-image-20260528", "hugging_face_id": null, "name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image)", "created": 1781754065, "description": "Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...", "context_length": 131072, "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image_output": "0.00006", "web_search": "0.014" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-flash-image-20260528/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "minimal" ], "default_effort": "minimal" } }, { "id": "google/gemini-3-pro-image", "canonical_slug": "google/gemini-3-pro-image-20260528", "hugging_face_id": null, "name": "Google: Nano Banana Pro (Gemini 3 Pro Image)", "created": 1781754054, "description": "Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...", "context_length": 131072, "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "image_output": "0.00012", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3-pro-image-20260528/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "cohere/north-mini-code:free", "canonical_slug": "cohere/north-mini-code-20260617", "hugging_face_id": "CohereLabs/North-Mini-Code-1.0", "name": "Cohere: North Mini Code (free)", "created": 1781723748, "description": "North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/cohere/north-mini-code-20260617/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 20.2, "coding_index": 36.5, "agentic_index": 3.1 } }, "reasoning": { "mandatory": false } }, { "id": "z-ai/glm-5.2", "canonical_slug": "z-ai/glm-5.2-20260616", "hugging_face_id": "zai-org/GLM-5.2", "name": "Z.ai: GLM 5.2", "created": 1781631930, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "context_length": 1048576, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.00000315", "input_cache_read": "0.000000115" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "parallel_tool_calls", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-5.2-20260616/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1184, "win_rate": 48.8, "rank": 10 }, { "arena": "agents", "category": "androidnative", "elo": 1208, "win_rate": 54.7, "rank": 15 }, { "arena": "agents", "category": "fullstack", "elo": 1264, "win_rate": 61.4, "rank": 9 }, { "arena": "agents", "category": "godotgamedev", "elo": 1142, "win_rate": 40.1, "rank": 18 }, { "arena": "agents", "category": "htmlslides", "elo": 1199, "win_rate": 50.1, "rank": 10 }, { "arena": "agents", "category": "mobileapps", "elo": 1209, "win_rate": 51.9, "rank": 14 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1207, "win_rate": 49.3, "rank": 11 }, { "arena": "agents", "category": "webapps", "elo": 1258, "win_rate": 56.8, "rank": 10 }, { "arena": "models", "category": "3d", "elo": 1353, "win_rate": 56.8, "rank": 6 }, { "arena": "models", "category": "asciiart", "elo": 1255, "win_rate": 52.2, "rank": 12 }, { "arena": "models", "category": "codecategories", "elo": 1326, "win_rate": 56.3, "rank": 6 }, { "arena": "models", "category": "dataviz", "elo": 1321, "win_rate": 53.7, "rank": 8 }, { "arena": "models", "category": "gamedev", "elo": 1327, "win_rate": 53.7, "rank": 11 }, { "arena": "models", "category": "svg", "elo": 1254, "win_rate": 55.5, "rank": 13 }, { "arena": "models", "category": "uicomponent", "elo": 1334, "win_rate": 56.9, "rank": 7 }, { "arena": "models", "category": "website", "elo": 1319, "win_rate": 57.5, "rank": 4 } ], "artificial_analysis": { "intelligence_index": 52.6, "coding_index": 68.8, "agentic_index": 45.7 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "xhigh", "high" ], "default_effort": "high" } }, { "id": "z-ai/glm-5.2:batch", "canonical_slug": "z-ai/glm-5.2-20260616", "hugging_face_id": "zai-org/GLM-5.2", "name": "Z.ai: GLM 5.2 (batch)", "created": 1781631930, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "context_length": 512000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000007", "completion": "0.0000022", "input_cache_read": "0.00000013" }, "top_provider": { "context_length": 512000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-5.2-20260616/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1184, "win_rate": 48.8, "rank": 10 }, { "arena": "agents", "category": "androidnative", "elo": 1208, "win_rate": 54.7, "rank": 15 }, { "arena": "agents", "category": "fullstack", "elo": 1264, "win_rate": 61.4, "rank": 9 }, { "arena": "agents", "category": "godotgamedev", "elo": 1142, "win_rate": 40.1, "rank": 18 }, { "arena": "agents", "category": "htmlslides", "elo": 1199, "win_rate": 50.1, "rank": 10 }, { "arena": "agents", "category": "mobileapps", "elo": 1209, "win_rate": 51.9, "rank": 14 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1207, "win_rate": 49.3, "rank": 11 }, { "arena": "agents", "category": "webapps", "elo": 1258, "win_rate": 56.8, "rank": 10 }, { "arena": "models", "category": "3d", "elo": 1353, "win_rate": 56.8, "rank": 6 }, { "arena": "models", "category": "asciiart", "elo": 1255, "win_rate": 52.2, "rank": 12 }, { "arena": "models", "category": "codecategories", "elo": 1326, "win_rate": 56.3, "rank": 6 }, { "arena": "models", "category": "dataviz", "elo": 1321, "win_rate": 53.7, "rank": 8 }, { "arena": "models", "category": "gamedev", "elo": 1327, "win_rate": 53.7, "rank": 11 }, { "arena": "models", "category": "svg", "elo": 1254, "win_rate": 55.5, "rank": 13 }, { "arena": "models", "category": "uicomponent", "elo": 1334, "win_rate": 56.9, "rank": 7 }, { "arena": "models", "category": "website", "elo": 1319, "win_rate": 57.5, "rank": 4 } ], "artificial_analysis": { "intelligence_index": 52.6, "coding_index": 68.8, "agentic_index": 45.7 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "xhigh", "high" ], "default_effort": "high" } }, { "id": "z-ai/glm-5.2:free", "canonical_slug": "z-ai/glm-5.2-20260616", "hugging_face_id": "zai-org/GLM-5.2", "name": "Z.ai: GLM 5.2 (free)", "created": 1781631930, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "seed", "stop", "temperature", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-5.2-20260616/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1184, "win_rate": 48.8, "rank": 10 }, { "arena": "agents", "category": "androidnative", "elo": 1208, "win_rate": 54.7, "rank": 15 }, { "arena": "agents", "category": "fullstack", "elo": 1264, "win_rate": 61.4, "rank": 9 }, { "arena": "agents", "category": "godotgamedev", "elo": 1142, "win_rate": 40.1, "rank": 18 }, { "arena": "agents", "category": "htmlslides", "elo": 1199, "win_rate": 50.1, "rank": 10 }, { "arena": "agents", "category": "mobileapps", "elo": 1209, "win_rate": 51.9, "rank": 14 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1207, "win_rate": 49.3, "rank": 11 }, { "arena": "agents", "category": "webapps", "elo": 1258, "win_rate": 56.8, "rank": 10 }, { "arena": "models", "category": "3d", "elo": 1353, "win_rate": 56.8, "rank": 6 }, { "arena": "models", "category": "asciiart", "elo": 1255, "win_rate": 52.2, "rank": 12 }, { "arena": "models", "category": "codecategories", "elo": 1326, "win_rate": 56.3, "rank": 6 }, { "arena": "models", "category": "dataviz", "elo": 1321, "win_rate": 53.7, "rank": 8 }, { "arena": "models", "category": "gamedev", "elo": 1327, "win_rate": 53.7, "rank": 11 }, { "arena": "models", "category": "svg", "elo": 1254, "win_rate": 55.5, "rank": 13 }, { "arena": "models", "category": "uicomponent", "elo": 1334, "win_rate": 56.9, "rank": 7 }, { "arena": "models", "category": "website", "elo": 1319, "win_rate": 57.5, "rank": 4 } ], "artificial_analysis": { "intelligence_index": 52.6, "coding_index": 68.8, "agentic_index": 45.7 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "xhigh", "high" ], "default_effort": "high" } }, { "id": "openrouter/fusion", "canonical_slug": "openrouter/fusion", "hugging_face_id": null, "name": "OpenRouter: Fusion", "created": 1781371647, "description": "Fusion turns your prompt into a small multi-model deliberation. A panel of expert models (see below) analyzes your prompt in parallel with web search and web fetch enabled, then a...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "-1", "completion": "-1" }, "top_provider": { "context_length": null, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openrouter/fusion/endpoints" } }, { "id": "moonshotai/kimi-k2.7-code", "canonical_slug": "moonshotai/kimi-k2.7-code-20260612", "hugging_face_id": "moonshotai/Kimi-K2.7-Code", "name": "MoonshotAI: Kimi K2.7 Code", "created": 1781266361, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000071", "completion": "0.0000035", "input_cache_read": "0.00000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "parallel_tool_calls", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k2.7-code-20260612/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1135, "win_rate": 42.9, "rank": 16 }, { "arena": "agents", "category": "androidnative", "elo": 1174, "win_rate": 47.5, "rank": 24 }, { "arena": "agents", "category": "fullstack", "elo": 1206, "win_rate": 53.3, "rank": 17 }, { "arena": "agents", "category": "godotgamedev", "elo": 1185, "win_rate": 49.3, "rank": 12 }, { "arena": "agents", "category": "htmlslides", "elo": 1226, "win_rate": 54, "rank": 5 }, { "arena": "agents", "category": "mobileapps", "elo": 1200, "win_rate": 49.8, "rank": 20 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1186, "win_rate": 47.1, "rank": 13 }, { "arena": "agents", "category": "webapps", "elo": 1213, "win_rate": 48.1, "rank": 20 }, { "arena": "models", "category": "3d", "elo": 1293, "win_rate": 52.2, "rank": 22 }, { "arena": "models", "category": "asciiart", "elo": 1244, "win_rate": 52, "rank": 14 }, { "arena": "models", "category": "codecategories", "elo": 1277, "win_rate": 52.2, "rank": 25 }, { "arena": "models", "category": "dataviz", "elo": 1249, "win_rate": 51.2, "rank": 33 }, { "arena": "models", "category": "gamedev", "elo": 1252, "win_rate": 49.6, "rank": 33 }, { "arena": "models", "category": "svg", "elo": 1211, "win_rate": 49.2, "rank": 25 }, { "arena": "models", "category": "uicomponent", "elo": 1292, "win_rate": 53.3, "rank": 23 }, { "arena": "models", "category": "website", "elo": 1284, "win_rate": 53.7, "rank": 20 } ], "artificial_analysis": { "intelligence_index": 43, "coding_index": 60.8, "agentic_index": 30.3 } }, "reasoning": { "mandatory": true, "default_enabled": true } }, { "id": "moonshotai/kimi-k2.7-code:batch", "canonical_slug": "moonshotai/kimi-k2.7-code-20260612", "hugging_face_id": "moonshotai/Kimi-K2.7-Code", "name": "MoonshotAI: Kimi K2.7 Code (batch)", "created": 1781266361, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000475", "completion": "0.000002", "input_cache_read": "0.000000095" }, "top_provider": { "context_length": 262144, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k2.7-code-20260612/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1135, "win_rate": 42.9, "rank": 16 }, { "arena": "agents", "category": "androidnative", "elo": 1174, "win_rate": 47.5, "rank": 24 }, { "arena": "agents", "category": "fullstack", "elo": 1206, "win_rate": 53.3, "rank": 17 }, { "arena": "agents", "category": "godotgamedev", "elo": 1185, "win_rate": 49.3, "rank": 12 }, { "arena": "agents", "category": "htmlslides", "elo": 1226, "win_rate": 54, "rank": 5 }, { "arena": "agents", "category": "mobileapps", "elo": 1200, "win_rate": 49.8, "rank": 20 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1186, "win_rate": 47.1, "rank": 13 }, { "arena": "agents", "category": "webapps", "elo": 1213, "win_rate": 48.1, "rank": 20 }, { "arena": "models", "category": "3d", "elo": 1293, "win_rate": 52.2, "rank": 22 }, { "arena": "models", "category": "asciiart", "elo": 1244, "win_rate": 52, "rank": 14 }, { "arena": "models", "category": "codecategories", "elo": 1277, "win_rate": 52.2, "rank": 25 }, { "arena": "models", "category": "dataviz", "elo": 1249, "win_rate": 51.2, "rank": 33 }, { "arena": "models", "category": "gamedev", "elo": 1252, "win_rate": 49.6, "rank": 33 }, { "arena": "models", "category": "svg", "elo": 1211, "win_rate": 49.2, "rank": 25 }, { "arena": "models", "category": "uicomponent", "elo": 1292, "win_rate": 53.3, "rank": 23 }, { "arena": "models", "category": "website", "elo": 1284, "win_rate": 53.7, "rank": 20 } ], "artificial_analysis": { "intelligence_index": 43, "coding_index": 60.8, "agentic_index": 30.3 } }, "reasoning": { "mandatory": true, "default_enabled": true } }, { "id": "~anthropic/claude-fable-latest", "canonical_slug": "~anthropic/claude-fable-latest", "alias_target": { "name": "Anthropic: Claude Fable 5", "slug": "anthropic/claude-fable-5" }, "hugging_face_id": null, "name": "Anthropic: Claude Fable Latest", "created": 1781029944, "description": "This model always redirects to the latest model in the Claude Fable family.", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~anthropic/claude-fable-latest/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-fable-5", "canonical_slug": "anthropic/claude-5-fable-20260609", "hugging_face_id": null, "name": "Anthropic: Claude Fable 5", "created": 1781007515, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-5-fable-20260609/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1296, "win_rate": 65.1, "rank": 1 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1254, "win_rate": 59.4, "rank": 1 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1252, "win_rate": 59.5, "rank": 1 }, { "arena": "agents", "category": "androidnative", "elo": 1291, "win_rate": 63.1, "rank": 1 }, { "arena": "agents", "category": "fullstack", "elo": 1288, "win_rate": 61, "rank": 4 }, { "arena": "agents", "category": "godotgamedev", "elo": 1346, "win_rate": 70.2, "rank": 1 }, { "arena": "agents", "category": "htmlslides", "elo": 1268, "win_rate": 59.7, "rank": 1 }, { "arena": "agents", "category": "mobileapps", "elo": 1256, "win_rate": 56.5, "rank": 5 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1305, "win_rate": 63, "rank": 3 }, { "arena": "agents", "category": "webapps", "elo": 1294, "win_rate": 59.5, "rank": 3 }, { "arena": "models", "category": "3d", "elo": 1370, "win_rate": 62.2, "rank": 5 }, { "arena": "models", "category": "asciiart", "elo": 1361, "win_rate": 69.7, "rank": 1 }, { "arena": "models", "category": "codecategories", "elo": 1334, "win_rate": 59, "rank": 4 }, { "arena": "models", "category": "dataviz", "elo": 1337, "win_rate": 58, "rank": 6 }, { "arena": "models", "category": "gamedev", "elo": 1382, "win_rate": 62.1, "rank": 2 }, { "arena": "models", "category": "svg", "elo": 1339, "win_rate": 64.1, "rank": 3 }, { "arena": "models", "category": "uicomponent", "elo": 1334, "win_rate": 56.1, "rank": 6 }, { "arena": "models", "category": "website", "elo": 1314, "win_rate": 58.6, "rank": 8 } ], "artificial_analysis": { "intelligence_index": 62.1, "coding_index": 76.5, "agentic_index": 56.6 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-fable-5:batch", "canonical_slug": "anthropic/claude-5-fable-20260609", "hugging_face_id": null, "name": "Anthropic: Claude Fable 5 (batch)", "created": 1781007515, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-5-fable-20260609/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1296, "win_rate": 65.1, "rank": 1 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1254, "win_rate": 59.4, "rank": 1 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1252, "win_rate": 59.5, "rank": 1 }, { "arena": "agents", "category": "androidnative", "elo": 1291, "win_rate": 63.1, "rank": 1 }, { "arena": "agents", "category": "fullstack", "elo": 1288, "win_rate": 61, "rank": 4 }, { "arena": "agents", "category": "godotgamedev", "elo": 1346, "win_rate": 70.2, "rank": 1 }, { "arena": "agents", "category": "htmlslides", "elo": 1268, "win_rate": 59.7, "rank": 1 }, { "arena": "agents", "category": "mobileapps", "elo": 1256, "win_rate": 56.5, "rank": 5 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1305, "win_rate": 63, "rank": 3 }, { "arena": "agents", "category": "webapps", "elo": 1294, "win_rate": 59.5, "rank": 3 }, { "arena": "models", "category": "3d", "elo": 1370, "win_rate": 62.2, "rank": 5 }, { "arena": "models", "category": "asciiart", "elo": 1361, "win_rate": 69.7, "rank": 1 }, { "arena": "models", "category": "codecategories", "elo": 1334, "win_rate": 59, "rank": 4 }, { "arena": "models", "category": "dataviz", "elo": 1337, "win_rate": 58, "rank": 6 }, { "arena": "models", "category": "gamedev", "elo": 1382, "win_rate": 62.1, "rank": 2 }, { "arena": "models", "category": "svg", "elo": 1339, "win_rate": 64.1, "rank": 3 }, { "arena": "models", "category": "uicomponent", "elo": 1334, "win_rate": 56.1, "rank": 6 }, { "arena": "models", "category": "website", "elo": 1314, "win_rate": 58.6, "rank": 8 } ], "artificial_analysis": { "intelligence_index": 62.1, "coding_index": 76.5, "agentic_index": 56.6 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "nex-agi/nex-n2-pro", "canonical_slug": "nex-agi/nex-n2-pro", "hugging_face_id": "nex-agi/Nex-N2-Pro", "name": "Nex AGI: Nex-N2-Pro", "created": 1780937140, "description": "Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.000001", "input_cache_read": "0.000000025" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "reasoning", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.7, "top_p": 0.95, "top_k": 40, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nex-agi/nex-n2-pro/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1300, "win_rate": 53, "rank": 20 }, { "arena": "models", "category": "asciiart", "elo": 1128, "win_rate": 37.6, "rank": 49 }, { "arena": "models", "category": "codecategories", "elo": 1252, "win_rate": 48.5, "rank": 36 }, { "arena": "models", "category": "dataviz", "elo": 1252, "win_rate": 49.6, "rank": 30 }, { "arena": "models", "category": "gamedev", "elo": 1259, "win_rate": 49.2, "rank": 30 }, { "arena": "models", "category": "svg", "elo": 1233, "win_rate": 51.3, "rank": 17 }, { "arena": "models", "category": "uicomponent", "elo": 1249, "win_rate": 48, "rank": 38 }, { "arena": "models", "category": "website", "elo": 1233, "win_rate": 46.5, "rank": 41 } ], "artificial_analysis": { "intelligence_index": null, "coding_index": 59.1, "agentic_index": null } }, "reasoning": { "mandatory": false } }, { "id": "nvidia/nemotron-3.5-content-safety:free", "canonical_slug": "nvidia/nemotron-3.5-content-safety-20260604", "hugging_face_id": "nvidia/Nemotron-3.5-Content-Safety", "name": "NVIDIA: Nemotron 3.5 Content Safety (free)", "created": 1780581864, "description": "NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...", "context_length": 128000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "seed", "temperature", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3.5-content-safety-20260604/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b", "canonical_slug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", "name": "NVIDIA: Nemotron 3 Ultra", "created": 1780551208, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "context_length": 512288, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "input_cache_read": "0.0000002" }, "top_provider": { "context_length": 512288, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-ultra-550b-a55b-20260604/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1183, "win_rate": 41.1, "rank": 56 }, { "arena": "models", "category": "asciiart", "elo": 1107, "win_rate": 36.8, "rank": 52 }, { "arena": "models", "category": "codecategories", "elo": 1150, "win_rate": 36, "rank": 77 }, { "arena": "models", "category": "dataviz", "elo": 1149, "win_rate": 37.3, "rank": 73 }, { "arena": "models", "category": "gamedev", "elo": 1161, "win_rate": 37, "rank": 68 }, { "arena": "models", "category": "svg", "elo": 1116, "win_rate": 37.4, "rank": 53 }, { "arena": "models", "category": "uicomponent", "elo": 1169, "win_rate": 38.3, "rank": 66 }, { "arena": "models", "category": "website", "elo": 1130, "win_rate": 33.5, "rank": 84 } ], "artificial_analysis": { "intelligence_index": 38.3, "coding_index": 49.3, "agentic_index": 27.5 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true, "supported_efforts": [ "high", "medium" ], "default_effort": "high" } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b:batch", "canonical_slug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", "name": "NVIDIA: Nemotron 3 Ultra (batch)", "created": 1780551208, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "context_length": 512288, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000018", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 512288, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-ultra-550b-a55b-20260604/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1183, "win_rate": 41.1, "rank": 56 }, { "arena": "models", "category": "asciiart", "elo": 1107, "win_rate": 36.8, "rank": 52 }, { "arena": "models", "category": "codecategories", "elo": 1150, "win_rate": 36, "rank": 77 }, { "arena": "models", "category": "dataviz", "elo": 1149, "win_rate": 37.3, "rank": 73 }, { "arena": "models", "category": "gamedev", "elo": 1161, "win_rate": 37, "rank": 68 }, { "arena": "models", "category": "svg", "elo": 1116, "win_rate": 37.4, "rank": 53 }, { "arena": "models", "category": "uicomponent", "elo": 1169, "win_rate": 38.3, "rank": 66 }, { "arena": "models", "category": "website", "elo": 1130, "win_rate": 33.5, "rank": 84 } ], "artificial_analysis": { "intelligence_index": 38.3, "coding_index": 49.3, "agentic_index": 27.5 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true, "supported_efforts": [ "high", "medium" ], "default_effort": "high" } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b:free", "canonical_slug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", "name": "NVIDIA: Nemotron 3 Ultra (free)", "created": 1780551208, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "seed", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-ultra-550b-a55b-20260604/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1183, "win_rate": 41.1, "rank": 56 }, { "arena": "models", "category": "asciiart", "elo": 1107, "win_rate": 36.8, "rank": 52 }, { "arena": "models", "category": "codecategories", "elo": 1150, "win_rate": 36, "rank": 77 }, { "arena": "models", "category": "dataviz", "elo": 1149, "win_rate": 37.3, "rank": 73 }, { "arena": "models", "category": "gamedev", "elo": 1161, "win_rate": 37, "rank": 68 }, { "arena": "models", "category": "svg", "elo": 1116, "win_rate": 37.4, "rank": 53 }, { "arena": "models", "category": "uicomponent", "elo": 1169, "win_rate": 38.3, "rank": 66 }, { "arena": "models", "category": "website", "elo": 1130, "win_rate": 33.5, "rank": 84 } ], "artificial_analysis": { "intelligence_index": 38.3, "coding_index": 49.3, "agentic_index": 27.5 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true, "supported_efforts": [ "high", "medium" ], "default_effort": "high" } }, { "id": "qwen/qwen3.7-plus", "canonical_slug": "qwen/qwen3.7-plus-20260602", "hugging_face_id": null, "name": "Qwen: Qwen3.7 Plus", "created": 1780491783, "description": "Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...", "context_length": 1000000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.00000032", "completion": "0.00000128", "input_cache_read": "0.000000064", "input_cache_write": "0.0000004", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000096", "completion": "0.00000384", "input_cache_read": "0.000000192", "input_cache_write": "0.0000012" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.7-plus-20260602/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1290, "win_rate": 49.2, "rank": 24 }, { "arena": "models", "category": "asciiart", "elo": 1172, "win_rate": 43.2, "rank": 36 }, { "arena": "models", "category": "codecategories", "elo": 1281, "win_rate": 50.1, "rank": 22 }, { "arena": "models", "category": "dataviz", "elo": 1266, "win_rate": 50.1, "rank": 22 }, { "arena": "models", "category": "gamedev", "elo": 1293, "win_rate": 50.7, "rank": 21 }, { "arena": "models", "category": "svg", "elo": 1246, "win_rate": 52.9, "rank": 14 }, { "arena": "models", "category": "uicomponent", "elo": 1275, "win_rate": 49, "rank": 27 }, { "arena": "models", "category": "website", "elo": 1281, "win_rate": 51.2, "rank": 22 } ], "artificial_analysis": { "intelligence_index": 39.4, "coding_index": 55.9, "agentic_index": 20.7 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "minimax/minimax-m3", "canonical_slug": "minimax/minimax-m3-20260531", "hugging_face_id": "MiniMaxAI/Minimax-M3", "name": "MiniMax: MiniMax M3", "created": 1780245374, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "context_length": 1048576, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006" }, "top_provider": { "context_length": 524288, "max_completion_tokens": 512000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m3-20260531/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1194, "win_rate": 52, "rank": 7 }, { "arena": "agents", "category": "androidnative", "elo": 1174, "win_rate": 46.1, "rank": 25 }, { "arena": "agents", "category": "fullstack", "elo": 1220, "win_rate": 49.7, "rank": 15 }, { "arena": "agents", "category": "htmlslides", "elo": 1178, "win_rate": 45.8, "rank": 14 }, { "arena": "agents", "category": "mobileapps", "elo": 1222, "win_rate": 50.1, "rank": 11 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1206, "win_rate": 49.5, "rank": 12 }, { "arena": "agents", "category": "webapps", "elo": 1234, "win_rate": 49.4, "rank": 16 }, { "arena": "models", "category": "3d", "elo": 1265, "win_rate": 52.9, "rank": 31 }, { "arena": "models", "category": "asciiart", "elo": 1188, "win_rate": 47, "rank": 25 }, { "arena": "models", "category": "codecategories", "elo": 1271, "win_rate": 53.2, "rank": 26 }, { "arena": "models", "category": "dataviz", "elo": 1253, "win_rate": 52, "rank": 29 }, { "arena": "models", "category": "gamedev", "elo": 1252, "win_rate": 48.5, "rank": 34 }, { "arena": "models", "category": "svg", "elo": 1207, "win_rate": 50.1, "rank": 27 }, { "arena": "models", "category": "uicomponent", "elo": 1273, "win_rate": 52.7, "rank": 28 }, { "arena": "models", "category": "website", "elo": 1272, "win_rate": 53.7, "rank": 25 } ], "artificial_analysis": { "intelligence_index": 45.4, "coding_index": 58.6, "agentic_index": 36.1 } }, "reasoning": { "mandatory": false } }, { "id": "minimax/minimax-m3:batch", "canonical_slug": "minimax/minimax-m3-20260531", "hugging_face_id": "MiniMaxAI/Minimax-M3", "name": "MiniMax: MiniMax M3 (batch)", "created": 1780245374, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "context_length": 524288, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 524288, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m3-20260531/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1194, "win_rate": 52, "rank": 7 }, { "arena": "agents", "category": "androidnative", "elo": 1174, "win_rate": 46.1, "rank": 25 }, { "arena": "agents", "category": "fullstack", "elo": 1220, "win_rate": 49.7, "rank": 15 }, { "arena": "agents", "category": "htmlslides", "elo": 1178, "win_rate": 45.8, "rank": 14 }, { "arena": "agents", "category": "mobileapps", "elo": 1222, "win_rate": 50.1, "rank": 11 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1206, "win_rate": 49.5, "rank": 12 }, { "arena": "agents", "category": "webapps", "elo": 1234, "win_rate": 49.4, "rank": 16 }, { "arena": "models", "category": "3d", "elo": 1265, "win_rate": 52.9, "rank": 31 }, { "arena": "models", "category": "asciiart", "elo": 1188, "win_rate": 47, "rank": 25 }, { "arena": "models", "category": "codecategories", "elo": 1271, "win_rate": 53.2, "rank": 26 }, { "arena": "models", "category": "dataviz", "elo": 1253, "win_rate": 52, "rank": 29 }, { "arena": "models", "category": "gamedev", "elo": 1252, "win_rate": 48.5, "rank": 34 }, { "arena": "models", "category": "svg", "elo": 1207, "win_rate": 50.1, "rank": 27 }, { "arena": "models", "category": "uicomponent", "elo": 1273, "win_rate": 52.7, "rank": 28 }, { "arena": "models", "category": "website", "elo": 1272, "win_rate": 53.7, "rank": 25 } ], "artificial_analysis": { "intelligence_index": 45.4, "coding_index": 58.6, "agentic_index": 36.1 } }, "reasoning": { "mandatory": false } }, { "id": "stepfun/step-3.7-flash", "canonical_slug": "stepfun/step-3.7-flash-20260528", "hugging_face_id": "stepfun-ai/Step-3.7-Flash", "name": "StepFun: Step 3.7 Flash", "created": 1779985069, "description": "Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.00000115", "input_cache_read": "0.00000004" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 256000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/stepfun/step-3.7-flash-20260528/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1174, "win_rate": 41.8, "rank": 59 }, { "arena": "models", "category": "asciiart", "elo": 1191, "win_rate": 46.7, "rank": 24 }, { "arena": "models", "category": "codecategories", "elo": 1196, "win_rate": 44.1, "rank": 53 }, { "arena": "models", "category": "dataviz", "elo": 1191, "win_rate": 44.4, "rank": 54 }, { "arena": "models", "category": "gamedev", "elo": 1193, "win_rate": 41, "rank": 55 }, { "arena": "models", "category": "svg", "elo": 1108, "win_rate": 38.8, "rank": 55 }, { "arena": "models", "category": "uicomponent", "elo": 1202, "win_rate": 43.8, "rank": 51 }, { "arena": "models", "category": "website", "elo": 1202, "win_rate": 45.4, "rank": 52 } ], "artificial_analysis": { "intelligence_index": 30.9, "coding_index": 39.6, "agentic_index": 21.7 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "anthropic/claude-opus-4.8-fast", "canonical_slug": "anthropic/claude-4.8-opus-fast-20260528", "hugging_face_id": null, "name": "Anthropic: Claude Opus 4.8 (Fast)", "created": 1779913703, "description": "Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.8-opus-fast-20260528/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-opus-4.8", "canonical_slug": "anthropic/claude-4.8-opus-20260528", "hugging_face_id": null, "name": "Anthropic: Claude Opus 4.8", "created": 1779905091, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.8-opus-20260528/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1257, "win_rate": 61.5, "rank": 2 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1227, "win_rate": 55.6, "rank": 4 }, { "arena": "agents", "category": "agenticslides", "elo": 1294, "win_rate": 64.8, "rank": 2 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1230, "win_rate": 56, "rank": 4 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1310, "win_rate": 68.9, "rank": 2 }, { "arena": "agents", "category": "androidnative", "elo": 1258, "win_rate": 58.4, "rank": 5 }, { "arena": "agents", "category": "fullstack", "elo": 1276, "win_rate": 59.7, "rank": 6 }, { "arena": "agents", "category": "godotgamedev", "elo": 1255, "win_rate": 58.9, "rank": 5 }, { "arena": "agents", "category": "htmlslides", "elo": 1241, "win_rate": 56.2, "rank": 3 }, { "arena": "agents", "category": "mobileapps", "elo": 1250, "win_rate": 56, "rank": 8 }, { "arena": "agents", "category": "pptxslides", "elo": 1306, "win_rate": 67.9, "rank": 2 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1298, "win_rate": 65.9, "rank": 4 }, { "arena": "agents", "category": "webapps", "elo": 1259, "win_rate": 53.3, "rank": 9 }, { "arena": "models", "category": "3d", "elo": 1273, "win_rate": 53.5, "rank": 28 }, { "arena": "models", "category": "asciiart", "elo": 1300, "win_rate": 61.9, "rank": 6 }, { "arena": "models", "category": "codecategories", "elo": 1266, "win_rate": 53.8, "rank": 27 }, { "arena": "models", "category": "dataviz", "elo": 1260, "win_rate": 54.7, "rank": 24 }, { "arena": "models", "category": "gamedev", "elo": 1288, "win_rate": 54.6, "rank": 24 }, { "arena": "models", "category": "svg", "elo": 1217, "win_rate": 53.8, "rank": 22 }, { "arena": "models", "category": "uicomponent", "elo": 1282, "win_rate": 55.2, "rank": 26 }, { "arena": "models", "category": "website", "elo": 1262, "win_rate": 54, "rank": 28 } ], "artificial_analysis": { "intelligence_index": 57.3, "coding_index": 74.3, "agentic_index": 49.4 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-opus-4.8:batch", "canonical_slug": "anthropic/claude-4.8-opus-20260528", "hugging_face_id": null, "name": "Anthropic: Claude Opus 4.8 (batch)", "created": 1779905091, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.8-opus-20260528/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1257, "win_rate": 61.5, "rank": 2 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1227, "win_rate": 55.6, "rank": 4 }, { "arena": "agents", "category": "agenticslides", "elo": 1294, "win_rate": 64.8, "rank": 2 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1230, "win_rate": 56, "rank": 4 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1310, "win_rate": 68.9, "rank": 2 }, { "arena": "agents", "category": "androidnative", "elo": 1258, "win_rate": 58.4, "rank": 5 }, { "arena": "agents", "category": "fullstack", "elo": 1276, "win_rate": 59.7, "rank": 6 }, { "arena": "agents", "category": "godotgamedev", "elo": 1255, "win_rate": 58.9, "rank": 5 }, { "arena": "agents", "category": "htmlslides", "elo": 1241, "win_rate": 56.2, "rank": 3 }, { "arena": "agents", "category": "mobileapps", "elo": 1250, "win_rate": 56, "rank": 8 }, { "arena": "agents", "category": "pptxslides", "elo": 1306, "win_rate": 67.9, "rank": 2 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1298, "win_rate": 65.9, "rank": 4 }, { "arena": "agents", "category": "webapps", "elo": 1259, "win_rate": 53.3, "rank": 9 }, { "arena": "models", "category": "3d", "elo": 1273, "win_rate": 53.5, "rank": 28 }, { "arena": "models", "category": "asciiart", "elo": 1300, "win_rate": 61.9, "rank": 6 }, { "arena": "models", "category": "codecategories", "elo": 1266, "win_rate": 53.8, "rank": 27 }, { "arena": "models", "category": "dataviz", "elo": 1260, "win_rate": 54.7, "rank": 24 }, { "arena": "models", "category": "gamedev", "elo": 1288, "win_rate": 54.6, "rank": 24 }, { "arena": "models", "category": "svg", "elo": 1217, "win_rate": 53.8, "rank": 22 }, { "arena": "models", "category": "uicomponent", "elo": 1282, "win_rate": 55.2, "rank": 26 }, { "arena": "models", "category": "website", "elo": 1262, "win_rate": 54, "rank": 28 } ], "artificial_analysis": { "intelligence_index": 57.3, "coding_index": 74.3, "agentic_index": 49.4 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "qwen/qwen3.7-max", "canonical_slug": "qwen/qwen3.7-max-20260520", "hugging_face_id": null, "name": "Qwen: Qwen3.7 Max", "created": 1779376861, "description": "Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.000001475", "completion": "0.000004425", "input_cache_read": "0.000000295", "input_cache_write": "0.00000184375" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.7-max-20260520/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1174, "win_rate": 48.3, "rank": 13 }, { "arena": "agents", "category": "androidnative", "elo": 1188, "win_rate": 49.3, "rank": 20 }, { "arena": "agents", "category": "fullstack", "elo": 1214, "win_rate": 50.5, "rank": 16 }, { "arena": "agents", "category": "godotgamedev", "elo": 1225, "win_rate": 55.4, "rank": 8 }, { "arena": "agents", "category": "htmlslides", "elo": 1194, "win_rate": 47.4, "rank": 12 }, { "arena": "agents", "category": "mobileapps", "elo": 1206, "win_rate": 48.6, "rank": 17 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1214, "win_rate": 51, "rank": 9 }, { "arena": "agents", "category": "webapps", "elo": 1248, "win_rate": 51.3, "rank": 12 }, { "arena": "models", "category": "3d", "elo": 1321, "win_rate": 56.3, "rank": 12 }, { "arena": "models", "category": "asciiart", "elo": 1248, "win_rate": 53.6, "rank": 13 }, { "arena": "models", "category": "codecategories", "elo": 1297, "win_rate": 55.7, "rank": 14 }, { "arena": "models", "category": "dataviz", "elo": 1298, "win_rate": 53.8, "rank": 14 }, { "arena": "models", "category": "gamedev", "elo": 1312, "win_rate": 55.6, "rank": 14 }, { "arena": "models", "category": "svg", "elo": 1256, "win_rate": 59.6, "rank": 11 }, { "arena": "models", "category": "uicomponent", "elo": 1300, "win_rate": 53.7, "rank": 17 }, { "arena": "models", "category": "website", "elo": 1286, "win_rate": 56, "rank": 17 } ], "artificial_analysis": { "intelligence_index": 46.7, "coding_index": 66, "agentic_index": 30.9 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "x-ai/grok-build-0.1", "canonical_slug": "x-ai/grok-build-0.1-20260520", "hugging_face_id": null, "name": "SpaceXAI: Grok Build 0.1", "created": 1779298123, "description": "Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...", "context_length": 256000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000002", "web_search": "0.005", "input_cache_read": "0.0000002", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000002", "completion": "0.000004", "input_cache_read": "0.0000004" } ] }, "top_provider": { "context_length": 256000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/x-ai/grok-build-0.1-20260520/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 40.7, "coding_index": 51.5, "agentic_index": 28.9 } }, "reasoning": { "mandatory": true } }, { "id": "google/gemini-3.5-flash", "canonical_slug": "google/gemini-3.5-flash-20260519", "hugging_face_id": null, "name": "Google: Gemini 3.5 Flash", "created": 1779193800, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000015", "completion": "0.000009", "image": "0.0000015", "audio": "0.000003", "input_audio_cache": "0.0000003", "web_search": "0.014", "internal_reasoning": "0.000009", "input_cache_read": "0.00000015", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-01", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.5-flash-20260519/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1183, "win_rate": 54, "rank": 11 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1162, "win_rate": 45.8, "rank": 7 }, { "arena": "agents", "category": "agenticslides", "elo": 1244, "win_rate": 57.5, "rank": 4 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1162, "win_rate": 45.7, "rank": 7 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1242, "win_rate": 57.8, "rank": 3 }, { "arena": "agents", "category": "androidnative", "elo": 1221, "win_rate": 55.6, "rank": 13 }, { "arena": "agents", "category": "fullstack", "elo": 1226, "win_rate": 55.8, "rank": 14 }, { "arena": "agents", "category": "godotgamedev", "elo": 1137, "win_rate": 42.9, "rank": 21 }, { "arena": "agents", "category": "htmlslides", "elo": 1170, "win_rate": 45.8, "rank": 15 }, { "arena": "agents", "category": "mobileapps", "elo": 1221, "win_rate": 53.6, "rank": 12 }, { "arena": "agents", "category": "pptxslides", "elo": 1244, "win_rate": 57.7, "rank": 3 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1247, "win_rate": 57.4, "rank": 6 }, { "arena": "agents", "category": "webapps", "elo": 1235, "win_rate": 52.6, "rank": 15 }, { "arena": "models", "category": "3d", "elo": 1292, "win_rate": 57.8, "rank": 23 }, { "arena": "models", "category": "asciiart", "elo": 1290, "win_rate": 59.5, "rank": 8 }, { "arena": "models", "category": "codecategories", "elo": 1282, "win_rate": 55.6, "rank": 21 }, { "arena": "models", "category": "dataviz", "elo": 1254, "win_rate": 54.5, "rank": 28 }, { "arena": "models", "category": "gamedev", "elo": 1307, "win_rate": 54.9, "rank": 16 }, { "arena": "models", "category": "svg", "elo": 1286, "win_rate": 60.3, "rank": 5 }, { "arena": "models", "category": "uicomponent", "elo": 1303, "win_rate": 56, "rank": 16 }, { "arena": "models", "category": "website", "elo": 1273, "win_rate": 55, "rank": 24 } ], "artificial_analysis": { "intelligence_index": 52, "coding_index": 70.1, "agentic_index": 39.7 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "google/gemini-3.5-flash:batch", "canonical_slug": "google/gemini-3.5-flash-20260519", "hugging_face_id": null, "name": "Google: Gemini 3.5 Flash (batch)", "created": 1779193800, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "image": "0.00000075", "audio": "0.0000015", "input_audio_cache": "0.00000015", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000075" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-01", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.5-flash-20260519/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1183, "win_rate": 54, "rank": 11 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1162, "win_rate": 45.8, "rank": 7 }, { "arena": "agents", "category": "agenticslides", "elo": 1244, "win_rate": 57.5, "rank": 4 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1162, "win_rate": 45.7, "rank": 7 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1242, "win_rate": 57.8, "rank": 3 }, { "arena": "agents", "category": "androidnative", "elo": 1221, "win_rate": 55.6, "rank": 13 }, { "arena": "agents", "category": "fullstack", "elo": 1226, "win_rate": 55.8, "rank": 14 }, { "arena": "agents", "category": "godotgamedev", "elo": 1137, "win_rate": 42.9, "rank": 21 }, { "arena": "agents", "category": "htmlslides", "elo": 1170, "win_rate": 45.8, "rank": 15 }, { "arena": "agents", "category": "mobileapps", "elo": 1221, "win_rate": 53.6, "rank": 12 }, { "arena": "agents", "category": "pptxslides", "elo": 1244, "win_rate": 57.7, "rank": 3 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1247, "win_rate": 57.4, "rank": 6 }, { "arena": "agents", "category": "webapps", "elo": 1235, "win_rate": 52.6, "rank": 15 }, { "arena": "models", "category": "3d", "elo": 1292, "win_rate": 57.8, "rank": 23 }, { "arena": "models", "category": "asciiart", "elo": 1290, "win_rate": 59.5, "rank": 8 }, { "arena": "models", "category": "codecategories", "elo": 1282, "win_rate": 55.6, "rank": 21 }, { "arena": "models", "category": "dataviz", "elo": 1254, "win_rate": 54.5, "rank": 28 }, { "arena": "models", "category": "gamedev", "elo": 1307, "win_rate": 54.9, "rank": 16 }, { "arena": "models", "category": "svg", "elo": 1286, "win_rate": 60.3, "rank": 5 }, { "arena": "models", "category": "uicomponent", "elo": 1303, "win_rate": 56, "rank": 16 }, { "arena": "models", "category": "website", "elo": 1273, "win_rate": 55, "rank": 24 } ], "artificial_analysis": { "intelligence_index": 52, "coding_index": 70.1, "agentic_index": 39.7 } }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "anthropic/claude-opus-4.7-fast", "canonical_slug": "anthropic/claude-4.7-opus-fast-20260512", "hugging_face_id": null, "name": "Anthropic: Claude Opus 4.7 (Fast)", "created": 1778613011, "description": "Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.00003", "completion": "0.00015", "web_search": "0.01", "input_cache_read": "0.000003", "input_cache_write": "0.0000375", "input_cache_write_1h": "0.00006" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.7-opus-fast-20260512/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "perceptron/perceptron-mk1", "canonical_slug": "perceptron/perceptron-mk1-20260512", "hugging_face_id": null, "name": "Perceptron: Perceptron Mk1", "created": 1778597029, "description": "Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...", "context_length": 32768, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000015" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/perceptron/perceptron-mk1-20260512/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "inclusionai/ring-2.6-1t", "canonical_slug": "inclusionai/ring-2.6-1t-20260508", "hugging_face_id": null, "name": "inclusionAI: Ring-2.6-1T", "created": 1778247440, "description": "Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000075", "completion": "0.000000625", "input_cache_read": "0.000000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/inclusionai/ring-2.6-1t-20260508/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 42.8, "agentic_index": null } }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high" ], "default_effort": "high" } }, { "id": "google/gemini-3.1-flash-lite", "canonical_slug": "google/gemini-3.1-flash-lite-20260507", "hugging_face_id": null, "name": "Google: Gemini 3.1 Flash Lite", "created": 1778168828, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "input_cache_read": "0.000000025", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-flash-lite-20260507/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "minimal" } }, { "id": "google/gemini-3.1-flash-lite:batch", "canonical_slug": "google/gemini-3.1-flash-lite-20260507", "hugging_face_id": null, "name": "Google: Gemini 3.1 Flash Lite (batch)", "created": 1778168828, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000000125", "completion": "0.00000075", "image": "0.000000125", "audio": "0.00000025", "input_audio_cache": "0.000000025", "web_search": "0.014", "internal_reasoning": "0.00000075", "input_cache_read": "0.0000000125" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-flash-lite-20260507/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "minimal" } }, { "id": "openai/gpt-chat-latest", "canonical_slug": "openai/gpt-chat-latest-20260505", "hugging_face_id": null, "name": "OpenAI: GPT Chat Latest", "created": 1778000212, "description": "GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-chat-latest-20260505/endpoints" } }, { "id": "x-ai/grok-4.3", "canonical_slug": "x-ai/grok-4.3-20260430", "hugging_face_id": null, "name": "SpaceXAI: Grok 4.3", "created": 1777591821, "description": "Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/x-ai/grok-4.3-20260430/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1007, "win_rate": 28, "rank": 19 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1066, "win_rate": 31.8, "rank": 10 }, { "arena": "agents", "category": "agenticslides", "elo": 1072, "win_rate": 31.9, "rank": 10 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1066, "win_rate": 31.7, "rank": 10 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1068, "win_rate": 32.4, "rank": 10 }, { "arena": "agents", "category": "androidnative", "elo": 999, "win_rate": 23.1, "rank": 36 }, { "arena": "agents", "category": "fullstack", "elo": 1037, "win_rate": 29.5, "rank": 36 }, { "arena": "agents", "category": "godotgamedev", "elo": 1035, "win_rate": 31.2, "rank": 29 }, { "arena": "agents", "category": "htmlslides", "elo": 1057, "win_rate": 30.1, "rank": 21 }, { "arena": "agents", "category": "mobileapps", "elo": 1113, "win_rate": 37, "rank": 35 }, { "arena": "agents", "category": "pptxslides", "elo": 1071, "win_rate": 32.5, "rank": 9 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1072, "win_rate": 30.8, "rank": 21 }, { "arena": "agents", "category": "webapps", "elo": 1168, "win_rate": 44.3, "rank": 24 }, { "arena": "models", "category": "3d", "elo": 1174, "win_rate": 43.3, "rank": 58 }, { "arena": "models", "category": "asciiart", "elo": 1169, "win_rate": 45.5, "rank": 37 }, { "arena": "models", "category": "codecategories", "elo": 1205, "win_rate": 46.5, "rank": 47 }, { "arena": "models", "category": "dataviz", "elo": 1202, "win_rate": 46.4, "rank": 49 }, { "arena": "models", "category": "gamedev", "elo": 1212, "win_rate": 47.7, "rank": 47 }, { "arena": "models", "category": "svg", "elo": 1122, "win_rate": 41.3, "rank": 52 }, { "arena": "models", "category": "uicomponent", "elo": 1221, "win_rate": 47, "rank": 44 }, { "arena": "models", "category": "website", "elo": 1204, "win_rate": 46.2, "rank": 50 } ], "artificial_analysis": { "intelligence_index": 37.9, "coding_index": 42.2, "agentic_index": 24.2 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "none" ], "default_effort": "low" } }, { "id": "ibm-granite/granite-4.1-8b", "canonical_slug": "ibm-granite/granite-4.1-8b-20260429", "hugging_face_id": "ibm-granite/granite-4.1-8b", "name": "IBM: Granite 4.1 8B", "created": 1777577071, "description": "Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000005", "completion": "0.0000001", "input_cache_read": "0.00000005" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/ibm-granite/granite-4.1-8b-20260429/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 9.5, "agentic_index": null } } }, { "id": "mistralai/mistral-medium-3-5", "canonical_slug": "mistralai/mistral-medium-3.5-20260430", "hugging_face_id": null, "name": "Mistral: Mistral Medium 3.5", "created": 1777570439, "description": "Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...", "context_length": 262144, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000015", "completion": "0.0000075" }, "top_provider": { "context_length": 262144, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-medium-3.5-20260430/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 30.4, "coding_index": 46.9, "agentic_index": 19.2 } }, "reasoning": { "mandatory": false, "supported_efforts": [ "high", "none" ], "default_effort": "high" } }, { "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", "canonical_slug": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428", "hugging_face_id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16", "name": "NVIDIA: Nemotron 3 Nano Omni (free)", "created": 1777393095, "description": "NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...", "context_length": 256000, "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "seed", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 13.8, "agentic_index": null } }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true } }, { "id": "~anthropic/claude-haiku-latest", "canonical_slug": "~anthropic/claude-haiku-latest", "alias_target": { "name": "Anthropic: Claude Haiku 4.5", "slug": "anthropic/claude-haiku-4.5" }, "hugging_face_id": null, "name": "Anthropic Claude Haiku Latest", "created": 1777318492, "description": "This model always redirects to the latest model in the Anthropic Claude Haiku family.", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~anthropic/claude-haiku-latest/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "~openai/gpt-mini-latest", "canonical_slug": "~openai/gpt-mini-latest", "alias_target": { "name": "OpenAI: GPT-5.4 Mini", "slug": "openai/gpt-5.4-mini" }, "hugging_face_id": null, "name": "OpenAI GPT Mini Latest", "created": 1777318471, "description": "This model always redirects to the latest model in the OpenAI GPT Mini family.", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "web_search": "0.01", "input_cache_read": "0.000000075" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/~openai/gpt-mini-latest/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "~google/gemini-pro-latest", "canonical_slug": "~google/gemini-pro-latest", "alias_target": { "name": "Google: Gemini 3.1 Pro Preview", "slug": "google/gemini-3.1-pro-preview" }, "hugging_face_id": null, "name": "Google Gemini Pro Latest", "created": 1777318451, "description": "This model always redirects to the latest model in the Google Gemini Pro family.", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000018", "audio": "0.000004", "input_audio_cache": "0.0000004", "input_cache_read": "0.0000004" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~google/gemini-pro-latest/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "~moonshotai/kimi-latest", "canonical_slug": "~moonshotai/kimi-latest", "alias_target": { "name": "MoonshotAI: Kimi K3", "slug": "moonshotai/kimi-k3" }, "hugging_face_id": null, "name": "MoonshotAI Kimi Latest", "created": 1777318428, "description": "This model always redirects to the latest model in the MoonshotAI Kimi family.", "context_length": 1048576, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.0000026", "completion": "0.000013", "input_cache_read": "0.00000029" }, "top_provider": { "context_length": 974842, "max_completion_tokens": 974842, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~moonshotai/kimi-latest/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "high", "low" ], "default_effort": "max" } }, { "id": "~google/gemini-flash-latest", "canonical_slug": "~google/gemini-flash-latest", "alias_target": { "name": "Google: Gemini 3.7 Flash", "slug": "google/gemini-3.7-flash" }, "hugging_face_id": null, "name": "Google Gemini Flash Latest", "created": 1777318398, "description": "This model always redirects to the latest model in the Google Gemini Flash family.", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000208333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~google/gemini-flash-latest/endpoints" }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "~anthropic/claude-sonnet-latest", "canonical_slug": "~anthropic/claude-sonnet-latest", "alias_target": { "name": "Anthropic: Claude Sonnet 5", "slug": "anthropic/claude-sonnet-5" }, "hugging_face_id": null, "name": "Anthropic Claude Sonnet Latest", "created": 1777318368, "description": "This model always redirects to the latest model in the Anthropic Claude Sonnet family.", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~anthropic/claude-sonnet-latest/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "~openai/gpt-latest", "canonical_slug": "~openai/gpt-latest", "alias_target": { "name": "OpenAI: GPT-5.6 Sol", "slug": "openai/gpt-5.6-sol" }, "hugging_face_id": null, "name": "OpenAI GPT Latest", "created": 1777318334, "description": "This model always redirects to the latest model in the OpenAI GPT family.", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2026-02-16", "expiration_date": null, "links": { "details": "/api/v1/models/~openai/gpt-latest/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "qwen/qwen3.5-plus-20260420", "canonical_slug": "qwen/qwen3.5-plus-20260420", "hugging_face_id": null, "name": "Qwen: Qwen3.5 Plus 2026-04-20", "created": 1777261368, "description": "Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...", "context_length": 1000000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000018", "input_cache_write": "0.000000375", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.000000375", "completion": "0.00000225", "input_cache_write": "0.00000046875" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-plus-20260420/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3.6-flash", "canonical_slug": "qwen/qwen3.6-flash", "hugging_face_id": null, "name": "Qwen: Qwen3.6 Flash", "created": 1777261362, "description": "Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...", "context_length": 1000000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000001875", "completion": "0.000001125", "input_cache_write": "0.000000234375", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000075", "completion": "0.000003", "input_cache_write": "0.0000009375" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.6-flash/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3.6-35b-a3b", "canonical_slug": "qwen/qwen3.6-35b-a3b-20260415", "hugging_face_id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen: Qwen3.6 35B A3B", "created": 1777260255, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.00000014", "completion": "0.000001", "input_cache_read": "0.00000005" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 20 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.6-35b-a3b-20260415/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 32.1, "coding_index": 41.9, "agentic_index": 21.6 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "qwen/qwen3.6-max-preview", "canonical_slug": "qwen/qwen3.6-max-preview-20260420", "hugging_face_id": null, "name": "Qwen: Qwen3.6 Max Preview", "created": 1777260242, "description": "Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.000001027", "completion": "0.000006162", "input_cache_write": "0.00000128375", "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.00000158", "completion": "0.00000948", "input_cache_write": "0.000001975" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.6-max-preview-20260420/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "qwen/qwen3.6-27b", "canonical_slug": "qwen/qwen3.6-27b-20260422", "hugging_face_id": "Qwen/Qwen3.6-27B", "name": "Qwen: Qwen3.6 27B", "created": 1777255064, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "input_cache_read": "0.00000012" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.6-27b-20260422/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 37.7, "coding_index": 53.7, "agentic_index": 27.5 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "openai/gpt-5.5-pro", "canonical_slug": "openai/gpt-5.5-pro-20260423", "hugging_face_id": "", "name": "OpenAI: GPT-5.5 Pro", "created": 1777051896, "description": "GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00003", "completion": "0.00018", "web_search": "0.01", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00006", "completion": "0.00027" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-12-01", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.5-pro-20260423/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.5-pro:batch", "canonical_slug": "openai/gpt-5.5-pro-20260423", "hugging_face_id": "", "name": "OpenAI: GPT-5.5 Pro (batch)", "created": 1777051896, "description": "GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000015", "completion": "0.00009", "web_search": "0.01", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00003", "completion": "0.000135" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-12-01", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.5-pro-20260423/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.5", "canonical_slug": "openai/gpt-5.5-20260423", "hugging_face_id": "", "name": "OpenAI: GPT-5.5", "created": 1777051893, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00001", "completion": "0.000045", "input_cache_read": "0.000001" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-12-01", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.5-20260423/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1180, "win_rate": 51.4, "rank": 12 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1084, "win_rate": 34.2, "rank": 9 }, { "arena": "agents", "category": "agenticslides", "elo": 1150, "win_rate": 43.5, "rank": 7 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1077, "win_rate": 33.2, "rank": 9 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1155, "win_rate": 45.2, "rank": 7 }, { "arena": "agents", "category": "androidnative", "elo": 1198, "win_rate": 50.8, "rank": 18 }, { "arena": "agents", "category": "fullstack", "elo": 1118, "win_rate": 43, "rank": 25 }, { "arena": "agents", "category": "godotgamedev", "elo": 1213, "win_rate": 52.4, "rank": 10 }, { "arena": "agents", "category": "htmlslides", "elo": 1089, "win_rate": 34.2, "rank": 19 }, { "arena": "agents", "category": "mobileapps", "elo": 1202, "win_rate": 51.6, "rank": 18 }, { "arena": "agents", "category": "pptxslides", "elo": 1157, "win_rate": 45.3, "rank": 7 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1152, "win_rate": 43.3, "rank": 18 }, { "arena": "agents", "category": "webapps", "elo": 1154, "win_rate": 42.7, "rank": 28 }, { "arena": "models", "category": "3d", "elo": 1247, "win_rate": 52, "rank": 36 }, { "arena": "models", "category": "asciiart", "elo": 1287, "win_rate": 61, "rank": 9 }, { "arena": "models", "category": "codecategories", "elo": 1279, "win_rate": 55.3, "rank": 24 }, { "arena": "models", "category": "dataviz", "elo": 1278, "win_rate": 56.5, "rank": 19 }, { "arena": "models", "category": "gamedev", "elo": 1340, "win_rate": 59.6, "rank": 8 }, { "arena": "models", "category": "svg", "elo": 1268, "win_rate": 57.4, "rank": 6 }, { "arena": "models", "category": "uicomponent", "elo": 1284, "win_rate": 55.6, "rank": 25 }, { "arena": "models", "category": "website", "elo": 1268, "win_rate": 54.4, "rank": 26 } ], "artificial_analysis": { "intelligence_index": 56.3, "coding_index": 74.9, "agentic_index": 47.4 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.5:batch", "canonical_slug": "openai/gpt-5.5-20260423", "hugging_face_id": "", "name": "OpenAI: GPT-5.5 (batch)", "created": 1777051893, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-12-01", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.5-20260423/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1180, "win_rate": 51.4, "rank": 12 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1084, "win_rate": 34.2, "rank": 9 }, { "arena": "agents", "category": "agenticslides", "elo": 1150, "win_rate": 43.5, "rank": 7 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1077, "win_rate": 33.2, "rank": 9 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1155, "win_rate": 45.2, "rank": 7 }, { "arena": "agents", "category": "androidnative", "elo": 1198, "win_rate": 50.8, "rank": 18 }, { "arena": "agents", "category": "fullstack", "elo": 1118, "win_rate": 43, "rank": 25 }, { "arena": "agents", "category": "godotgamedev", "elo": 1213, "win_rate": 52.4, "rank": 10 }, { "arena": "agents", "category": "htmlslides", "elo": 1089, "win_rate": 34.2, "rank": 19 }, { "arena": "agents", "category": "mobileapps", "elo": 1202, "win_rate": 51.6, "rank": 18 }, { "arena": "agents", "category": "pptxslides", "elo": 1157, "win_rate": 45.3, "rank": 7 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1152, "win_rate": 43.3, "rank": 18 }, { "arena": "agents", "category": "webapps", "elo": 1154, "win_rate": 42.7, "rank": 28 }, { "arena": "models", "category": "3d", "elo": 1247, "win_rate": 52, "rank": 36 }, { "arena": "models", "category": "asciiart", "elo": 1287, "win_rate": 61, "rank": 9 }, { "arena": "models", "category": "codecategories", "elo": 1279, "win_rate": 55.3, "rank": 24 }, { "arena": "models", "category": "dataviz", "elo": 1278, "win_rate": 56.5, "rank": 19 }, { "arena": "models", "category": "gamedev", "elo": 1340, "win_rate": 59.6, "rank": 8 }, { "arena": "models", "category": "svg", "elo": 1268, "win_rate": 57.4, "rank": 6 }, { "arena": "models", "category": "uicomponent", "elo": 1284, "win_rate": 55.6, "rank": 25 }, { "arena": "models", "category": "website", "elo": 1268, "win_rate": 54.4, "rank": 26 } ], "artificial_analysis": { "intelligence_index": 56.3, "coding_index": 74.9, "agentic_index": 47.4 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "deepseek/deepseek-v4-pro", "canonical_slug": "deepseek/deepseek-v4-pro-20260423", "hugging_face_id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek: DeepSeek V4 Pro 0423", "created": 1777000679, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "context_length": 1048576, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044", "overrides": [ { "utc_start": 1000, "utc_end": 100, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 100, "utc_end": 400, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" }, { "utc_start": 400, "utc_end": 600, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 600, "utc_end": 1000, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 384000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 1 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-v4-pro-20260423/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "fullstack", "elo": 948, "win_rate": 22.1, "rank": 39 }, { "arena": "agents", "category": "godotgamedev", "elo": 1059, "win_rate": 34, "rank": 27 }, { "arena": "agents", "category": "webapps", "elo": 1000, "win_rate": 26.4, "rank": 37 }, { "arena": "models", "category": "3d", "elo": 1307, "win_rate": 57.6, "rank": 16 }, { "arena": "models", "category": "asciiart", "elo": 1183, "win_rate": 46.7, "rank": 30 }, { "arena": "models", "category": "codecategories", "elo": 1261, "win_rate": 53, "rank": 31 }, { "arena": "models", "category": "dataviz", "elo": 1222, "win_rate": 49, "rank": 43 }, { "arena": "models", "category": "gamedev", "elo": 1271, "win_rate": 54.2, "rank": 28 }, { "arena": "models", "category": "svg", "elo": 1173, "win_rate": 46.7, "rank": 39 }, { "arena": "models", "category": "uicomponent", "elo": 1251, "win_rate": 51.4, "rank": 37 }, { "arena": "models", "category": "website", "elo": 1247, "win_rate": 51.5, "rank": 35 } ], "artificial_analysis": { "intelligence_index": 45.3, "coding_index": 59.4, "agentic_index": 37.8 } }, "reasoning": { "mandatory": false, "supported_efforts": [ "xhigh", "high" ], "default_effort": "high" } }, { "id": "deepseek/deepseek-v4-flash", "canonical_slug": "deepseek/deepseek-v4-flash-20260423", "hugging_face_id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek: DeepSeek V4 Flash 0423", "created": 1777000666, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "context_length": 1048576, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "pricing": { "prompt": "0.0000000826", "completion": "0.0000001652", "input_cache_read": "0.00000001652" }, "top_provider": { "context_length": 1024000, "max_completion_tokens": 384000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-v4-flash-20260423/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1239, "win_rate": 49.3, "rank": 40 }, { "arena": "models", "category": "asciiart", "elo": 1142, "win_rate": 42.8, "rank": 46 }, { "arena": "models", "category": "codecategories", "elo": 1225, "win_rate": 48.9, "rank": 43 }, { "arena": "models", "category": "dataviz", "elo": 1146, "win_rate": 40.6, "rank": 75 }, { "arena": "models", "category": "gamedev", "elo": 1235, "win_rate": 50.2, "rank": 39 }, { "arena": "models", "category": "svg", "elo": 1190, "win_rate": 48.9, "rank": 31 }, { "arena": "models", "category": "uicomponent", "elo": 1192, "win_rate": 44.7, "rank": 58 }, { "arena": "models", "category": "website", "elo": 1219, "win_rate": 49.1, "rank": 44 } ], "artificial_analysis": { "intelligence_index": 42.1, "coding_index": 56.2, "agentic_index": 33.7 } }, "reasoning": { "mandatory": false, "supported_efforts": [ "xhigh", "high" ], "default_effort": "high" } }, { "id": "inclusionai/ling-2.6-1t", "canonical_slug": "inclusionai/ling-2.6-1t-20260423", "hugging_face_id": null, "name": "inclusionAI: Ling-2.6-1T", "created": 1776948238, "description": "Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000075", "completion": "0.000000625", "input_cache_read": "0.000000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/inclusionai/ling-2.6-1t-20260423/endpoints" } }, { "id": "tencent/hy3-preview", "canonical_slug": "tencent/hy3-preview-20260421", "hugging_face_id": "tencent/Hy3-preview", "name": "Tencent: Hy3 preview", "created": 1776878150, "description": "Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000018", "completion": "0.0000006", "input_cache_read": "0.00000006" }, "top_provider": { "context_length": 262144, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "seed", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.9, "top_p": 1, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/tencent/hy3-preview-20260421/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 42.2, "coding_index": 58.8, "agentic_index": 31.4 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "low", "none" ], "default_effort": "high" } }, { "id": "xiaomi/mimo-v2.5-pro", "canonical_slug": "xiaomi/mimo-v2.5-pro-20260422", "hugging_face_id": "XiaomiMiMo/MiMo-V2.5-Pro", "name": "Xiaomi: MiMo-V2.5-Pro", "created": 1776874273, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "context_length": 1050000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000435", "completion": "0.00000087", "input_cache_read": "0.0000000036" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/xiaomi/mimo-v2.5-pro-20260422/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1298, "win_rate": 55.8, "rank": 21 }, { "arena": "models", "category": "asciiart", "elo": 1178, "win_rate": 47.4, "rank": 32 }, { "arena": "models", "category": "codecategories", "elo": 1292, "win_rate": 54.6, "rank": 18 }, { "arena": "models", "category": "dataviz", "elo": 1286, "win_rate": 51.8, "rank": 16 }, { "arena": "models", "category": "gamedev", "elo": 1303, "win_rate": 55.1, "rank": 17 }, { "arena": "models", "category": "svg", "elo": 1216, "win_rate": 51.6, "rank": 24 }, { "arena": "models", "category": "uicomponent", "elo": 1295, "win_rate": 55.8, "rank": 21 }, { "arena": "models", "category": "website", "elo": 1287, "win_rate": 54.5, "rank": 15 } ], "artificial_analysis": { "intelligence_index": 42.9, "coding_index": 60.2, "agentic_index": 29.5 } }, "reasoning": { "mandatory": false } }, { "id": "xiaomi/mimo-v2.5", "canonical_slug": "xiaomi/mimo-v2.5-20260422", "hugging_face_id": "XiaomiMiMo/MiMo-V2.5", "name": "Xiaomi: MiMo-V2.5", "created": 1776874269, "description": "MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...", "context_length": 1050000, "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.0000000028" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/xiaomi/mimo-v2.5-20260422/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1265, "win_rate": 51.7, "rank": 30 }, { "arena": "models", "category": "asciiart", "elo": 1165, "win_rate": 45.5, "rank": 39 }, { "arena": "models", "category": "codecategories", "elo": 1280, "win_rate": 54.8, "rank": 23 }, { "arena": "models", "category": "dataviz", "elo": 1273, "win_rate": 55.2, "rank": 20 }, { "arena": "models", "category": "gamedev", "elo": 1280, "win_rate": 55, "rank": 25 }, { "arena": "models", "category": "svg", "elo": 1209, "win_rate": 52.5, "rank": 26 }, { "arena": "models", "category": "uicomponent", "elo": 1287, "win_rate": 55, "rank": 24 }, { "arena": "models", "category": "website", "elo": 1280, "win_rate": 54.9, "rank": 23 } ], "artificial_analysis": { "intelligence_index": 38, "coding_index": 56.8, "agentic_index": 24.4 } }, "reasoning": { "mandatory": false } }, { "id": "openai/gpt-5.4-image-2", "canonical_slug": "openai/gpt-5.4-image-2-20260421", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 Image 2", "created": 1776797528, "description": "[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...", "context_length": 272000, "architecture": { "modality": "text+image+file->text+image", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "image", "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000008", "completion": "0.000015", "image_output": "0.00003", "web_search": "0.01", "input_cache_read": "0.000002" }, "top_provider": { "context_length": 272000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "top_logprobs" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-image-2-20260421/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "inclusionai/ling-2.6-flash", "canonical_slug": "inclusionai/ling-2.6-flash-20260421", "hugging_face_id": "", "name": "inclusionAI: Ling-2.6-flash", "created": 1776795886, "description": "Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000001", "completion": "0.00000003", "input_cache_read": "0.000000002" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/inclusionai/ling-2.6-flash-20260421/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 14.2, "coding_index": 25.3, "agentic_index": 2.3 } } }, { "id": "~anthropic/claude-opus-latest", "canonical_slug": "~anthropic/claude-opus-latest", "alias_target": { "name": "Claude Opus 5", "slug": "anthropic/claude-opus-5" }, "hugging_face_id": "", "name": "Anthropic: Claude Opus Latest", "created": 1776795361, "description": "This model always redirects to the latest model in the Claude Opus family.", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/~anthropic/claude-opus-latest/endpoints" }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "openrouter/pareto-code", "canonical_slug": "openrouter/pareto-code", "hugging_face_id": "", "name": "Pareto Code Router", "created": 1776747900, "description": "The Pareto Router maintains a tiered shortlist of strong coding models, ranked by [Artificial Analysis](https://artificialanalysis.ai/) coding percentiles. Set min_coding_score between 0 and 1 on the [pareto-router plugin](https://openrouter.ai/docs/guides/routing/routers/pareto-router#the-min_coding_score-parameter) to control how...", "context_length": 2000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "-1", "completion": "-1" }, "top_provider": { "context_length": null, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openrouter/pareto-code/endpoints" } }, { "id": "moonshotai/kimi-k2.6", "canonical_slug": "moonshotai/kimi-k2.6-20260420", "hugging_face_id": "moonshotai/Kimi-K2.6", "name": "MoonshotAI: Kimi K2.6", "created": 1776699402, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "parallel_tool_calls", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k2.6-20260420/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1146, "win_rate": 47.9, "rank": 15 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1248, "win_rate": 59, "rank": 2 }, { "arena": "agents", "category": "agenticslides", "elo": 1187, "win_rate": 45.8, "rank": 5 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1252, "win_rate": 59.2, "rank": 2 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1186, "win_rate": 45.5, "rank": 5 }, { "arena": "agents", "category": "androidnative", "elo": 1197, "win_rate": 51.4, "rank": 19 }, { "arena": "agents", "category": "fullstack", "elo": 1187, "win_rate": 53.7, "rank": 21 }, { "arena": "agents", "category": "godotgamedev", "elo": 1161, "win_rate": 47.1, "rank": 15 }, { "arena": "agents", "category": "htmlslides", "elo": 1221, "win_rate": 53.4, "rank": 7 }, { "arena": "agents", "category": "mobileapps", "elo": 1216, "win_rate": 53.5, "rank": 13 }, { "arena": "agents", "category": "pptxslides", "elo": 1181, "win_rate": 44.3, "rank": 5 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1180, "win_rate": 42.1, "rank": 15 }, { "arena": "agents", "category": "webapps", "elo": 1268, "win_rate": 59.3, "rank": 7 }, { "arena": "models", "category": "3d", "elo": 1320, "win_rate": 57.8, "rank": 13 }, { "arena": "models", "category": "asciiart", "elo": 1186, "win_rate": 46.4, "rank": 26 }, { "arena": "models", "category": "codecategories", "elo": 1294, "win_rate": 55.2, "rank": 17 }, { "arena": "models", "category": "dataviz", "elo": 1271, "win_rate": 52.1, "rank": 21 }, { "arena": "models", "category": "gamedev", "elo": 1291, "win_rate": 55.2, "rank": 22 }, { "arena": "models", "category": "svg", "elo": 1217, "win_rate": 51.8, "rank": 23 }, { "arena": "models", "category": "uicomponent", "elo": 1296, "win_rate": 56, "rank": 20 }, { "arena": "models", "category": "website", "elo": 1284, "win_rate": 54.9, "rank": 19 } ], "artificial_analysis": { "intelligence_index": 45.1, "coding_index": 61.8, "agentic_index": 31.2 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "anthropic/claude-opus-4.7", "canonical_slug": "anthropic/claude-4.7-opus-20260416", "hugging_face_id": null, "name": "Anthropic: Claude Opus 4.7", "created": 1776351100, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.7-opus-20260416/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1250, "win_rate": 61.4, "rank": 3 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1243, "win_rate": 58, "rank": 3 }, { "arena": "agents", "category": "agenticslides", "elo": 1334, "win_rate": 64.7, "rank": 1 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1242, "win_rate": 57.8, "rank": 3 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1331, "win_rate": 65.2, "rank": 1 }, { "arena": "agents", "category": "androidnative", "elo": 1263, "win_rate": 56, "rank": 4 }, { "arena": "agents", "category": "fullstack", "elo": 1503, "win_rate": 80.1, "rank": 1 }, { "arena": "agents", "category": "godotgamedev", "elo": 1270, "win_rate": 60.9, "rank": 3 }, { "arena": "agents", "category": "htmlslides", "elo": 1253, "win_rate": 58.3, "rank": 2 }, { "arena": "agents", "category": "pptxslides", "elo": 1333, "win_rate": 64.9, "rank": 1 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1336, "win_rate": 63.1, "rank": 2 }, { "arena": "agents", "category": "webapps", "elo": 1296, "win_rate": 62.1, "rank": 2 }, { "arena": "models", "category": "3d", "elo": 1302, "win_rate": 56.8, "rank": 19 }, { "arena": "models", "category": "asciiart", "elo": 1316, "win_rate": 65.5, "rank": 4 }, { "arena": "models", "category": "codecategories", "elo": 1311, "win_rate": 58, "rank": 10 }, { "arena": "models", "category": "dataviz", "elo": 1311, "win_rate": 56.3, "rank": 10 }, { "arena": "models", "category": "gamedev", "elo": 1330, "win_rate": 59.2, "rank": 10 }, { "arena": "models", "category": "svg", "elo": 1261, "win_rate": 58.8, "rank": 8 }, { "arena": "models", "category": "uicomponent", "elo": 1332, "win_rate": 59.4, "rank": 8 }, { "arena": "models", "category": "website", "elo": 1307, "win_rate": 58.3, "rank": 9 } ], "artificial_analysis": { "intelligence_index": 55, "coding_index": 73.6, "agentic_index": 46.3 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-opus-4.7:batch", "canonical_slug": "anthropic/claude-4.7-opus-20260416", "hugging_face_id": null, "name": "Anthropic: Claude Opus 4.7 (batch)", "created": 1776351100, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.7-opus-20260416/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1250, "win_rate": 61.4, "rank": 3 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1243, "win_rate": 58, "rank": 3 }, { "arena": "agents", "category": "agenticslides", "elo": 1334, "win_rate": 64.7, "rank": 1 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1242, "win_rate": 57.8, "rank": 3 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1331, "win_rate": 65.2, "rank": 1 }, { "arena": "agents", "category": "androidnative", "elo": 1263, "win_rate": 56, "rank": 4 }, { "arena": "agents", "category": "fullstack", "elo": 1503, "win_rate": 80.1, "rank": 1 }, { "arena": "agents", "category": "godotgamedev", "elo": 1270, "win_rate": 60.9, "rank": 3 }, { "arena": "agents", "category": "htmlslides", "elo": 1253, "win_rate": 58.3, "rank": 2 }, { "arena": "agents", "category": "pptxslides", "elo": 1333, "win_rate": 64.9, "rank": 1 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1336, "win_rate": 63.1, "rank": 2 }, { "arena": "agents", "category": "webapps", "elo": 1296, "win_rate": 62.1, "rank": 2 }, { "arena": "models", "category": "3d", "elo": 1302, "win_rate": 56.8, "rank": 19 }, { "arena": "models", "category": "asciiart", "elo": 1316, "win_rate": 65.5, "rank": 4 }, { "arena": "models", "category": "codecategories", "elo": 1311, "win_rate": 58, "rank": 10 }, { "arena": "models", "category": "dataviz", "elo": 1311, "win_rate": 56.3, "rank": 10 }, { "arena": "models", "category": "gamedev", "elo": 1330, "win_rate": 59.2, "rank": 10 }, { "arena": "models", "category": "svg", "elo": 1261, "win_rate": 58.8, "rank": 8 }, { "arena": "models", "category": "uicomponent", "elo": 1332, "win_rate": 59.4, "rank": 8 }, { "arena": "models", "category": "website", "elo": 1307, "win_rate": 58.3, "rank": 9 } ], "artificial_analysis": { "intelligence_index": 55, "coding_index": 73.6, "agentic_index": 46.3 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "max", "xhigh", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "z-ai/glm-5.1", "canonical_slug": "z-ai/glm-5.1-20260406", "hugging_face_id": "zai-org/GLM-5.1", "name": "Z.ai: GLM 5.1", "created": 1775578025, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000966", "completion": "0.000003036", "input_cache_read": "0.0000001794" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-5.1-20260406/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1336, "win_rate": 62.6, "rank": 8 }, { "arena": "models", "category": "asciiart", "elo": 1165, "win_rate": 45.7, "rank": 38 }, { "arena": "models", "category": "codecategories", "elo": 1288, "win_rate": 54.5, "rank": 19 }, { "arena": "models", "category": "dataviz", "elo": 1366, "win_rate": 67, "rank": 2 }, { "arena": "models", "category": "gamedev", "elo": 1299, "win_rate": 56.3, "rank": 19 }, { "arena": "models", "category": "svg", "elo": 1254, "win_rate": 58.8, "rank": 12 }, { "arena": "models", "category": "uicomponent", "elo": 1299, "win_rate": 54.1, "rank": 18 }, { "arena": "models", "category": "website", "elo": 1290, "win_rate": 55.1, "rank": 14 }, { "arena": "agents", "category": "agenticgamedev", "elo": 1173, "win_rate": 50.7, "rank": 14 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1205, "win_rate": 52, "rank": 6 }, { "arena": "agents", "category": "agenticslides", "elo": 1245, "win_rate": 54.4, "rank": 3 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1204, "win_rate": 51.8, "rank": 6 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1240, "win_rate": 53.3, "rank": 4 }, { "arena": "agents", "category": "androidnative", "elo": 1203, "win_rate": 51.3, "rank": 16 }, { "arena": "agents", "category": "fullstack", "elo": 1197, "win_rate": 55.1, "rank": 19 }, { "arena": "agents", "category": "godotgamedev", "elo": 1104, "win_rate": 39.5, "rank": 25 }, { "arena": "agents", "category": "htmlslides", "elo": 1197, "win_rate": 48.8, "rank": 11 }, { "arena": "agents", "category": "mobileapps", "elo": 1208, "win_rate": 53.4, "rank": 15 }, { "arena": "agents", "category": "pptxslides", "elo": 1241, "win_rate": 53.5, "rank": 4 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1258, "win_rate": 54.2, "rank": 5 }, { "arena": "agents", "category": "webapps", "elo": 1214, "win_rate": 52.6, "rank": 19 } ], "artificial_analysis": { "intelligence_index": 41, "coding_index": 55.8, "agentic_index": 30.6 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "google/gemma-4-26b-a4b-it", "canonical_slug": "google/gemma-4-26b-a4b-it-20260403", "hugging_face_id": "google/gemma-4-26B-A4B-it", "name": "Google: Gemma 4 26B A4B ", "created": 1775227989, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "pricing": { "prompt": "0.00000007", "completion": "0.00000034" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 64 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 26.1, "coding_index": 39.3, "agentic_index": 11 } }, "reasoning": { "mandatory": false, "default_enabled": false } }, { "id": "google/gemma-4-26b-a4b-it:free", "canonical_slug": "google/gemma-4-26b-a4b-it-20260403", "hugging_face_id": "google/gemma-4-26B-A4B-it", "name": "Google: Gemma 4 26B A4B (free)", "created": 1775227989, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 64 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 26.1, "coding_index": 39.3, "agentic_index": 11 } }, "reasoning": { "mandatory": false, "default_enabled": false } }, { "id": "google/gemma-4-31b-it", "canonical_slug": "google/gemma-4-31b-it-20260402", "hugging_face_id": "google/gemma-4-31B-it", "name": "Google: Gemma 4 31B", "created": 1775148486, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.00000034", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 64, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-4-31b-it-20260402/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 29.7, "coding_index": 43.4, "agentic_index": 14.4 } }, "reasoning": { "mandatory": false, "default_enabled": false } }, { "id": "google/gemma-4-31b-it:free", "canonical_slug": "google/gemma-4-31b-it-20260402", "hugging_face_id": "google/gemma-4-31B-it", "name": "Google: Gemma 4 31B (free)", "created": 1775148486, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 64, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-4-31b-it-20260402/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 29.7, "coding_index": 43.4, "agentic_index": 14.4 } }, "reasoning": { "mandatory": false, "default_enabled": false } }, { "id": "qwen/qwen3.6-plus", "canonical_slug": "qwen/qwen3.6-plus-04-02", "hugging_face_id": "", "name": "Qwen: Qwen3.6 Plus", "created": 1775133557, "description": "Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...", "context_length": 1000000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.000000325", "completion": "0.00000195", "input_cache_write": "0.00000040625", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.0000013", "completion": "0.0000039", "input_cache_write": "0.000001625" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.6-plus-04-02/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1251, "win_rate": 51.4, "rank": 35 }, { "arena": "models", "category": "asciiart", "elo": 1147, "win_rate": 43.1, "rank": 45 }, { "arena": "models", "category": "codecategories", "elo": 1254, "win_rate": 51.7, "rank": 34 }, { "arena": "models", "category": "dataviz", "elo": 1248, "win_rate": 51.3, "rank": 34 }, { "arena": "models", "category": "gamedev", "elo": 1245, "win_rate": 50.5, "rank": 37 }, { "arena": "models", "category": "svg", "elo": 1198, "win_rate": 51.7, "rank": 30 }, { "arena": "models", "category": "uicomponent", "elo": 1266, "win_rate": 52.5, "rank": 32 }, { "arena": "models", "category": "website", "elo": 1252, "win_rate": 52, "rank": 34 } ], "artificial_analysis": { "intelligence_index": 40.5, "coding_index": 54.5, "agentic_index": 29 } }, "reasoning": { "mandatory": false } }, { "id": "z-ai/glm-5v-turbo", "canonical_slug": "z-ai/glm-5v-turbo-20260401", "hugging_face_id": "", "name": "Z.ai: GLM 5V Turbo", "created": 1775061458, "description": "GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...", "context_length": 202752, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000012", "completion": "0.000004", "input_cache_read": "0.00000024" }, "top_provider": { "context_length": 202752, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": "2098-12-31", "links": { "details": "/api/v1/models/z-ai/glm-5v-turbo-20260401/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1102, "win_rate": 41.4, "rank": 18 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1134, "win_rate": 41.7, "rank": 8 }, { "arena": "agents", "category": "agenticslides", "elo": 1171, "win_rate": 51.6, "rank": 6 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1137, "win_rate": 41.8, "rank": 8 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1183, "win_rate": 53.9, "rank": 6 }, { "arena": "agents", "category": "androidnative", "elo": 1267, "win_rate": 54.8, "rank": 3 }, { "arena": "agents", "category": "fullstack", "elo": 1179, "win_rate": 52, "rank": 22 }, { "arena": "agents", "category": "godotgamedev", "elo": 997, "win_rate": 26.8, "rank": 30 }, { "arena": "agents", "category": "htmlslides", "elo": 1151, "win_rate": 43.5, "rank": 18 }, { "arena": "agents", "category": "mobileapps", "elo": 1185, "win_rate": 50.5, "rank": 25 }, { "arena": "agents", "category": "pptxslides", "elo": 1164, "win_rate": 52.1, "rank": 6 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1169, "win_rate": 50.2, "rank": 16 }, { "arena": "agents", "category": "webapps", "elo": 1161, "win_rate": 44.6, "rank": 26 }, { "arena": "models", "category": "3d", "elo": 1260, "win_rate": 53.9, "rank": 33 }, { "arena": "models", "category": "asciiart", "elo": 1139, "win_rate": 43, "rank": 47 }, { "arena": "models", "category": "codecategories", "elo": 1248, "win_rate": 51.3, "rank": 37 }, { "arena": "models", "category": "dataviz", "elo": 1218, "win_rate": 48, "rank": 45 }, { "arena": "models", "category": "gamedev", "elo": 1255, "win_rate": 52.6, "rank": 32 }, { "arena": "models", "category": "svg", "elo": 1186, "win_rate": 50.9, "rank": 34 }, { "arena": "models", "category": "uicomponent", "elo": 1242, "win_rate": 50.3, "rank": 39 }, { "arena": "models", "category": "website", "elo": 1242, "win_rate": 50.3, "rank": 36 } ] }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "arcee-ai/trinity-large-thinking", "canonical_slug": "arcee-ai/trinity-large-thinking", "hugging_face_id": "arcee-ai/Trinity-Large-Thinking", "name": "Arcee AI: Trinity Large Thinking", "created": 1775058318, "description": "Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000022", "completion": "0.00000085", "input_cache_read": "0.00000006" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.3, "top_p": 0.8, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/arcee-ai/trinity-large-thinking/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1132, "win_rate": 41.3, "rank": 74 }, { "arena": "models", "category": "asciiart", "elo": 1073, "win_rate": 37.1, "rank": 54 }, { "arena": "models", "category": "codecategories", "elo": 1136, "win_rate": 40.1, "rank": 80 }, { "arena": "models", "category": "dataviz", "elo": 1119, "win_rate": 39.3, "rank": 84 }, { "arena": "models", "category": "gamedev", "elo": 1117, "win_rate": 38.4, "rank": 85 }, { "arena": "models", "category": "svg", "elo": 1051, "win_rate": 35.2, "rank": 68 }, { "arena": "models", "category": "uicomponent", "elo": 1073, "win_rate": 32.6, "rank": 89 }, { "arena": "models", "category": "website", "elo": 1148, "win_rate": 41.3, "rank": 77 } ], "artificial_analysis": { "intelligence_index": null, "coding_index": 25.8, "agentic_index": null } }, "reasoning": { "mandatory": true } }, { "id": "x-ai/grok-4.20-multi-agent", "canonical_slug": "x-ai/grok-4.20-multi-agent-20260309", "hugging_face_id": "", "name": "SpaceXAI: Grok 4.20 Multi-Agent", "created": 1774979158, "description": "Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...", "context_length": 2000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "top_provider": { "context_length": 2000000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-09-01", "expiration_date": null, "links": { "details": "/api/v1/models/x-ai/grok-4.20-multi-agent-20260309/endpoints" }, "reasoning": { "mandatory": true, "default_enabled": true, "supported_efforts": [ "xhigh", "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "x-ai/grok-4.20", "canonical_slug": "x-ai/grok-4.20-20260309", "hugging_face_id": "", "name": "SpaceXAI: Grok 4.20", "created": 1774979019, "description": "Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...", "context_length": 2000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "top_provider": { "context_length": 2000000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-09-01", "expiration_date": null, "links": { "details": "/api/v1/models/x-ai/grok-4.20-20260309/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1079, "win_rate": 35.1, "rank": 34 }, { "arena": "agents", "category": "fullstack", "elo": 1083, "win_rate": 41, "rank": 30 }, { "arena": "agents", "category": "godotgamedev", "elo": 1089, "win_rate": 38.5, "rank": 26 }, { "arena": "agents", "category": "htmlslides", "elo": 1180, "win_rate": 45.7, "rank": 13 }, { "arena": "agents", "category": "mobileapps", "elo": 1141, "win_rate": 45.4, "rank": 31 }, { "arena": "agents", "category": "webapps", "elo": 1180, "win_rate": 48.6, "rank": 22 }, { "arena": "models", "category": "3d", "elo": 1243, "win_rate": 52.9, "rank": 38 }, { "arena": "models", "category": "asciiart", "elo": 1210, "win_rate": 48.6, "rank": 19 }, { "arena": "models", "category": "codecategories", "elo": 1239, "win_rate": 52.9, "rank": 38 }, { "arena": "models", "category": "dataviz", "elo": 1239, "win_rate": 52.8, "rank": 37 }, { "arena": "models", "category": "gamedev", "elo": 1235, "win_rate": 52.1, "rank": 40 }, { "arena": "models", "category": "svg", "elo": 1201, "win_rate": 53.5, "rank": 29 }, { "arena": "models", "category": "uicomponent", "elo": 1229, "win_rate": 50.3, "rank": 43 }, { "arena": "models", "category": "website", "elo": 1239, "win_rate": 52.9, "rank": 38 } ] }, "reasoning": { "mandatory": false, "default_enabled": false } }, { "id": "google/lyria-3-pro-preview", "canonical_slug": "google/lyria-3-pro-preview-20260330", "hugging_face_id": null, "name": "Google: Lyria 3 Pro Preview", "created": 1774907286, "description": "Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...", "context_length": 1048576, "architecture": { "modality": "text+image->text+audio", "input_modalities": [ "text", "image" ], "output_modalities": [ "text", "audio" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "temperature", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/lyria-3-pro-preview-20260330/endpoints" } }, { "id": "google/lyria-3-clip-preview", "canonical_slug": "google/lyria-3-clip-preview-20260330", "hugging_face_id": null, "name": "Google: Lyria 3 Clip Preview", "created": 1774907255, "description": "30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...", "context_length": 1048576, "architecture": { "modality": "text+image->text+audio", "input_modalities": [ "text", "image" ], "output_modalities": [ "text", "audio" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "temperature", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/lyria-3-clip-preview-20260330/endpoints" } }, { "id": "kwaipilot/kat-coder-pro-v2", "canonical_slug": "kwaipilot/kat-coder-pro-v2-20260327", "hugging_face_id": "", "name": "Kwaipilot: KAT-Coder-Pro V2", "created": 1774649310, "description": "KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 80000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/kwaipilot/kat-coder-pro-v2-20260327/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 59.5, "agentic_index": null } } }, { "id": "rekaai/reka-edge", "canonical_slug": "rekaai/reka-edge-2603", "hugging_face_id": "RekaAI/reka-edge-2603", "name": "Reka Edge", "created": 1774026965, "description": "Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...", "context_length": 16384, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000001" }, "top_provider": { "context_length": 16384, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/rekaai/reka-edge-2603/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "minimax/minimax-m2.7", "canonical_slug": "minimax/minimax-m2.7-20260318", "hugging_face_id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax: MiniMax M2.7", "created": 1773836697, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006" }, "top_provider": { "context_length": 204800, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m2.7-20260318/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1245, "win_rate": 50.9, "rank": 37 }, { "arena": "models", "category": "asciiart", "elo": 1172, "win_rate": 47.7, "rank": 35 }, { "arena": "models", "category": "codecategories", "elo": 1254, "win_rate": 52.5, "rank": 33 }, { "arena": "models", "category": "dataviz", "elo": 1251, "win_rate": 52.6, "rank": 31 }, { "arena": "models", "category": "gamedev", "elo": 1245, "win_rate": 52.1, "rank": 35 }, { "arena": "models", "category": "svg", "elo": 1174, "win_rate": 49.9, "rank": 38 }, { "arena": "models", "category": "uicomponent", "elo": 1239, "win_rate": 49.7, "rank": 40 }, { "arena": "models", "category": "website", "elo": 1258, "win_rate": 53.3, "rank": 32 } ], "artificial_analysis": { "intelligence_index": 38.9, "coding_index": 52.6, "agentic_index": 25.9 } }, "reasoning": { "mandatory": true } }, { "id": "openai/gpt-5.4-nano", "canonical_slug": "openai/gpt-5.4-nano-20260317", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 Nano", "created": 1773748187, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.00000125", "web_search": "0.01", "input_cache_read": "0.00000002" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-nano-20260317/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 39.7, "coding_index": 56.1, "agentic_index": 29.7 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.4-nano:batch", "canonical_slug": "openai/gpt-5.4-nano-20260317", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 Nano (batch)", "created": 1773748187, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.000000625", "web_search": "0.01", "input_cache_read": "0.00000001" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-nano-20260317/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 39.7, "coding_index": 56.1, "agentic_index": 29.7 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.4-mini", "canonical_slug": "openai/gpt-5.4-mini-20260317", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 Mini", "created": 1773748178, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "web_search": "0.01", "input_cache_read": "0.000000075" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-mini-20260317/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 40.9, "coding_index": 56.1, "agentic_index": 31.5 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.4-mini:batch", "canonical_slug": "openai/gpt-5.4-mini-20260317", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 Mini (batch)", "created": 1773748178, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000375", "completion": "0.00000225", "web_search": "0.01", "input_cache_read": "0.0000000375" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-mini-20260317/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 40.9, "coding_index": 56.1, "agentic_index": 31.5 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "mistralai/mistral-small-2603", "canonical_slug": "mistralai/mistral-small-2603", "hugging_face_id": "mistralai/Mistral-Small-4-119B-2603", "name": "Mistral: Mistral Small 4", "created": 1773695685, "description": "Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-small-2603/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 19.7, "coding_index": 26.6, "agentic_index": 4.6 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "high", "none" ], "default_effort": "high" } }, { "id": "z-ai/glm-5-turbo", "canonical_slug": "z-ai/glm-5-turbo-20260315", "hugging_face_id": "", "name": "Z.ai: GLM 5 Turbo", "created": 1773583573, "description": "GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...", "context_length": 202752, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000012", "completion": "0.000004", "input_cache_read": "0.00000024" }, "top_provider": { "context_length": 202752, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": "2098-12-31", "links": { "details": "/api/v1/models/z-ai/glm-5-turbo-20260315/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1303, "win_rate": 57.9, "rank": 18 }, { "arena": "models", "category": "asciiart", "elo": 1185, "win_rate": 49.4, "rank": 27 }, { "arena": "models", "category": "codecategories", "elo": 1286, "win_rate": 55.5, "rank": 20 }, { "arena": "models", "category": "dataviz", "elo": 1285, "win_rate": 57.1, "rank": 18 }, { "arena": "models", "category": "gamedev", "elo": 1290, "win_rate": 54.6, "rank": 23 }, { "arena": "models", "category": "svg", "elo": 1241, "win_rate": 56.5, "rank": 15 }, { "arena": "models", "category": "uicomponent", "elo": 1292, "win_rate": 56.7, "rank": 22 }, { "arena": "models", "category": "website", "elo": 1281, "win_rate": 54.9, "rank": 21 } ] }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "nvidia/nemotron-3-super-120b-a12b", "canonical_slug": "nvidia/nemotron-3-super-120b-a12b-20230311", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", "name": "NVIDIA: Nemotron 3 Super", "created": 1773245239, "description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000085", "completion": "0.0000004" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-super-120b-a12b-20230311/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 25.7, "coding_index": 37.7, "agentic_index": 8.8 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true, "supported_efforts": [ "medium", "low" ], "default_effort": "medium" } }, { "id": "nvidia/nemotron-3-super-120b-a12b:free", "canonical_slug": "nvidia/nemotron-3-super-120b-a12b-20230311", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", "name": "NVIDIA: Nemotron 3 Super (free)", "created": 1773245239, "description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-super-120b-a12b-20230311/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 25.7, "coding_index": 37.7, "agentic_index": 8.8 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supports_max_tokens": true, "supported_efforts": [ "medium", "low" ], "default_effort": "medium" } }, { "id": "bytedance-seed/seed-2.0-lite", "canonical_slug": "bytedance-seed/seed-2.0-lite-20260309", "hugging_face_id": null, "name": "ByteDance Seed: Seed-2.0-Lite", "created": 1773157231, "description": "Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000005", "completion": "0.000004" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/bytedance-seed/seed-2.0-lite-20260309/endpoints" }, "reasoning": { "mandatory": false, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "qwen/qwen3.5-9b", "canonical_slug": "qwen/qwen3.5-9b-20260310", "hugging_face_id": "Qwen/Qwen3.5-9B", "name": "Qwen: Qwen3.5-9B", "created": 1773152396, "description": "Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.00000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-9b-20260310/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 21.8, "coding_index": 28.7, "agentic_index": 7 } }, "reasoning": { "mandatory": false } }, { "id": "openai/gpt-5.4-pro", "canonical_slug": "openai/gpt-5.4-pro-20260305", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 Pro", "created": 1772734366, "description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00003", "completion": "0.00018", "web_search": "0.01", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00006", "completion": "0.00027" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-pro-20260305/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.4-pro:batch", "canonical_slug": "openai/gpt-5.4-pro-20260305", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 Pro (batch)", "created": 1772734366, "description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000015", "completion": "0.00009", "web_search": "0.01", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00003", "completion": "0.000135" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-pro-20260305/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.4", "canonical_slug": "openai/gpt-5.4-20260305", "hugging_face_id": "", "name": "OpenAI: GPT-5.4", "created": 1772734352, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-20260305/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1152, "win_rate": 42.4, "rank": 66 }, { "arena": "models", "category": "asciiart", "elo": 1232, "win_rate": 55.5, "rank": 17 }, { "arena": "models", "category": "codecategories", "elo": 1231, "win_rate": 52.5, "rank": 40 }, { "arena": "models", "category": "dataviz", "elo": 1255, "win_rate": 56.6, "rank": 27 }, { "arena": "models", "category": "gamedev", "elo": 1277, "win_rate": 57.6, "rank": 26 }, { "arena": "models", "category": "svg", "elo": 1231, "win_rate": 57.9, "rank": 18 }, { "arena": "models", "category": "uicomponent", "elo": 1268, "win_rate": 57.5, "rank": 30 }, { "arena": "models", "category": "website", "elo": 1231, "win_rate": 52.5, "rank": 42 }, { "arena": "agents", "category": "androidnative", "elo": 1073, "win_rate": 47.4, "rank": 35 }, { "arena": "agents", "category": "fullstack", "elo": 1049, "win_rate": 40.8, "rank": 35 }, { "arena": "agents", "category": "godotgamedev", "elo": 1135, "win_rate": 46.9, "rank": 22 }, { "arena": "agents", "category": "mobileapps", "elo": 1139, "win_rate": 46, "rank": 33 }, { "arena": "agents", "category": "webapps", "elo": 1097, "win_rate": 39.3, "rank": 32 } ], "artificial_analysis": { "intelligence_index": 53.1, "coding_index": 71.1, "agentic_index": 44.2 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.4:batch", "canonical_slug": "openai/gpt-5.4-20260305", "hugging_face_id": "", "name": "OpenAI: GPT-5.4 (batch)", "created": 1772734352, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "context_length": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "top_provider": { "context_length": 1050000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.4-20260305/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1152, "win_rate": 42.4, "rank": 66 }, { "arena": "models", "category": "asciiart", "elo": 1232, "win_rate": 55.5, "rank": 17 }, { "arena": "models", "category": "codecategories", "elo": 1231, "win_rate": 52.5, "rank": 40 }, { "arena": "models", "category": "dataviz", "elo": 1255, "win_rate": 56.6, "rank": 27 }, { "arena": "models", "category": "gamedev", "elo": 1277, "win_rate": 57.6, "rank": 26 }, { "arena": "models", "category": "svg", "elo": 1231, "win_rate": 57.9, "rank": 18 }, { "arena": "models", "category": "uicomponent", "elo": 1268, "win_rate": 57.5, "rank": 30 }, { "arena": "models", "category": "website", "elo": 1231, "win_rate": 52.5, "rank": 42 }, { "arena": "agents", "category": "androidnative", "elo": 1073, "win_rate": 47.4, "rank": 35 }, { "arena": "agents", "category": "fullstack", "elo": 1049, "win_rate": 40.8, "rank": 35 }, { "arena": "agents", "category": "godotgamedev", "elo": 1135, "win_rate": 46.9, "rank": 22 }, { "arena": "agents", "category": "mobileapps", "elo": 1139, "win_rate": 46, "rank": 33 }, { "arena": "agents", "category": "webapps", "elo": 1097, "win_rate": 39.3, "rank": 32 } ], "artificial_analysis": { "intelligence_index": 53.1, "coding_index": 71.1, "agentic_index": 44.2 } }, "reasoning": { "mandatory": false, "default_enabled": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "inception/mercury-2", "canonical_slug": "inception/mercury-2-20260304", "hugging_face_id": null, "name": "Inception: Mercury 2", "created": 1772636275, "description": "Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.00000075", "input_cache_read": "0.000000025" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 50000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools" ], "default_parameters": { "temperature": 0.75, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/inception/mercury-2-20260304/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1031, "win_rate": 23.4, "rank": 99 }, { "arena": "models", "category": "asciiart", "elo": 1043, "win_rate": 28, "rank": 56 }, { "arena": "models", "category": "codecategories", "elo": 1019, "win_rate": 20.8, "rank": 108 }, { "arena": "models", "category": "dataviz", "elo": 1015, "win_rate": 21.9, "rank": 100 }, { "arena": "models", "category": "gamedev", "elo": 1009, "win_rate": 19.7, "rank": 107 }, { "arena": "models", "category": "svg", "elo": 1008, "win_rate": 24.1, "rank": 77 }, { "arena": "models", "category": "uicomponent", "elo": 995, "win_rate": 18.5, "rank": 100 }, { "arena": "models", "category": "website", "elo": 1016, "win_rate": 20.4, "rank": 110 } ], "artificial_analysis": { "intelligence_index": 21.9, "coding_index": 31.1, "agentic_index": 9.5 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "google/gemini-3.1-flash-lite-preview", "canonical_slug": "google/gemini-3.1-flash-lite-preview-20260303", "hugging_face_id": "", "name": "Google: Gemini 3.1 Flash Lite Preview", "created": 1772512673, "description": "Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "input_cache_read": "0.000000025", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-flash-lite-preview-20260303/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1098, "win_rate": 38.7, "rank": 87 }, { "arena": "models", "category": "asciiart", "elo": 1199, "win_rate": 50.6, "rank": 21 }, { "arena": "models", "category": "codecategories", "elo": 1091, "win_rate": 36.4, "rank": 92 }, { "arena": "models", "category": "dataviz", "elo": 1065, "win_rate": 33.3, "rank": 95 }, { "arena": "models", "category": "gamedev", "elo": 1068, "win_rate": 33.7, "rank": 95 }, { "arena": "models", "category": "svg", "elo": 1088, "win_rate": 42.4, "rank": 58 }, { "arena": "models", "category": "uicomponent", "elo": 1102, "win_rate": 38.2, "rank": 85 }, { "arena": "models", "category": "website", "elo": 1094, "win_rate": 36.5, "rank": 94 } ], "artificial_analysis": { "intelligence_index": 25.6, "coding_index": 34.7, "agentic_index": 6.5 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "minimal" } }, { "id": "bytedance-seed/seed-2.0-mini", "canonical_slug": "bytedance-seed/seed-2.0-mini-20260224", "hugging_face_id": "", "name": "ByteDance Seed: Seed-2.0-Mini", "created": 1772131107, "description": "Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000002", "completion": "0.0000008" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/bytedance-seed/seed-2.0-mini-20260224/endpoints" }, "reasoning": { "mandatory": false, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "google/gemini-3.1-flash-image-preview", "canonical_slug": "google/gemini-3.1-flash-image-preview-20260226", "hugging_face_id": "", "name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)", "created": 1772119558, "description": "Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...", "context_length": 65536, "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image_output": "0.00006", "web_search": "0.014" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-flash-image-preview-20260226/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "graphicdesign", "elo": 1286, "win_rate": 66.3, "rank": 2 }, { "arena": "models", "category": "image", "elo": 1296, "win_rate": 65.1, "rank": 2 }, { "arena": "models", "category": "logo", "elo": 1275, "win_rate": 62.9, "rank": 2 } ] }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "minimal" ], "default_effort": "minimal" } }, { "id": "qwen/qwen3.5-35b-a3b", "canonical_slug": "qwen/qwen3.5-35b-a3b-20260224", "hugging_face_id": "Qwen/Qwen3.5-35B-A3B", "name": "Qwen: Qwen3.5-35B-A3B", "created": 1772053822, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.000000225", "completion": "0.0000018", "input_cache_read": "0.000000225" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-35b-a3b-20260224/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 24.3, "coding_index": 37, "agentic_index": 11.8 } }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3.5-27b", "canonical_slug": "qwen/qwen3.5-27b-20260224", "hugging_face_id": "Qwen/Qwen3.5-27B", "name": "Qwen: Qwen3.5-27B", "created": 1772053810, "description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.000000195", "completion": "0.00000156" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": 0.95, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-27b-20260224/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3.5-122b-a10b", "canonical_slug": "qwen/qwen3.5-122b-a10b-20260224", "hugging_face_id": "Qwen/Qwen3.5-122B-A10B", "name": "Qwen: Qwen3.5-122B-A10B", "created": 1772053789, "description": "The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000026", "completion": "0.00000208" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": 0.95, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-122b-a10b-20260224/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 32.8, "coding_index": 45.7, "agentic_index": 21.3 } }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3.5-flash-02-23", "canonical_slug": "qwen/qwen3.5-flash-20260224", "hugging_face_id": null, "name": "Qwen: Qwen3.5-Flash", "created": 1772053776, "description": "The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...", "context_length": 1000000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.000000065", "completion": "0.00000026" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-flash-20260224/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "google/gemini-3.1-pro-preview-customtools", "canonical_slug": "google/gemini-3.1-pro-preview-customtools-20260219", "hugging_face_id": null, "name": "Google: Gemini 3.1 Pro Preview Custom Tools", "created": 1772045923, "description": "Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "audio", "image", "video", "file" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000018", "audio": "0.000004", "input_audio_cache": "0.0000004", "input_cache_read": "0.0000004" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-pro-preview-customtools-20260219/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.3-codex", "canonical_slug": "openai/gpt-5.3-codex-20260224", "hugging_face_id": "", "name": "OpenAI: GPT-5.3-Codex", "created": 1771959164, "description": "GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.3-codex-20260224/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1084, "win_rate": 35.2, "rank": 33 }, { "arena": "agents", "category": "fullstack", "elo": 1014, "win_rate": 36.4, "rank": 38 }, { "arena": "agents", "category": "godotgamedev", "elo": 1124, "win_rate": 45.1, "rank": 24 }, { "arena": "agents", "category": "mobileapps", "elo": 1105, "win_rate": 41.4, "rank": 36 }, { "arena": "agents", "category": "webapps", "elo": 1073, "win_rate": 36.1, "rank": 34 }, { "arena": "models", "category": "3d", "elo": 1059, "win_rate": 35.3, "rank": 93 }, { "arena": "models", "category": "asciiart", "elo": 1184, "win_rate": 51.2, "rank": 29 }, { "arena": "models", "category": "codecategories", "elo": 1164, "win_rate": 47.3, "rank": 67 }, { "arena": "models", "category": "dataviz", "elo": 1184, "win_rate": 50.6, "rank": 59 }, { "arena": "models", "category": "gamedev", "elo": 1200, "win_rate": 51.3, "rank": 52 }, { "arena": "models", "category": "svg", "elo": 1165, "win_rate": 53.9, "rank": 43 }, { "arena": "models", "category": "uicomponent", "elo": 1171, "win_rate": 48.3, "rank": 65 }, { "arena": "models", "category": "website", "elo": 1175, "win_rate": 48.5, "rank": 66 } ] }, "reasoning": { "mandatory": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "aion-labs/aion-2.0", "canonical_slug": "aion-labs/aion-2.0-20260223", "hugging_face_id": null, "name": "AionLabs: Aion-2.0", "created": 1771881306, "description": "Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000008", "completion": "0.0000016", "input_cache_read": "0.0000002" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/aion-labs/aion-2.0-20260223/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "google/gemini-3.1-pro-preview", "canonical_slug": "google/gemini-3.1-pro-preview-20260219", "hugging_face_id": "", "name": "Google: Gemini 3.1 Pro Preview", "created": 1771509627, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000018", "audio": "0.000004", "input_audio_cache": "0.0000004", "input_cache_read": "0.0000004" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-pro-preview-20260219/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1120, "win_rate": 44, "rank": 17 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1226, "win_rate": 55.8, "rank": 5 }, { "arena": "agents", "category": "agenticslides", "elo": 1112, "win_rate": 33.8, "rank": 8 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1219, "win_rate": 54.4, "rank": 5 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1107, "win_rate": 33.9, "rank": 8 }, { "arena": "agents", "category": "androidnative", "elo": 1091, "win_rate": 41.6, "rank": 32 }, { "arena": "agents", "category": "fullstack", "elo": 1099, "win_rate": 42.4, "rank": 26 }, { "arena": "agents", "category": "godotgamedev", "elo": 1236, "win_rate": 60, "rank": 6 }, { "arena": "agents", "category": "htmlslides", "elo": 1203, "win_rate": 51.4, "rank": 9 }, { "arena": "agents", "category": "mobileapps", "elo": 1142, "win_rate": 44.4, "rank": 29 }, { "arena": "agents", "category": "pptxslides", "elo": 1110, "win_rate": 34.1, "rank": 8 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1109, "win_rate": 31.9, "rank": 20 }, { "arena": "agents", "category": "webapps", "elo": 1169, "win_rate": 46, "rank": 23 }, { "arena": "models", "category": "3d", "elo": 1284, "win_rate": 59.3, "rank": 26 }, { "arena": "models", "category": "asciiart", "elo": 1304, "win_rate": 63.7, "rank": 5 }, { "arena": "models", "category": "codecategories", "elo": 1264, "win_rate": 64.4, "rank": 29 }, { "arena": "models", "category": "dataviz", "elo": 1259, "win_rate": 61.9, "rank": 25 }, { "arena": "models", "category": "gamedev", "elo": 1240, "win_rate": 53.5, "rank": 38 }, { "arena": "models", "category": "svg", "elo": 1322, "win_rate": 67.9, "rank": 4 }, { "arena": "models", "category": "uicomponent", "elo": 1306, "win_rate": 69.3, "rank": 15 }, { "arena": "models", "category": "website", "elo": 1266, "win_rate": 64.4, "rank": 27 } ], "artificial_analysis": { "intelligence_index": 47.7, "coding_index": 68.8, "agentic_index": 23 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "google/gemini-3.1-pro-preview:batch", "canonical_slug": "google/gemini-3.1-pro-preview-20260219", "hugging_face_id": "", "name": "Google: Gemini 3.1 Pro Preview (batch)", "created": 1771509627, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000006", "image": "0.000001", "audio": "0.000001", "web_search": "0.014", "internal_reasoning": "0.000006", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000002", "completion": "0.000009", "audio": "0.000002" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3.1-pro-preview-20260219/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1120, "win_rate": 44, "rank": 17 }, { "arena": "agents", "category": "agentichtmlslides", "elo": 1226, "win_rate": 55.8, "rank": 5 }, { "arena": "agents", "category": "agenticslides", "elo": 1112, "win_rate": 33.8, "rank": 8 }, { "arena": "agents", "category": "agenticslides(html)", "elo": 1219, "win_rate": 54.4, "rank": 5 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1107, "win_rate": 33.9, "rank": 8 }, { "arena": "agents", "category": "androidnative", "elo": 1091, "win_rate": 41.6, "rank": 32 }, { "arena": "agents", "category": "fullstack", "elo": 1099, "win_rate": 42.4, "rank": 26 }, { "arena": "agents", "category": "godotgamedev", "elo": 1236, "win_rate": 60, "rank": 6 }, { "arena": "agents", "category": "htmlslides", "elo": 1203, "win_rate": 51.4, "rank": 9 }, { "arena": "agents", "category": "mobileapps", "elo": 1142, "win_rate": 44.4, "rank": 29 }, { "arena": "agents", "category": "pptxslides", "elo": 1110, "win_rate": 34.1, "rank": 8 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1109, "win_rate": 31.9, "rank": 20 }, { "arena": "agents", "category": "webapps", "elo": 1169, "win_rate": 46, "rank": 23 }, { "arena": "models", "category": "3d", "elo": 1284, "win_rate": 59.3, "rank": 26 }, { "arena": "models", "category": "asciiart", "elo": 1304, "win_rate": 63.7, "rank": 5 }, { "arena": "models", "category": "codecategories", "elo": 1264, "win_rate": 64.4, "rank": 29 }, { "arena": "models", "category": "dataviz", "elo": 1259, "win_rate": 61.9, "rank": 25 }, { "arena": "models", "category": "gamedev", "elo": 1240, "win_rate": 53.5, "rank": 38 }, { "arena": "models", "category": "svg", "elo": 1322, "win_rate": 67.9, "rank": 4 }, { "arena": "models", "category": "uicomponent", "elo": 1306, "win_rate": 69.3, "rank": 15 }, { "arena": "models", "category": "website", "elo": 1266, "win_rate": 64.4, "rank": 27 } ], "artificial_analysis": { "intelligence_index": 47.7, "coding_index": 68.8, "agentic_index": 23 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "anthropic/claude-sonnet-4.6", "canonical_slug": "anthropic/claude-4.6-sonnet-20260217", "hugging_face_id": "", "name": "Anthropic: Claude Sonnet 4.6", "created": 1771342990, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.6-sonnet-20260217/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1185, "win_rate": 52.2, "rank": 9 }, { "arena": "agents", "category": "androidnative", "elo": 1236, "win_rate": 62, "rank": 9 }, { "arena": "agents", "category": "fullstack", "elo": 1243, "win_rate": 64.1, "rank": 11 }, { "arena": "agents", "category": "godotgamedev", "elo": 1226, "win_rate": 60.6, "rank": 7 }, { "arena": "agents", "category": "mobileapps", "elo": 1262, "win_rate": 63.4, "rank": 4 }, { "arena": "agents", "category": "webapps", "elo": 1226, "win_rate": 55.4, "rank": 18 }, { "arena": "models", "category": "3d", "elo": 1285, "win_rate": 57.6, "rank": 25 }, { "arena": "models", "category": "asciiart", "elo": 1266, "win_rate": 60.1, "rank": 11 }, { "arena": "models", "category": "codecategories", "elo": 1296, "win_rate": 59.1, "rank": 15 }, { "arena": "models", "category": "dataviz", "elo": 1300, "win_rate": 57.4, "rank": 13 }, { "arena": "models", "category": "gamedev", "elo": 1298, "win_rate": 58.8, "rank": 20 }, { "arena": "models", "category": "svg", "elo": 1234, "win_rate": 59, "rank": 16 }, { "arena": "models", "category": "uicomponent", "elo": 1298, "win_rate": 57.3, "rank": 19 }, { "arena": "models", "category": "website", "elo": 1297, "win_rate": 59.8, "rank": 13 } ], "artificial_analysis": { "intelligence_index": 48.4, "coding_index": 63, "agentic_index": 42.1 } }, "reasoning": { "mandatory": false, "supported_efforts": [ "max", "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "anthropic/claude-sonnet-4.6:batch", "canonical_slug": "anthropic/claude-4.6-sonnet-20260217", "hugging_face_id": "", "name": "Anthropic: Claude Sonnet 4.6 (batch)", "created": 1771342990, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000015", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.00000015", "input_cache_write": "0.000001875", "input_cache_write_1h": "0.000003" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.6-sonnet-20260217/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticgamedev", "elo": 1185, "win_rate": 52.2, "rank": 9 }, { "arena": "agents", "category": "androidnative", "elo": 1236, "win_rate": 62, "rank": 9 }, { "arena": "agents", "category": "fullstack", "elo": 1243, "win_rate": 64.1, "rank": 11 }, { "arena": "agents", "category": "godotgamedev", "elo": 1226, "win_rate": 60.6, "rank": 7 }, { "arena": "agents", "category": "mobileapps", "elo": 1262, "win_rate": 63.4, "rank": 4 }, { "arena": "agents", "category": "webapps", "elo": 1226, "win_rate": 55.4, "rank": 18 }, { "arena": "models", "category": "3d", "elo": 1285, "win_rate": 57.6, "rank": 25 }, { "arena": "models", "category": "asciiart", "elo": 1266, "win_rate": 60.1, "rank": 11 }, { "arena": "models", "category": "codecategories", "elo": 1296, "win_rate": 59.1, "rank": 15 }, { "arena": "models", "category": "dataviz", "elo": 1300, "win_rate": 57.4, "rank": 13 }, { "arena": "models", "category": "gamedev", "elo": 1298, "win_rate": 58.8, "rank": 20 }, { "arena": "models", "category": "svg", "elo": 1234, "win_rate": 59, "rank": 16 }, { "arena": "models", "category": "uicomponent", "elo": 1298, "win_rate": 57.3, "rank": 19 }, { "arena": "models", "category": "website", "elo": 1297, "win_rate": 59.8, "rank": 13 } ], "artificial_analysis": { "intelligence_index": 48.4, "coding_index": 63, "agentic_index": 42.1 } }, "reasoning": { "mandatory": false, "supported_efforts": [ "max", "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "qwen/qwen3.5-plus-02-15", "canonical_slug": "qwen/qwen3.5-plus-20260216", "hugging_face_id": "", "name": "Qwen: Qwen3.5 Plus 2026-02-15", "created": 1771229416, "description": "The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...", "context_length": 1000000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000026", "completion": "0.00000156", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.000000325", "completion": "0.00000195" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-plus-20260216/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1170, "win_rate": 47.7, "rank": 63 }, { "arena": "models", "category": "asciiart", "elo": 1126, "win_rate": 43.2, "rank": 50 }, { "arena": "models", "category": "codecategories", "elo": 1186, "win_rate": 48.5, "rank": 60 }, { "arena": "models", "category": "dataviz", "elo": 1156, "win_rate": 44.9, "rank": 70 }, { "arena": "models", "category": "gamedev", "elo": 1144, "win_rate": 42.6, "rank": 71 }, { "arena": "models", "category": "svg", "elo": 1140, "win_rate": 48.5, "rank": 48 }, { "arena": "models", "category": "uicomponent", "elo": 1208, "win_rate": 52.7, "rank": 47 }, { "arena": "models", "category": "website", "elo": 1199, "win_rate": 50, "rank": 54 } ] }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3.5-397b-a17b", "canonical_slug": "qwen/qwen3.5-397b-a17b-20260216", "hugging_face_id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen: Qwen3.5 397B A17B", "created": 1771223018, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000039", "completion": "0.00000234" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": 0.95, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3.5-397b-a17b-20260216/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1211, "win_rate": 56.6, "rank": 47 }, { "arena": "models", "category": "codecategories", "elo": 1200, "win_rate": 52.7, "rank": 50 }, { "arena": "models", "category": "dataviz", "elo": 1196, "win_rate": 53.2, "rank": 52 }, { "arena": "models", "category": "gamedev", "elo": 1179, "win_rate": 50.2, "rank": 59 }, { "arena": "models", "category": "svg", "elo": 1171, "win_rate": 55.2, "rank": 41 }, { "arena": "models", "category": "uicomponent", "elo": 1194, "win_rate": 52.2, "rank": 57 }, { "arena": "models", "category": "website", "elo": 1203, "win_rate": 52.5, "rank": 51 } ], "artificial_analysis": { "intelligence_index": 34.3, "coding_index": 48.2, "agentic_index": 19.8 } }, "reasoning": { "mandatory": false } }, { "id": "minimax/minimax-m2.5", "canonical_slug": "minimax/minimax-m2.5-20260211", "hugging_face_id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax: MiniMax M2.5", "created": 1770908502, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000022", "completion": "0.0000009", "input_cache_read": "0.00000005" }, "top_provider": { "context_length": 196608, "max_completion_tokens": 196608, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "parallel_tool_calls", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m2.5-20260211/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1220, "win_rate": 57.5, "rank": 44 }, { "arena": "models", "category": "codecategories", "elo": 1226, "win_rate": 56.7, "rank": 42 }, { "arena": "models", "category": "dataviz", "elo": 1192, "win_rate": 51, "rank": 53 }, { "arena": "models", "category": "gamedev", "elo": 1216, "win_rate": 55.6, "rank": 45 }, { "arena": "models", "category": "svg", "elo": 1186, "win_rate": 54.6, "rank": 33 }, { "arena": "models", "category": "uicomponent", "elo": 1200, "win_rate": 53.3, "rank": 52 }, { "arena": "models", "category": "website", "elo": 1234, "win_rate": 57.4, "rank": 40 } ] }, "reasoning": { "mandatory": true } }, { "id": "z-ai/glm-5", "canonical_slug": "z-ai/glm-5-20260211", "hugging_face_id": "zai-org/GLM-5", "name": "Z.ai: GLM 5", "created": 1770829182, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.00000192", "input_cache_read": "0.00000012" }, "top_provider": { "context_length": 198000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-5-20260211/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1182, "win_rate": 55.2, "rank": 21 }, { "arena": "agents", "category": "fullstack", "elo": 1152, "win_rate": 51.6, "rank": 23 }, { "arena": "agents", "category": "godotgamedev", "elo": 1144, "win_rate": 46.7, "rank": 16 }, { "arena": "agents", "category": "htmlslides", "elo": 1167, "win_rate": 45, "rank": 16 }, { "arena": "agents", "category": "mobileapps", "elo": 1188, "win_rate": 51.8, "rank": 23 }, { "arena": "models", "category": "3d", "elo": 1279, "win_rate": 56.3, "rank": 27 }, { "arena": "models", "category": "asciiart", "elo": 1181, "win_rate": 48, "rank": 31 }, { "arena": "models", "category": "codecategories", "elo": 1266, "win_rate": 55.5, "rank": 28 }, { "arena": "models", "category": "dataviz", "elo": 1249, "win_rate": 53, "rank": 32 }, { "arena": "models", "category": "gamedev", "elo": 1273, "win_rate": 57.4, "rank": 27 }, { "arena": "models", "category": "svg", "elo": 1204, "win_rate": 54.4, "rank": 28 }, { "arena": "models", "category": "uicomponent", "elo": 1261, "win_rate": 54, "rank": 35 }, { "arena": "models", "category": "website", "elo": 1261, "win_rate": 55, "rank": 29 } ] }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "qwen/qwen3-max-thinking", "canonical_slug": "qwen/qwen3-max-thinking-20260123", "hugging_face_id": null, "name": "Qwen: Qwen3 Max Thinking", "created": 1770671901, "description": "Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.00000078", "completion": "0.0000039", "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.00000156", "completion": "0.0000078" }, { "min_prompt_tokens": 128000, "prompt": "0.00000195", "completion": "0.00000975" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-max-thinking-20260123/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "anthropic/claude-opus-4.6", "canonical_slug": "anthropic/claude-4.6-opus-20260205", "hugging_face_id": "", "name": "Anthropic: Claude Opus 4.6", "created": 1770219050, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.6-opus-20260205/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1247, "win_rate": 65.9, "rank": 7 }, { "arena": "agents", "category": "fullstack", "elo": 1249, "win_rate": 63.1, "rank": 10 }, { "arena": "agents", "category": "mobileapps", "elo": 1264, "win_rate": 62.8, "rank": 3 }, { "arena": "agents", "category": "webapps", "elo": 1232, "win_rate": 55.9, "rank": 17 }, { "arena": "models", "category": "3d", "elo": 1326, "win_rate": 62.4, "rank": 11 }, { "arena": "models", "category": "asciiart", "elo": 1285, "win_rate": 61.7, "rank": 10 }, { "arena": "models", "category": "codecategories", "elo": 1309, "win_rate": 61, "rank": 11 }, { "arena": "models", "category": "dataviz", "elo": 1303, "win_rate": 58.1, "rank": 11 }, { "arena": "models", "category": "gamedev", "elo": 1315, "win_rate": 59.3, "rank": 13 }, { "arena": "models", "category": "svg", "elo": 1265, "win_rate": 60.6, "rank": 7 }, { "arena": "models", "category": "uicomponent", "elo": 1312, "win_rate": 59.1, "rank": 11 }, { "arena": "models", "category": "website", "elo": 1304, "win_rate": 61.1, "rank": 10 } ] }, "reasoning": { "mandatory": false, "default_enabled": false, "supports_max_tokens": true, "supported_efforts": [ "max", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "anthropic/claude-opus-4.6:batch", "canonical_slug": "anthropic/claude-4.6-opus-20260205", "hugging_face_id": "", "name": "Anthropic: Claude Opus 4.6 (batch)", "created": 1770219050, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.6-opus-20260205/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1247, "win_rate": 65.9, "rank": 7 }, { "arena": "agents", "category": "fullstack", "elo": 1249, "win_rate": 63.1, "rank": 10 }, { "arena": "agents", "category": "mobileapps", "elo": 1264, "win_rate": 62.8, "rank": 3 }, { "arena": "agents", "category": "webapps", "elo": 1232, "win_rate": 55.9, "rank": 17 }, { "arena": "models", "category": "3d", "elo": 1326, "win_rate": 62.4, "rank": 11 }, { "arena": "models", "category": "asciiart", "elo": 1285, "win_rate": 61.7, "rank": 10 }, { "arena": "models", "category": "codecategories", "elo": 1309, "win_rate": 61, "rank": 11 }, { "arena": "models", "category": "dataviz", "elo": 1303, "win_rate": 58.1, "rank": 11 }, { "arena": "models", "category": "gamedev", "elo": 1315, "win_rate": 59.3, "rank": 13 }, { "arena": "models", "category": "svg", "elo": 1265, "win_rate": 60.6, "rank": 7 }, { "arena": "models", "category": "uicomponent", "elo": 1312, "win_rate": 59.1, "rank": 11 }, { "arena": "models", "category": "website", "elo": 1304, "win_rate": 61.1, "rank": 10 } ] }, "reasoning": { "mandatory": false, "default_enabled": false, "supports_max_tokens": true, "supported_efforts": [ "max", "high", "medium", "low" ], "default_effort": "high" } }, { "id": "qwen/qwen3-coder-next", "canonical_slug": "qwen/qwen3-coder-next-2025-02-03", "hugging_face_id": "Qwen/Qwen3-Coder-Next", "name": "Qwen: Qwen3 Coder Next", "created": 1770164101, "description": "Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.00000012", "completion": "0.0000008", "input_cache_read": "0.00000007" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-coder-next-2025-02-03/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 21.3, "coding_index": 36.2, "agentic_index": 8.9 } } }, { "id": "openrouter/free", "canonical_slug": "openrouter/free", "hugging_face_id": "", "name": "Free Models Router", "created": 1769917427, "description": "The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...", "context_length": 200000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": null, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openrouter/free/endpoints" } }, { "id": "stepfun/step-3.5-flash", "canonical_slug": "stepfun/step-3.5-flash", "hugging_face_id": "stepfun-ai/Step-3.5-Flash", "name": "StepFun: Step 3.5 Flash", "created": 1769728337, "description": "Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000003" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/stepfun/step-3.5-flash/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "moonshotai/kimi-k2.5", "canonical_slug": "moonshotai/kimi-k2.5-0127", "hugging_face_id": "moonshotai/Kimi-K2.5", "name": "MoonshotAI: Kimi K2.5", "created": 1769487076, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000057", "completion": "0.00000285", "input_cache_read": "0.000000095" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k2.5-0127/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1161, "win_rate": 57.9, "rank": 26 }, { "arena": "agents", "category": "fullstack", "elo": 1145, "win_rate": 54.2, "rank": 24 }, { "arena": "agents", "category": "godotgamedev", "elo": 1219, "win_rate": 59.8, "rank": 9 }, { "arena": "agents", "category": "mobileapps", "elo": 1196, "win_rate": 54.2, "rank": 21 }, { "arena": "agents", "category": "webapps", "elo": 1166, "win_rate": 50.1, "rank": 25 }, { "arena": "models", "category": "3d", "elo": 1256, "win_rate": 53.1, "rank": 34 }, { "arena": "models", "category": "asciiart", "elo": 1202, "win_rate": 46.5, "rank": 20 }, { "arena": "models", "category": "codecategories", "elo": 1256, "win_rate": 54.1, "rank": 32 }, { "arena": "models", "category": "dataviz", "elo": 1239, "win_rate": 51.3, "rank": 38 }, { "arena": "models", "category": "gamedev", "elo": 1245, "win_rate": 53.5, "rank": 36 }, { "arena": "models", "category": "svg", "elo": 1184, "win_rate": 48.3, "rank": 36 }, { "arena": "models", "category": "uicomponent", "elo": 1266, "win_rate": 53.7, "rank": 31 }, { "arena": "models", "category": "website", "elo": 1261, "win_rate": 55.3, "rank": 30 } ], "artificial_analysis": { "intelligence_index": 36, "coding_index": 46.8, "agentic_index": 21.7 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "upstage/solar-pro-3", "canonical_slug": "upstage/solar-pro-3", "hugging_face_id": "", "name": "Upstage: Solar Pro 3", "created": 1769481200, "description": "Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000015" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "response_format", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/upstage/solar-pro-3/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 14.5, "coding_index": 16.2, "agentic_index": 2.9 } }, "reasoning": { "mandatory": false } }, { "id": "minimax/minimax-m2-her", "canonical_slug": "minimax/minimax-m2-her-20260123", "hugging_face_id": "", "name": "MiniMax: MiniMax M2-her", "created": 1769177239, "description": "MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...", "context_length": 65536, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 2048, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m2-her-20260123/endpoints" } }, { "id": "writer/palmyra-x5", "canonical_slug": "writer/palmyra-x5-20250428", "hugging_face_id": "", "name": "Writer: Palmyra X5", "created": 1769003823, "description": "Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...", "context_length": 1040000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.000006" }, "top_provider": { "context_length": 1040000, "max_completion_tokens": 8192, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "stop", "temperature", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/writer/palmyra-x5-20250428/endpoints" } }, { "id": "openai/gpt-audio", "canonical_slug": "openai/gpt-audio", "hugging_face_id": "", "name": "OpenAI: GPT Audio", "created": 1768862569, "description": "The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...", "context_length": 128000, "architecture": { "modality": "text+audio->text+audio", "input_modalities": [ "text", "audio" ], "output_modalities": [ "text", "audio" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "audio": "0.000032", "audio_output": "0.000064" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-audio/endpoints" } }, { "id": "openai/gpt-audio-mini", "canonical_slug": "openai/gpt-audio-mini", "hugging_face_id": "", "name": "OpenAI: GPT Audio Mini", "created": 1768859419, "description": "A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...", "context_length": 128000, "architecture": { "modality": "text+audio->text+audio", "input_modalities": [ "text", "audio" ], "output_modalities": [ "text", "audio" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.0000024", "audio": "0.0000006", "audio_output": "0.0000024" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-audio-mini/endpoints" } }, { "id": "z-ai/glm-4.7-flash", "canonical_slug": "z-ai/glm-4.7-flash-20260119", "hugging_face_id": "zai-org/GLM-4.7-Flash", "name": "Z.ai: GLM 4.7 Flash", "created": 1768833913, "description": "As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...", "context_length": 202752, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000006", "completion": "0.0000004", "input_cache_read": "0.00000001" }, "top_provider": { "context_length": 202752, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-4.7-flash-20260119/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1172, "win_rate": 51.4, "rank": 62 }, { "arena": "models", "category": "codecategories", "elo": 1197, "win_rate": 53.1, "rank": 51 }, { "arena": "models", "category": "dataviz", "elo": 1144, "win_rate": 45.6, "rank": 77 }, { "arena": "models", "category": "gamedev", "elo": 1173, "win_rate": 50, "rank": 63 }, { "arena": "models", "category": "svg", "elo": 1065, "win_rate": 42.4, "rank": 63 }, { "arena": "models", "category": "uicomponent", "elo": 1234, "win_rate": 57.8, "rank": 41 }, { "arena": "models", "category": "website", "elo": 1205, "win_rate": 54, "rank": 49 } ] }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "openai/gpt-5.2-codex", "canonical_slug": "openai/gpt-5.2-codex-20260114", "hugging_face_id": "", "name": "OpenAI: GPT-5.2-Codex", "created": 1768409315, "description": "GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....", "context_length": 400000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.2-codex-20260114/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1176, "win_rate": 47.5, "rank": 23 }, { "arena": "agents", "category": "fullstack", "elo": 1023, "win_rate": 37, "rank": 37 }, { "arena": "agents", "category": "godotgamedev", "elo": 1142, "win_rate": 47.8, "rank": 20 }, { "arena": "agents", "category": "mobileapps", "elo": 1140, "win_rate": 46.4, "rank": 32 }, { "arena": "agents", "category": "webapps", "elo": 1088, "win_rate": 39.6, "rank": 33 } ] }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "bytedance-seed/seed-1.6-flash", "canonical_slug": "bytedance-seed/seed-1.6-flash-20250625", "hugging_face_id": "", "name": "ByteDance Seed: Seed 1.6 Flash", "created": 1766505011, "description": "Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000075", "completion": "0.0000003", "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000001", "completion": "0.0000008" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/bytedance-seed/seed-1.6-flash-20250625/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "bytedance-seed/seed-1.6", "canonical_slug": "bytedance-seed/seed-1.6-20250625", "hugging_face_id": "", "name": "ByteDance Seed: Seed 1.6", "created": 1766504997, "description": "Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.", "context_length": 262144, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000005", "completion": "0.000004" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/bytedance-seed/seed-1.6-20250625/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "minimax/minimax-m2.1", "canonical_slug": "minimax/minimax-m2.1", "hugging_face_id": "MiniMaxAI/MiniMax-M2.1", "name": "MiniMax: MiniMax M2.1", "created": 1766454997, "description": "MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 204800, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.9, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m2.1/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1216, "win_rate": 57.6, "rank": 45 }, { "arena": "models", "category": "codecategories", "elo": 1210, "win_rate": 55.3, "rank": 45 }, { "arena": "models", "category": "dataviz", "elo": 1227, "win_rate": 56.5, "rank": 41 }, { "arena": "models", "category": "gamedev", "elo": 1173, "win_rate": 50.5, "rank": 64 }, { "arena": "models", "category": "svg", "elo": 1166, "win_rate": 54.2, "rank": 42 }, { "arena": "models", "category": "uicomponent", "elo": 1252, "win_rate": 61.2, "rank": 36 }, { "arena": "models", "category": "website", "elo": 1213, "win_rate": 55.4, "rank": 45 } ] }, "reasoning": { "mandatory": true } }, { "id": "z-ai/glm-4.7", "canonical_slug": "z-ai/glm-4.7-20251222", "hugging_face_id": "zai-org/GLM-4.7", "name": "Z.ai: GLM 4.7", "created": 1766378014, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000004", "completion": "0.00000175", "input_cache_read": "0.00000008" }, "top_provider": { "context_length": 202752, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-4.7-20251222/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1155, "win_rate": 56.2, "rank": 27 }, { "arena": "agents", "category": "fullstack", "elo": 1080, "win_rate": 44.9, "rank": 31 }, { "arena": "agents", "category": "godotgamedev", "elo": 1058, "win_rate": 35.8, "rank": 28 }, { "arena": "agents", "category": "mobileapps", "elo": 1158, "win_rate": 49, "rank": 26 }, { "arena": "models", "category": "3d", "elo": 1241, "win_rate": 54.3, "rank": 39 }, { "arena": "models", "category": "asciiart", "elo": 1198, "win_rate": 48.1, "rank": 23 }, { "arena": "models", "category": "codecategories", "elo": 1237, "win_rate": 54.9, "rank": 39 }, { "arena": "models", "category": "dataviz", "elo": 1216, "win_rate": 51.3, "rank": 46 }, { "arena": "models", "category": "gamedev", "elo": 1227, "win_rate": 55.2, "rank": 42 }, { "arena": "models", "category": "svg", "elo": 1180, "win_rate": 54, "rank": 37 }, { "arena": "models", "category": "uicomponent", "elo": 1229, "win_rate": 51.4, "rank": 42 }, { "arena": "models", "category": "website", "elo": 1239, "win_rate": 55.3, "rank": 37 } ], "artificial_analysis": { "intelligence_index": 34.5, "coding_index": 45.3, "agentic_index": 26.2 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "google/gemini-3-flash-preview", "canonical_slug": "google/gemini-3-flash-preview-20251217", "hugging_face_id": "", "name": "Google: Gemini 3 Flash Preview", "created": 1765987078, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image": "0.0000005", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.000003", "input_cache_read": "0.00000005", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3-flash-preview-20251217/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticslides", "elo": 1073, "win_rate": 39.3, "rank": 9 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1075, "win_rate": 39.3, "rank": 9 }, { "arena": "agents", "category": "androidnative", "elo": 1093, "win_rate": 48.1, "rank": 31 }, { "arena": "agents", "category": "fullstack", "elo": 1093, "win_rate": 47.1, "rank": 27 }, { "arena": "agents", "category": "godotgamedev", "elo": 1161, "win_rate": 50.6, "rank": 14 }, { "arena": "agents", "category": "mobileapps", "elo": 1141, "win_rate": 46.6, "rank": 30 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1035, "win_rate": 38.3, "rank": 22 }, { "arena": "agents", "category": "webapps", "elo": 1158, "win_rate": 49.6, "rank": 27 }, { "arena": "models", "category": "3d", "elo": 1234, "win_rate": 62.7, "rank": 41 }, { "arena": "models", "category": "codecategories", "elo": 1208, "win_rate": 57.6, "rank": 46 }, { "arena": "models", "category": "gamedev", "elo": 1205, "win_rate": 58.3, "rank": 50 }, { "arena": "models", "category": "website", "elo": 1207, "win_rate": 57, "rank": 47 } ] }, "reasoning": { "mandatory": false, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "google/gemini-3-flash-preview:batch", "canonical_slug": "google/gemini-3-flash-preview-20251217", "hugging_face_id": "", "name": "Google: Gemini 3 Flash Preview (batch)", "created": 1765987078, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "web_search": "0.014", "internal_reasoning": "0.0000015" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3-flash-preview-20251217/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "agenticslides", "elo": 1073, "win_rate": 39.3, "rank": 9 }, { "arena": "agents", "category": "agenticslides(python-pptx)", "elo": 1075, "win_rate": 39.3, "rank": 9 }, { "arena": "agents", "category": "androidnative", "elo": 1093, "win_rate": 48.1, "rank": 31 }, { "arena": "agents", "category": "fullstack", "elo": 1093, "win_rate": 47.1, "rank": 27 }, { "arena": "agents", "category": "godotgamedev", "elo": 1161, "win_rate": 50.6, "rank": 14 }, { "arena": "agents", "category": "mobileapps", "elo": 1141, "win_rate": 46.6, "rank": 30 }, { "arena": "agents", "category": "python-pptxslides", "elo": 1035, "win_rate": 38.3, "rank": 22 }, { "arena": "agents", "category": "webapps", "elo": 1158, "win_rate": 49.6, "rank": 27 }, { "arena": "models", "category": "3d", "elo": 1234, "win_rate": 62.7, "rank": 41 }, { "arena": "models", "category": "codecategories", "elo": 1208, "win_rate": 57.6, "rank": 46 }, { "arena": "models", "category": "gamedev", "elo": 1205, "win_rate": 58.3, "rank": 50 }, { "arena": "models", "category": "website", "elo": 1207, "win_rate": 57, "rank": 47 } ] }, "reasoning": { "mandatory": false, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "nvidia/nemotron-3-nano-30b-a3b", "canonical_slug": "nvidia/nemotron-3-nano-30b-a3b", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16", "name": "NVIDIA: Nemotron 3 Nano 30B A3B", "created": 1765731275, "description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-nano-30b-a3b/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 14.5, "coding_index": 14.4, "agentic_index": 2 } }, "reasoning": { "mandatory": false } }, { "id": "nvidia/nemotron-3-nano-30b-a3b:free", "canonical_slug": "nvidia/nemotron-3-nano-30b-a3b", "hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16", "name": "NVIDIA: Nemotron 3 Nano 30B A3B (free)", "created": 1765731275, "description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 256000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "seed", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-3-nano-30b-a3b/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 14.5, "coding_index": 14.4, "agentic_index": 2 } }, "reasoning": { "mandatory": false } }, { "id": "openai/gpt-5.2-chat", "canonical_slug": "openai/gpt-5.2-chat-20251211", "hugging_face_id": "", "name": "OpenAI: GPT-5.2 Chat", "created": 1765389783, "description": "GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_completion_tokens", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.2-chat-20251211/endpoints" } }, { "id": "openai/gpt-5.2-pro", "canonical_slug": "openai/gpt-5.2-pro-20251211", "hugging_face_id": "", "name": "OpenAI: GPT-5.2 Pro", "created": 1765389780, "description": "GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000021", "completion": "0.000168", "web_search": "0.01" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.2-pro-20251211/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.2-pro:batch", "canonical_slug": "openai/gpt-5.2-pro-20251211", "hugging_face_id": "", "name": "OpenAI: GPT-5.2 Pro (batch)", "created": 1765389780, "description": "GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000105", "completion": "0.000084", "web_search": "0.01" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.2-pro-20251211/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.2", "canonical_slug": "openai/gpt-5.2-20251211", "hugging_face_id": "", "name": "OpenAI: GPT-5.2", "created": 1765389775, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.2-20251211/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "website", "elo": 1206, "win_rate": 54.4, "rank": 48 }, { "arena": "agents", "category": "androidnative", "elo": 1101, "win_rate": 49.1, "rank": 30 }, { "arena": "agents", "category": "fullstack", "elo": 1074, "win_rate": 44.1, "rank": 33 }, { "arena": "agents", "category": "godotgamedev", "elo": 1142, "win_rate": 48.1, "rank": 19 }, { "arena": "agents", "category": "mobileapps", "elo": 1143, "win_rate": 46.6, "rank": 28 }, { "arena": "agents", "category": "webapps", "elo": 1113, "win_rate": 43.8, "rank": 29 }, { "arena": "models", "category": "3d", "elo": 1129, "win_rate": 41.5, "rank": 78 }, { "arena": "models", "category": "codecategories", "elo": 1189, "win_rate": 49.5, "rank": 58 }, { "arena": "models", "category": "dataviz", "elo": 1221, "win_rate": 56.1, "rank": 44 }, { "arena": "models", "category": "gamedev", "elo": 1233, "win_rate": 56, "rank": 41 }, { "arena": "models", "category": "uicomponent", "elo": 1216, "win_rate": 51.3, "rank": 45 }, { "arena": "models", "category": "asciiart", "elo": 1185, "win_rate": 50.7, "rank": 28 }, { "arena": "models", "category": "svg", "elo": 1171, "win_rate": 52.6, "rank": 40 } ] }, "reasoning": { "mandatory": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.2:batch", "canonical_slug": "openai/gpt-5.2-20251211", "hugging_face_id": "", "name": "OpenAI: GPT-5.2 (batch)", "created": 1765389775, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000875", "completion": "0.000007", "web_search": "0.01", "input_cache_read": "0.0000000875" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.2-20251211/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "website", "elo": 1206, "win_rate": 54.4, "rank": 48 }, { "arena": "agents", "category": "androidnative", "elo": 1101, "win_rate": 49.1, "rank": 30 }, { "arena": "agents", "category": "fullstack", "elo": 1074, "win_rate": 44.1, "rank": 33 }, { "arena": "agents", "category": "godotgamedev", "elo": 1142, "win_rate": 48.1, "rank": 19 }, { "arena": "agents", "category": "mobileapps", "elo": 1143, "win_rate": 46.6, "rank": 28 }, { "arena": "agents", "category": "webapps", "elo": 1113, "win_rate": 43.8, "rank": 29 }, { "arena": "models", "category": "3d", "elo": 1129, "win_rate": 41.5, "rank": 78 }, { "arena": "models", "category": "codecategories", "elo": 1189, "win_rate": 49.5, "rank": 58 }, { "arena": "models", "category": "dataviz", "elo": 1221, "win_rate": 56.1, "rank": 44 }, { "arena": "models", "category": "gamedev", "elo": 1233, "win_rate": 56, "rank": 41 }, { "arena": "models", "category": "uicomponent", "elo": 1216, "win_rate": 51.3, "rank": 45 }, { "arena": "models", "category": "asciiart", "elo": 1185, "win_rate": 50.7, "rank": 28 }, { "arena": "models", "category": "svg", "elo": 1171, "win_rate": 52.6, "rank": 40 } ] }, "reasoning": { "mandatory": false, "supported_efforts": [ "xhigh", "high", "medium", "low", "none" ], "default_effort": "medium" } }, { "id": "relace/relace-search", "canonical_slug": "relace/relace-search-20251208", "hugging_face_id": null, "name": "Relace: Relace Search", "created": 1765213560, "description": "The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000003" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/relace/relace-search-20251208/endpoints" } }, { "id": "z-ai/glm-4.6v", "canonical_slug": "z-ai/glm-4.6-20251208", "hugging_face_id": "zai-org/GLM-4.6V", "name": "Z.ai: GLM 4.6V", "created": 1765207462, "description": "GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...", "context_length": 131072, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000009", "input_cache_read": "0.000000055" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 0.8, "top_p": 0.6, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-4.6-20251208/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "openrouter/bodybuilder", "canonical_slug": "openrouter/bodybuilder", "hugging_face_id": "", "name": "Body Builder (beta)", "created": 1764903653, "description": "Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "-1", "completion": "-1" }, "top_provider": { "context_length": null, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openrouter/bodybuilder/endpoints" } }, { "id": "openai/gpt-5.1-codex-max", "canonical_slug": "openai/gpt-5.1-codex-max-20251204", "hugging_face_id": "", "name": "OpenAI: GPT-5.1-Codex-Max", "created": 1764878934, "description": "GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...", "context_length": 400000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.1-codex-max-20251204/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "xhigh", "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "amazon/nova-2-lite-v1", "canonical_slug": "amazon/nova-2-lite-v1", "hugging_face_id": "", "name": "Amazon: Nova 2 Lite", "created": 1764696672, "description": "Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...", "context_length": 1000000, "architecture": { "modality": "text+image+file+video->text", "input_modalities": [ "text", "image", "video", "file" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000025" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65535, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/amazon/nova-2-lite-v1/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 23, "agentic_index": null } }, "reasoning": { "mandatory": false } }, { "id": "mistralai/ministral-14b-2512", "canonical_slug": "mistralai/ministral-14b-2512", "hugging_face_id": "mistralai/Ministral-3-14B-Instruct-2512", "name": "Mistral: Ministral 3 14B 2512", "created": 1764681735, "description": "The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000002", "input_cache_read": "0.00000002" }, "top_provider": { "context_length": 262144, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/ministral-14b-2512/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1041, "win_rate": 39.6, "rank": 96 }, { "arena": "models", "category": "codecategories", "elo": 1084, "win_rate": 44, "rank": 94 }, { "arena": "models", "category": "gamedev", "elo": 1077, "win_rate": 43.6, "rank": 93 }, { "arena": "models", "category": "website", "elo": 1092, "win_rate": 44.8, "rank": 95 } ], "artificial_analysis": { "intelligence_index": 11.2, "coding_index": 14.4, "agentic_index": 2.2 } } }, { "id": "mistralai/ministral-8b-2512", "canonical_slug": "mistralai/ministral-8b-2512", "hugging_face_id": "mistralai/Ministral-3-8B-Instruct-2512", "name": "Mistral: Ministral 3 8B 2512", "created": 1764681654, "description": "A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.00000015", "input_cache_read": "0.000000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/ministral-8b-2512/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1082, "win_rate": 46.2, "rank": 89 }, { "arena": "models", "category": "codecategories", "elo": 1072, "win_rate": 42.9, "rank": 95 }, { "arena": "models", "category": "gamedev", "elo": 1029, "win_rate": 38.7, "rank": 103 }, { "arena": "models", "category": "website", "elo": 1076, "win_rate": 42.9, "rank": 97 } ], "artificial_analysis": { "intelligence_index": 9, "coding_index": 9.7, "agentic_index": 1.2 } } }, { "id": "mistralai/ministral-3b-2512", "canonical_slug": "mistralai/ministral-3b-2512", "hugging_face_id": "mistralai/Ministral-3-3B-Instruct-2512", "name": "Mistral: Ministral 3 3B 2512", "created": 1764681560, "description": "The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.", "context_length": 131072, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000001", "input_cache_read": "0.00000001" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/ministral-3b-2512/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1013, "win_rate": 35.9, "rank": 102 }, { "arena": "models", "category": "codecategories", "elo": 1029, "win_rate": 37.3, "rank": 104 }, { "arena": "models", "category": "gamedev", "elo": 987, "win_rate": 33, "rank": 111 }, { "arena": "models", "category": "website", "elo": 1039, "win_rate": 38.2, "rank": 106 } ], "artificial_analysis": { "intelligence_index": 7.1, "coding_index": 4.8, "agentic_index": 1.6 } } }, { "id": "mistralai/mistral-large-2512", "canonical_slug": "mistralai/mistral-large-2512", "hugging_face_id": "", "name": "Mistral: Mistral Large 3 2512", "created": 1764624472, "description": "Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.", "context_length": 262144, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.0000015", "input_cache_read": "0.00000005" }, "top_provider": { "context_length": 262144, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.0645, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-large-2512/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1150, "win_rate": 47, "rank": 67 }, { "arena": "models", "category": "asciiart", "elo": 1101, "win_rate": 40.3, "rank": 53 }, { "arena": "models", "category": "codecategories", "elo": 1161, "win_rate": 47.6, "rank": 70 }, { "arena": "models", "category": "dataviz", "elo": 1157, "win_rate": 45.8, "rank": 69 }, { "arena": "models", "category": "gamedev", "elo": 1119, "win_rate": 41.6, "rank": 83 }, { "arena": "models", "category": "svg", "elo": 1028, "win_rate": 37.7, "rank": 73 }, { "arena": "models", "category": "uicomponent", "elo": 1130, "win_rate": 43.3, "rank": 75 }, { "arena": "models", "category": "website", "elo": 1174, "win_rate": 49.4, "rank": 67 } ], "artificial_analysis": { "intelligence_index": 15.9, "coding_index": 20.1, "agentic_index": 5.5 } } }, { "id": "deepseek/deepseek-v3.2", "canonical_slug": "deepseek/deepseek-v3.2-20251201", "hugging_face_id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek: DeepSeek V3.2", "created": 1764594642, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "context_length": 163840, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "pricing": { "prompt": "0.000000269", "completion": "0.0000004", "input_cache_read": "0.0000001345" }, "top_provider": { "context_length": 163840, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-v3.2-20251201/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1184, "win_rate": 49.6, "rank": 55 }, { "arena": "models", "category": "asciiart", "elo": 1115, "win_rate": 40.5, "rank": 51 }, { "arena": "models", "category": "codecategories", "elo": 1183, "win_rate": 49.3, "rank": 63 }, { "arena": "models", "category": "dataviz", "elo": 1179, "win_rate": 48.2, "rank": 61 }, { "arena": "models", "category": "gamedev", "elo": 1170, "win_rate": 46.6, "rank": 66 }, { "arena": "models", "category": "svg", "elo": 1063, "win_rate": 39.8, "rank": 66 }, { "arena": "models", "category": "uicomponent", "elo": 1175, "win_rate": 46.8, "rank": 62 }, { "arena": "models", "category": "website", "elo": 1186, "win_rate": 50.1, "rank": 61 } ], "artificial_analysis": { "intelligence_index": null, "coding_index": 44.2, "agentic_index": null } }, "reasoning": { "mandatory": false, "default_enabled": false } }, { "id": "anthropic/claude-opus-4.5", "canonical_slug": "anthropic/claude-4.5-opus-20251124", "hugging_face_id": "", "name": "Anthropic: Claude Opus 4.5", "created": 1764010580, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.5-opus-20251124/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1265, "win_rate": 58.6, "rank": 29 }, { "arena": "models", "category": "asciiart", "elo": 1222, "win_rate": 54.7, "rank": 18 }, { "arena": "models", "category": "codecategories", "elo": 1262, "win_rate": 59.5, "rank": 30 }, { "arena": "models", "category": "dataviz", "elo": 1263, "win_rate": 58.5, "rank": 23 }, { "arena": "models", "category": "gamedev", "elo": 1267, "win_rate": 59.3, "rank": 29 }, { "arena": "models", "category": "svg", "elo": 1218, "win_rate": 58.5, "rank": 21 }, { "arena": "models", "category": "uicomponent", "elo": 1265, "win_rate": 58.1, "rank": 33 }, { "arena": "models", "category": "website", "elo": 1259, "win_rate": 59.6, "rank": 31 }, { "arena": "agents", "category": "androidnative", "elo": 1218, "win_rate": 65.5, "rank": 14 }, { "arena": "agents", "category": "fullstack", "elo": 1187, "win_rate": 59.9, "rank": 20 }, { "arena": "agents", "category": "mobileapps", "elo": 1240, "win_rate": 60.5, "rank": 9 }, { "arena": "agents", "category": "webapps", "elo": 1189, "win_rate": 53, "rank": 21 } ] }, "reasoning": { "mandatory": false } }, { "id": "anthropic/claude-opus-4.5:batch", "canonical_slug": "anthropic/claude-4.5-opus-20251124", "hugging_face_id": "", "name": "Anthropic: Claude Opus 4.5 (batch)", "created": 1764010580, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "verbosity" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.5-opus-20251124/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1265, "win_rate": 58.6, "rank": 29 }, { "arena": "models", "category": "asciiart", "elo": 1222, "win_rate": 54.7, "rank": 18 }, { "arena": "models", "category": "codecategories", "elo": 1262, "win_rate": 59.5, "rank": 30 }, { "arena": "models", "category": "dataviz", "elo": 1263, "win_rate": 58.5, "rank": 23 }, { "arena": "models", "category": "gamedev", "elo": 1267, "win_rate": 59.3, "rank": 29 }, { "arena": "models", "category": "svg", "elo": 1218, "win_rate": 58.5, "rank": 21 }, { "arena": "models", "category": "uicomponent", "elo": 1265, "win_rate": 58.1, "rank": 33 }, { "arena": "models", "category": "website", "elo": 1259, "win_rate": 59.6, "rank": 31 }, { "arena": "agents", "category": "androidnative", "elo": 1218, "win_rate": 65.5, "rank": 14 }, { "arena": "agents", "category": "fullstack", "elo": 1187, "win_rate": 59.9, "rank": 20 }, { "arena": "agents", "category": "mobileapps", "elo": 1240, "win_rate": 60.5, "rank": 9 }, { "arena": "agents", "category": "webapps", "elo": 1189, "win_rate": 53, "rank": 21 } ] }, "reasoning": { "mandatory": false } }, { "id": "allenai/olmo-3-32b-think", "canonical_slug": "allenai/olmo-3-32b-think-20251121", "hugging_face_id": "allenai/Olmo-3-32B-Think", "name": "AllenAI: Olmo 3 32B Think", "created": 1763758276, "description": "Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...", "context_length": 65536, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000005" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/allenai/olmo-3-32b-think-20251121/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "google/gemini-3-pro-image-preview", "canonical_slug": "google/gemini-3-pro-image-preview-20251120", "hugging_face_id": "", "name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)", "created": 1763653797, "description": "Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...", "context_length": 65536, "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "image_output": "0.00012", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-3-pro-image-preview-20251120/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "graphicdesign", "elo": 1275, "win_rate": 65.8, "rank": 3 }, { "arena": "models", "category": "image", "elo": 1263, "win_rate": 62.1, "rank": 3 }, { "arena": "models", "category": "logo", "elo": 1253, "win_rate": 61, "rank": 4 }, { "arena": "models", "category": "imageediting", "elo": 1270, "win_rate": 65.4, "rank": 2 } ] }, "reasoning": { "mandatory": true } }, { "id": "deepcogito/cogito-v2.1-671b", "canonical_slug": "deepcogito/cogito-v2.1-671b-20251118", "hugging_face_id": "", "name": "Deep Cogito: Cogito v2.1 671B", "created": 1763071233, "description": "Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00000125" }, "top_provider": { "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/deepcogito/cogito-v2.1-671b-20251118/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "openai/gpt-5.1", "canonical_slug": "openai/gpt-5.1-20251113", "hugging_face_id": "", "name": "OpenAI: GPT-5.1", "created": 1763060305, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.1-20251113/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "webapps", "elo": 1029, "win_rate": 39.9, "rank": 36 }, { "arena": "models", "category": "3d", "elo": 1110, "win_rate": 43.7, "rank": 85 }, { "arena": "models", "category": "asciiart", "elo": 1151, "win_rate": 48.6, "rank": 44 }, { "arena": "models", "category": "codecategories", "elo": 1190, "win_rate": 52.9, "rank": 57 }, { "arena": "models", "category": "dataviz", "elo": 1225, "win_rate": 58.1, "rank": 42 }, { "arena": "models", "category": "gamedev", "elo": 1217, "win_rate": 55.9, "rank": 44 }, { "arena": "models", "category": "svg", "elo": 1186, "win_rate": 56.3, "rank": 35 }, { "arena": "models", "category": "uicomponent", "elo": 1196, "win_rate": 53.4, "rank": 54 }, { "arena": "models", "category": "website", "elo": 1198, "win_rate": 53.9, "rank": 56 } ], "artificial_analysis": { "intelligence_index": 37.5, "coding_index": 49.4, "agentic_index": 21.6 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "none" ], "default_effort": "none" } }, { "id": "openai/gpt-5.1:batch", "canonical_slug": "openai/gpt-5.1-20251113", "hugging_face_id": "", "name": "OpenAI: GPT-5.1 (batch)", "created": 1763060305, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000000625" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.1-20251113/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "webapps", "elo": 1029, "win_rate": 39.9, "rank": 36 }, { "arena": "models", "category": "3d", "elo": 1110, "win_rate": 43.7, "rank": 85 }, { "arena": "models", "category": "asciiart", "elo": 1151, "win_rate": 48.6, "rank": 44 }, { "arena": "models", "category": "codecategories", "elo": 1190, "win_rate": 52.9, "rank": 57 }, { "arena": "models", "category": "dataviz", "elo": 1225, "win_rate": 58.1, "rank": 42 }, { "arena": "models", "category": "gamedev", "elo": 1217, "win_rate": 55.9, "rank": 44 }, { "arena": "models", "category": "svg", "elo": 1186, "win_rate": 56.3, "rank": 35 }, { "arena": "models", "category": "uicomponent", "elo": 1196, "win_rate": 53.4, "rank": 54 }, { "arena": "models", "category": "website", "elo": 1198, "win_rate": 53.9, "rank": 56 } ], "artificial_analysis": { "intelligence_index": 37.5, "coding_index": 49.4, "agentic_index": 21.6 } }, "reasoning": { "mandatory": false, "default_enabled": true, "supported_efforts": [ "high", "medium", "low", "none" ], "default_effort": "none" } }, { "id": "openai/gpt-5.1-codex", "canonical_slug": "openai/gpt-5.1-codex-20251113", "hugging_face_id": "", "name": "OpenAI: GPT-5.1-Codex", "created": 1763060298, "description": "GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....", "context_length": 400000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.00000013" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.1-codex-20251113/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "fullstack", "elo": 1086, "win_rate": 44.5, "rank": 28 }, { "arena": "agents", "category": "mobileapps", "elo": 1191, "win_rate": 53.4, "rank": 22 }, { "arena": "agents", "category": "webapps", "elo": 1057, "win_rate": 41.5, "rank": 35 }, { "arena": "models", "category": "codecategories", "elo": 1171, "win_rate": 54.6, "rank": 66 }, { "arena": "models", "category": "dataviz", "elo": 1234, "win_rate": 55.9, "rank": 40 }, { "arena": "models", "category": "gamedev", "elo": 1177, "win_rate": 50.5, "rank": 61 }, { "arena": "models", "category": "uicomponent", "elo": 1264, "win_rate": 60, "rank": 34 }, { "arena": "models", "category": "website", "elo": 1173, "win_rate": 56.1, "rank": 68 } ] }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "openai/gpt-5.1-codex-mini", "canonical_slug": "openai/gpt-5.1-codex-mini-20251113", "hugging_face_id": "", "name": "OpenAI: GPT-5.1-Codex-Mini", "created": 1763057820, "description": "GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex", "context_length": 400000, "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "web_search": "0.01", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5.1-codex-mini-20251113/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1039, "win_rate": 32.8, "rank": 97 }, { "arena": "models", "category": "asciiart", "elo": 1137, "win_rate": 43, "rank": 48 }, { "arena": "models", "category": "codecategories", "elo": 1113, "win_rate": 41.5, "rank": 88 }, { "arena": "models", "category": "dataviz", "elo": 1117, "win_rate": 40.5, "rank": 86 }, { "arena": "models", "category": "gamedev", "elo": 1131, "win_rate": 43.5, "rank": 80 }, { "arena": "models", "category": "svg", "elo": 1012, "win_rate": 35, "rank": 75 }, { "arena": "models", "category": "uicomponent", "elo": 1108, "win_rate": 40.9, "rank": 82 }, { "arena": "models", "category": "website", "elo": 1122, "win_rate": 42.8, "rank": 88 } ] }, "reasoning": { "mandatory": false, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "moonshotai/kimi-k2-thinking", "canonical_slug": "moonshotai/kimi-k2-thinking-20251106", "hugging_face_id": "moonshotai/Kimi-K2-Thinking", "name": "MoonshotAI: Kimi K2 Thinking", "created": 1762440622, "description": "Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.0000025", "input_cache_read": "0.00000015" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 100352, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k2-thinking-20251106/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "website", "elo": 1124, "win_rate": 48.8, "rank": 87 } ], "artificial_analysis": { "intelligence_index": null, "coding_index": 21, "agentic_index": null } }, "reasoning": { "mandatory": true, "default_enabled": true } }, { "id": "amazon/nova-premier-v1", "canonical_slug": "amazon/nova-premier-v1", "hugging_face_id": "", "name": "Amazon: Nova Premier 1.0", "created": 1761950332, "description": "Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.", "context_length": 1000000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "input_cache_read": "0.000000625" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 32000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "stop", "temperature", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/amazon/nova-premier-v1/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "website", "elo": 846, "win_rate": 26.2, "rank": 123 } ] } }, { "id": "perplexity/sonar-pro-search", "canonical_slug": "perplexity/sonar-pro-search", "hugging_face_id": "", "name": "Perplexity: Sonar Pro Search", "created": 1761854366, "description": "Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...", "context_length": 200000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.018" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 8000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "structured_outputs", "temperature", "top_k", "top_p", "web_search_options" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/perplexity/sonar-pro-search/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "mistralai/voxtral-small-24b-2507", "canonical_slug": "mistralai/voxtral-small-24b-2507", "hugging_face_id": "mistralai/Voxtral-Small-24B-2507", "name": "Mistral: Voxtral Small 24B 2507", "created": 1761835144, "description": "Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...", "context_length": 32000, "architecture": { "modality": "text+file+audio->text", "input_modalities": [ "text", "audio", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "audio": "0.0001", "input_cache_read": "0.00000001" }, "top_provider": { "context_length": 32000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.2, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/voxtral-small-24b-2507/endpoints" } }, { "id": "openai/gpt-oss-safeguard-20b", "canonical_slug": "openai/gpt-oss-safeguard-20b", "hugging_face_id": "openai/gpt-oss-safeguard-20b", "name": "OpenAI: gpt-oss-safeguard-20b", "created": 1761752836, "description": "gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000075", "completion": "0.0000003", "input_cache_read": "0.0000000375" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-oss-safeguard-20b/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "nvidia/nemotron-nano-12b-v2-vl:free", "canonical_slug": "nvidia/nemotron-nano-12b-v2-vl", "hugging_face_id": "nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16", "name": "NVIDIA: Nemotron Nano 12B 2 VL (free)", "created": 1761675565, "description": "NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...", "context_length": 128000, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "seed", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-nano-12b-v2-vl/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "minimax/minimax-m2", "canonical_slug": "minimax/minimax-m2", "hugging_face_id": "MiniMaxAI/MiniMax-M2", "name": "MiniMax: MiniMax M2", "created": 1761252093, "description": "MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000255", "completion": "0.00000102" }, "top_provider": { "context_length": 204800, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m2/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1148, "win_rate": 48.6, "rank": 68 }, { "arena": "models", "category": "codecategories", "elo": 1153, "win_rate": 48.1, "rank": 75 }, { "arena": "models", "category": "dataviz", "elo": 1161, "win_rate": 49.8, "rank": 68 }, { "arena": "models", "category": "gamedev", "elo": 1155, "win_rate": 48.3, "rank": 69 }, { "arena": "models", "category": "svg", "elo": 1133, "win_rate": 54, "rank": 49 }, { "arena": "models", "category": "uicomponent", "elo": 1165, "win_rate": 49.7, "rank": 68 }, { "arena": "models", "category": "website", "elo": 1153, "win_rate": 47.9, "rank": 76 } ] }, "reasoning": { "mandatory": true } }, { "id": "qwen/qwen3-vl-32b-instruct", "canonical_slug": "qwen/qwen3-vl-32b-instruct", "hugging_face_id": "Qwen/Qwen3-VL-32B-Instruct", "name": "Qwen: Qwen3 VL 32B Instruct", "created": 1761231332, "description": "Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...", "context_length": 131072, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.000000104", "completion": "0.000000416" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.7, "top_p": 0.8, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": 1 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-vl-32b-instruct/endpoints" } }, { "id": "ibm-granite/granite-4.0-h-micro", "canonical_slug": "ibm-granite/granite-4.0-h-micro", "hugging_face_id": "ibm-granite/granite-4.0-h-micro", "name": "IBM: Granite 4.0 Micro", "created": 1760927695, "description": "Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...", "context_length": 131000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000000017", "completion": "0.000000112" }, "top_provider": { "context_length": 131000, "max_completion_tokens": 131000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/ibm-granite/granite-4.0-h-micro/endpoints" } }, { "id": "openai/gpt-5-image-mini", "canonical_slug": "openai/gpt-5-image-mini", "hugging_face_id": "", "name": "OpenAI: GPT-5 Image Mini", "created": 1760624583, "description": "GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...", "context_length": 400000, "architecture": { "modality": "text+image+file->text+image", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.000002", "image_output": "0.000008", "web_search": "0.01", "input_cache_read": "0.00000025" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-image-mini/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "graphicdesign", "elo": 1190, "win_rate": 47.7, "rank": 9 }, { "arena": "models", "category": "image", "elo": 1203, "win_rate": 51, "rank": 9 }, { "arena": "models", "category": "logo", "elo": 1221, "win_rate": 51.4, "rank": 6 } ] }, "reasoning": { "mandatory": true } }, { "id": "anthropic/claude-haiku-4.5", "canonical_slug": "anthropic/claude-4.5-haiku-20251001", "hugging_face_id": "", "name": "Anthropic: Claude Haiku 4.5", "created": 1760547638, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.5-haiku-20251001/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1123, "win_rate": 41.2, "rank": 82 }, { "arena": "models", "category": "asciiart", "elo": 1172, "win_rate": 49.3, "rank": 34 }, { "arena": "models", "category": "codecategories", "elo": 1134, "win_rate": 44.9, "rank": 81 }, { "arena": "models", "category": "dataviz", "elo": 1145, "win_rate": 45.7, "rank": 76 }, { "arena": "models", "category": "gamedev", "elo": 1136, "win_rate": 44.7, "rank": 76 }, { "arena": "models", "category": "svg", "elo": 1063, "win_rate": 39, "rank": 65 }, { "arena": "models", "category": "uicomponent", "elo": 1129, "win_rate": 43, "rank": 76 }, { "arena": "models", "category": "website", "elo": 1133, "win_rate": 45, "rank": 82 } ], "artificial_analysis": { "intelligence_index": 29.9, "coding_index": 43.9, "agentic_index": 16.5 } }, "reasoning": { "mandatory": false } }, { "id": "anthropic/claude-haiku-4.5:batch", "canonical_slug": "anthropic/claude-4.5-haiku-20251001", "hugging_face_id": "", "name": "Anthropic: Claude Haiku 4.5 (batch)", "created": 1760547638, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.0000025", "web_search": "0.01", "input_cache_read": "0.00000005", "input_cache_write": "0.000000625", "input_cache_write_1h": "0.000001" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.5-haiku-20251001/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1123, "win_rate": 41.2, "rank": 82 }, { "arena": "models", "category": "asciiart", "elo": 1172, "win_rate": 49.3, "rank": 34 }, { "arena": "models", "category": "codecategories", "elo": 1134, "win_rate": 44.9, "rank": 81 }, { "arena": "models", "category": "dataviz", "elo": 1145, "win_rate": 45.7, "rank": 76 }, { "arena": "models", "category": "gamedev", "elo": 1136, "win_rate": 44.7, "rank": 76 }, { "arena": "models", "category": "svg", "elo": 1063, "win_rate": 39, "rank": 65 }, { "arena": "models", "category": "uicomponent", "elo": 1129, "win_rate": 43, "rank": 76 }, { "arena": "models", "category": "website", "elo": 1133, "win_rate": 45, "rank": 82 } ], "artificial_analysis": { "intelligence_index": 29.9, "coding_index": 43.9, "agentic_index": 16.5 } }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3-vl-8b-thinking", "canonical_slug": "qwen/qwen3-vl-8b-thinking", "hugging_face_id": "Qwen/Qwen3-VL-8B-Thinking", "name": "Qwen: Qwen3 VL 8B Thinking", "created": 1760463746, "description": "Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...", "context_length": 131072, "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000018", "completion": "0.0000021" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 0.95 }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-vl-8b-thinking/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "qwen/qwen3-vl-8b-instruct", "canonical_slug": "qwen/qwen3-vl-8b-instruct", "hugging_face_id": "Qwen/Qwen3-VL-8B-Instruct", "name": "Qwen: Qwen3 VL 8B Instruct", "created": 1760463308, "description": "Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.000000117", "completion": "0.000000455" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.7, "top_p": 0.8, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-vl-8b-instruct/endpoints" } }, { "id": "openai/gpt-5-image", "canonical_slug": "openai/gpt-5-image", "hugging_face_id": "", "name": "OpenAI: GPT-5 Image", "created": 1760447986, "description": "[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...", "context_length": 400000, "architecture": { "modality": "text+image+file->text+image", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "image", "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00001", "image_output": "0.00004", "web_search": "0.01", "input_cache_read": "0.00000125" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-image/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "graphicdesign", "elo": 1197, "win_rate": 48.9, "rank": 7 }, { "arena": "models", "category": "image", "elo": 1211, "win_rate": 53.9, "rank": 7 }, { "arena": "models", "category": "logo", "elo": 1214, "win_rate": 52.7, "rank": 7 } ] }, "reasoning": { "mandatory": true } }, { "id": "google/gemini-2.5-flash-image", "canonical_slug": "google/gemini-2.5-flash-image", "hugging_face_id": "", "name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "created": 1759870431, "description": "Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...", "context_length": 32768, "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "image_output": "0.00003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-flash-image/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "graphicdesign", "elo": 1192, "win_rate": 56.9, "rank": 8 }, { "arena": "models", "category": "image", "elo": 1206, "win_rate": 55.6, "rank": 8 }, { "arena": "models", "category": "logo", "elo": 1181, "win_rate": 51.4, "rank": 9 } ] } }, { "id": "qwen/qwen3-vl-30b-a3b-thinking", "canonical_slug": "qwen/qwen3-vl-30b-a3b-thinking", "hugging_face_id": "Qwen/Qwen3-VL-30B-A3B-Thinking", "name": "Qwen: Qwen3 VL 30B A3B Thinking", "created": 1759794479, "description": "Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000024" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.8, "top_p": 0.95, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": 1 }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-vl-30b-a3b-thinking/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "qwen/qwen3-vl-30b-a3b-instruct", "canonical_slug": "qwen/qwen3-vl-30b-a3b-instruct", "hugging_face_id": "Qwen/Qwen3-VL-30B-A3B-Instruct", "name": "Qwen: Qwen3 VL 30B A3B Instruct", "created": 1759794476, "description": "Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000013", "completion": "0.00000052" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.7, "top_p": 0.8, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": 1 }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-vl-30b-a3b-instruct/endpoints" } }, { "id": "openai/gpt-5-pro", "canonical_slug": "openai/gpt-5-pro-2025-10-06", "hugging_face_id": "", "name": "OpenAI: GPT-5 Pro", "created": 1759776663, "description": "GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000015", "completion": "0.00012", "web_search": "0.01" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-pro-2025-10-06/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "high" ], "default_effort": "high" } }, { "id": "openai/gpt-5-pro:batch", "canonical_slug": "openai/gpt-5-pro-2025-10-06", "hugging_face_id": "", "name": "OpenAI: GPT-5 Pro (batch)", "created": 1759776663, "description": "GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000075", "completion": "0.00006", "web_search": "0.01" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-pro-2025-10-06/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "high" ], "default_effort": "high" } }, { "id": "z-ai/glm-4.6", "canonical_slug": "z-ai/glm-4.6", "hugging_face_id": "zai-org/GLM-4.6", "name": "Z.ai: GLM 4.6", "created": 1759235576, "description": "Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...", "context_length": 204800, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.000002", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 202752, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-4.6/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "androidnative", "elo": 1127, "win_rate": 52.7, "rank": 28 }, { "arena": "agents", "category": "fullstack", "elo": 1062, "win_rate": 42.3, "rank": 34 }, { "arena": "agents", "category": "godotgamedev", "elo": 1180, "win_rate": 53.2, "rank": 13 }, { "arena": "agents", "category": "mobileapps", "elo": 1145, "win_rate": 46.8, "rank": 27 }, { "arena": "models", "category": "3d", "elo": 1180, "win_rate": 54.3, "rank": 57 }, { "arena": "models", "category": "codecategories", "elo": 1185, "win_rate": 54.2, "rank": 61 }, { "arena": "models", "category": "dataviz", "elo": 1181, "win_rate": 52.5, "rank": 60 }, { "arena": "models", "category": "gamedev", "elo": 1191, "win_rate": 55.1, "rank": 56 }, { "arena": "models", "category": "svg", "elo": 1141, "win_rate": 50.4, "rank": 47 }, { "arena": "models", "category": "uicomponent", "elo": 1175, "win_rate": 52.3, "rank": 63 }, { "arena": "models", "category": "website", "elo": 1185, "win_rate": 54.2, "rank": 62 } ], "artificial_analysis": { "intelligence_index": 29.3, "coding_index": 45.8, "agentic_index": 18.6 } }, "reasoning": { "mandatory": false } }, { "id": "anthropic/claude-sonnet-4.5", "canonical_slug": "anthropic/claude-4.5-sonnet-20250929", "hugging_face_id": "", "name": "Anthropic: Claude Sonnet 4.5", "created": 1759161676, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 1, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.5-sonnet-20250929/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1208, "win_rate": 51.1, "rank": 48 }, { "arena": "models", "category": "asciiart", "elo": 1236, "win_rate": 56.1, "rank": 16 }, { "arena": "models", "category": "codecategories", "elo": 1202, "win_rate": 51.4, "rank": 49 }, { "arena": "models", "category": "dataviz", "elo": 1187, "win_rate": 47.4, "rank": 56 }, { "arena": "models", "category": "gamedev", "elo": 1205, "win_rate": 51.1, "rank": 49 }, { "arena": "models", "category": "svg", "elo": 1152, "win_rate": 51.8, "rank": 45 }, { "arena": "models", "category": "uicomponent", "elo": 1199, "win_rate": 49.5, "rank": 53 }, { "arena": "models", "category": "website", "elo": 1201, "win_rate": 51.7, "rank": 53 }, { "arena": "agents", "category": "fullstack", "elo": 1083, "win_rate": 43.5, "rank": 29 }, { "arena": "agents", "category": "mobileapps", "elo": 1187, "win_rate": 52.3, "rank": 24 }, { "arena": "agents", "category": "webapps", "elo": 1100, "win_rate": 43.6, "rank": 31 } ], "artificial_analysis": { "intelligence_index": 37.4, "coding_index": 52.1, "agentic_index": 26.4 } }, "reasoning": { "mandatory": false } }, { "id": "anthropic/claude-sonnet-4.5:batch", "canonical_slug": "anthropic/claude-4.5-sonnet-20250929", "hugging_face_id": "", "name": "Anthropic: Claude Sonnet 4.5 (batch)", "created": 1759161676, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000015", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.00000015", "input_cache_write": "0.000001875", "input_cache_write_1h": "0.000003", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000003", "completion": "0.00001125", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 1, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.5-sonnet-20250929/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1208, "win_rate": 51.1, "rank": 48 }, { "arena": "models", "category": "asciiart", "elo": 1236, "win_rate": 56.1, "rank": 16 }, { "arena": "models", "category": "codecategories", "elo": 1202, "win_rate": 51.4, "rank": 49 }, { "arena": "models", "category": "dataviz", "elo": 1187, "win_rate": 47.4, "rank": 56 }, { "arena": "models", "category": "gamedev", "elo": 1205, "win_rate": 51.1, "rank": 49 }, { "arena": "models", "category": "svg", "elo": 1152, "win_rate": 51.8, "rank": 45 }, { "arena": "models", "category": "uicomponent", "elo": 1199, "win_rate": 49.5, "rank": 53 }, { "arena": "models", "category": "website", "elo": 1201, "win_rate": 51.7, "rank": 53 }, { "arena": "agents", "category": "fullstack", "elo": 1083, "win_rate": 43.5, "rank": 29 }, { "arena": "agents", "category": "mobileapps", "elo": 1187, "win_rate": 52.3, "rank": 24 }, { "arena": "agents", "category": "webapps", "elo": 1100, "win_rate": 43.6, "rank": 31 } ], "artificial_analysis": { "intelligence_index": 37.4, "coding_index": 52.1, "agentic_index": 26.4 } }, "reasoning": { "mandatory": false } }, { "id": "deepseek/deepseek-v3.2-exp", "canonical_slug": "deepseek/deepseek-v3.2-exp", "hugging_face_id": "deepseek-ai/DeepSeek-V3.2-Exp", "name": "DeepSeek: DeepSeek V3.2 Exp", "created": 1759150481, "description": "DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "context_length": 163840, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "pricing": { "prompt": "0.00000027", "completion": "0.00000041" }, "top_provider": { "context_length": 163840, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": 0.95, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-07-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-v3.2-exp/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1199, "win_rate": 56.4, "rank": 50 }, { "arena": "models", "category": "codecategories", "elo": 1190, "win_rate": 54.2, "rank": 56 }, { "arena": "models", "category": "dataviz", "elo": 1172, "win_rate": 50.9, "rank": 65 }, { "arena": "models", "category": "gamedev", "elo": 1184, "win_rate": 53.2, "rank": 58 }, { "arena": "models", "category": "svg", "elo": 1066, "win_rate": 41.2, "rank": 62 }, { "arena": "models", "category": "uicomponent", "elo": 1194, "win_rate": 53.4, "rank": 56 }, { "arena": "models", "category": "website", "elo": 1191, "win_rate": 54.2, "rank": 59 } ] }, "reasoning": { "mandatory": false } }, { "id": "thedrummer/cydonia-24b-v4.1", "canonical_slug": "thedrummer/cydonia-24b-v4.1", "hugging_face_id": "thedrummer/cydonia-24b-v4.1", "name": "TheDrummer: Cydonia 24B V4.1", "created": 1758931878, "description": "Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000005", "input_cache_read": "0.00000015" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-04-30", "expiration_date": null, "links": { "details": "/api/v1/models/thedrummer/cydonia-24b-v4.1/endpoints" } }, { "id": "relace/relace-apply-3", "canonical_slug": "relace/relace-apply-3", "hugging_face_id": "", "name": "Relace: Relace Apply 3", "created": 1758891572, "description": "Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000085", "completion": "0.00000125" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "seed", "stop" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/relace/relace-apply-3/endpoints" } }, { "id": "qwen/qwen3-vl-235b-a22b-thinking", "canonical_slug": "qwen/qwen3-vl-235b-a22b-thinking", "hugging_face_id": "Qwen/Qwen3-VL-235B-A22B-Thinking", "name": "Qwen: Qwen3 VL 235B A22B Thinking", "created": 1758668690, "description": "Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....", "context_length": 131072, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000004", "completion": "0.000004" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.8, "top_p": 0.95, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": 1 }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-vl-235b-a22b-thinking/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "qwen/qwen3-vl-235b-a22b-instruct", "canonical_slug": "qwen/qwen3-vl-235b-a22b-instruct", "hugging_face_id": "Qwen/Qwen3-VL-235B-A22B-Instruct", "name": "Qwen: Qwen3 VL 235B A22B Instruct", "created": 1758668687, "description": "Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000021", "completion": "0.0000019", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.7, "top_p": 0.8, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-vl-235b-a22b-instruct/endpoints" } }, { "id": "qwen/qwen3-max", "canonical_slug": "qwen/qwen3-max", "hugging_face_id": "", "name": "Qwen: Qwen3 Max", "created": 1758662808, "description": "Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000078", "completion": "0.0000039", "input_cache_read": "0.000000156", "input_cache_write": "0.000000975", "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.00000156", "completion": "0.0000078", "input_cache_read": "0.000000312", "input_cache_write": "0.00000195" }, { "min_prompt_tokens": 128000, "prompt": "0.00000195", "completion": "0.00000975", "input_cache_read": "0.00000039", "input_cache_write": "0.0000024375" } ] }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 1, "top_p": 1, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-max/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1124, "win_rate": 43.6, "rank": 81 }, { "arena": "models", "category": "asciiart", "elo": 1160, "win_rate": 47.2, "rank": 40 }, { "arena": "models", "category": "codecategories", "elo": 1129, "win_rate": 44, "rank": 84 }, { "arena": "models", "category": "dataviz", "elo": 1120, "win_rate": 41.3, "rank": 83 }, { "arena": "models", "category": "gamedev", "elo": 1133, "win_rate": 43.9, "rank": 79 }, { "arena": "models", "category": "svg", "elo": 1046, "win_rate": 36.9, "rank": 70 }, { "arena": "models", "category": "uicomponent", "elo": 1103, "win_rate": 40, "rank": 84 }, { "arena": "models", "category": "website", "elo": 1130, "win_rate": 44.4, "rank": 85 } ] }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3-coder-plus", "canonical_slug": "qwen/qwen3-coder-plus", "hugging_face_id": "", "name": "Qwen: Qwen3 Coder Plus", "created": 1758662707, "description": "Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000065", "completion": "0.00000325", "input_cache_read": "0.00000013", "input_cache_write": "0.0000008125", "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.00000117", "completion": "0.00000585", "input_cache_read": "0.000000234", "input_cache_write": "0.0000014625" }, { "min_prompt_tokens": 128000, "prompt": "0.00000195", "completion": "0.00000975", "input_cache_read": "0.00000039", "input_cache_write": "0.0000024375" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-coder-plus/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "openai/gpt-5-codex:batch", "canonical_slug": "openai/gpt-5-codex", "hugging_face_id": "", "name": "OpenAI: GPT-5 Codex (batch)", "created": 1758643403, "description": "GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....", "context_length": 400000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000000625" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-codex/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "agents", "category": "webapps", "elo": 1109, "win_rate": 51, "rank": 30 } ] }, "reasoning": { "mandatory": true } }, { "id": "deepseek/deepseek-v3.1-terminus", "canonical_slug": "deepseek/deepseek-v3.1-terminus", "hugging_face_id": "deepseek-ai/DeepSeek-V3.1-Terminus", "name": "DeepSeek: DeepSeek V3.1 Terminus", "created": 1758548275, "description": "DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...", "context_length": 163840, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "pricing": { "prompt": "0.00000027", "completion": "0.000001" }, "top_provider": { "context_length": 163840, "max_completion_tokens": 163840, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-v3.1-terminus/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1191, "win_rate": 56.1, "rank": 53 }, { "arena": "models", "category": "codecategories", "elo": 1194, "win_rate": 56, "rank": 54 }, { "arena": "models", "category": "dataviz", "elo": 1188, "win_rate": 53.9, "rank": 55 }, { "arena": "models", "category": "gamedev", "elo": 1170, "win_rate": 52.7, "rank": 65 }, { "arena": "models", "category": "svg", "elo": 1096, "win_rate": 48.1, "rank": 57 }, { "arena": "models", "category": "uicomponent", "elo": 1210, "win_rate": 59.5, "rank": 46 }, { "arena": "models", "category": "website", "elo": 1198, "win_rate": 56.2, "rank": 55 } ], "artificial_analysis": { "intelligence_index": null, "coding_index": 43.5, "agentic_index": null } }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3-coder-flash", "canonical_slug": "qwen/qwen3-coder-flash", "hugging_face_id": "", "name": "Qwen: Qwen3 Coder Flash", "created": 1758115536, "description": "Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.000000195", "completion": "0.000000975", "input_cache_read": "0.000000039", "input_cache_write": "0.00000024375", "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.000000325", "completion": "0.000001625", "input_cache_read": "0.000000065", "input_cache_write": "0.00000040625" }, { "min_prompt_tokens": 128000, "prompt": "0.00000052", "completion": "0.0000026", "input_cache_read": "0.000000104", "input_cache_write": "0.00000065" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-coder-flash/endpoints" } }, { "id": "qwen/qwen3-next-80b-a3b-thinking", "canonical_slug": "qwen/qwen3-next-80b-a3b-thinking-2509", "hugging_face_id": "Qwen/Qwen3-Next-80B-A3B-Thinking", "name": "Qwen: Qwen3 Next 80B A3B Thinking", "created": 1757612284, "description": "Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000012" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-next-80b-a3b-thinking-2509/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 16.9, "coding_index": 17.4, "agentic_index": 2.1 } }, "reasoning": { "mandatory": true } }, { "id": "qwen/qwen3-next-80b-a3b-instruct", "canonical_slug": "qwen/qwen3-next-80b-a3b-instruct-2509", "hugging_face_id": "Qwen/Qwen3-Next-80B-A3B-Instruct", "name": "Qwen: Qwen3 Next 80B A3B Instruct", "created": 1757612213, "description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000011", "input_cache_read": "0.00000007" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-next-80b-a3b-instruct-2509/endpoints" } }, { "id": "qwen/qwen-plus-2025-07-28", "canonical_slug": "qwen/qwen-plus-2025-07-28", "hugging_face_id": "", "name": "Qwen: Qwen Plus 0728", "created": 1757347599, "description": "Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000026", "completion": "0.00000078", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000078", "completion": "0.00000234" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen-plus-2025-07-28/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen-plus-2025-07-28:thinking", "canonical_slug": "qwen/qwen-plus-2025-07-28", "hugging_face_id": "", "name": "Qwen: Qwen Plus 0728 (thinking)", "created": 1757347599, "description": "Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000026", "completion": "0.00000078", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000078", "completion": "0.00000234" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen-plus-2025-07-28/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "nvidia/nemotron-nano-9b-v2:free", "canonical_slug": "nvidia/nemotron-nano-9b-v2", "hugging_face_id": "nvidia/NVIDIA-Nemotron-Nano-9B-v2", "name": "NVIDIA: Nemotron Nano 9B V2 (free)", "created": 1757106807, "description": "NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/nvidia/nemotron-nano-9b-v2/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "moonshotai/kimi-k2-0905", "canonical_slug": "moonshotai/kimi-k2-0905", "hugging_face_id": "moonshotai/Kimi-K2-Instruct-0905", "name": "MoonshotAI: Kimi K2 0905", "created": 1757021147, "description": "Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.0000025" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 100352, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k2-0905/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 1117, "win_rate": 48.5, "rank": 87 }, { "arena": "models", "category": "website", "elo": 1118, "win_rate": 48.3, "rank": 89 } ] } }, { "id": "qwen/qwen3-30b-a3b-thinking-2507", "canonical_slug": "qwen/qwen3-30b-a3b-thinking-2507", "hugging_face_id": "Qwen/Qwen3-30B-A3B-Thinking-2507", "name": "Qwen: Qwen3 30B A3B Thinking 2507", "created": 1756399192, "description": "Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...", "context_length": 81920, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000024" }, "top_provider": { "context_length": 81920, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-30b-a3b-thinking-2507/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "dataviz", "elo": 944, "win_rate": 33.3, "rank": 110 }, { "arena": "models", "category": "website", "elo": 941, "win_rate": 35.5, "rank": 118 } ], "artificial_analysis": { "intelligence_index": 14.6, "coding_index": 12.1, "agentic_index": 1.8 } }, "reasoning": { "mandatory": true } }, { "id": "nousresearch/hermes-4-70b", "canonical_slug": "nousresearch/hermes-4-70b", "hugging_face_id": "NousResearch/Hermes-4-70B", "name": "Nous: Hermes 4 70B", "created": 1756236182, "description": "Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": null }, "pricing": { "prompt": "0.00000013", "completion": "0.0000004" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/nousresearch/hermes-4-70b/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "nousresearch/hermes-4-405b", "canonical_slug": "nousresearch/hermes-4-405b", "hugging_face_id": "NousResearch/Hermes-4-405B", "name": "Nous: Hermes 4 405B", "created": 1756235463, "description": "Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000003" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/nousresearch/hermes-4-405b/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "deepseek/deepseek-chat-v3.1", "canonical_slug": "deepseek/deepseek-chat-v3.1", "hugging_face_id": "deepseek-ai/DeepSeek-V3.1", "name": "DeepSeek: DeepSeek V3.1", "created": 1755779628, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "context_length": 163840, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "pricing": { "prompt": "0.00000025", "completion": "0.00000095", "input_cache_read": "0.00000013" }, "top_provider": { "context_length": 163840, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-chat-v3.1/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1129, "win_rate": 48.5, "rank": 77 }, { "arena": "models", "category": "codecategories", "elo": 1131, "win_rate": 47.8, "rank": 83 }, { "arena": "models", "category": "dataviz", "elo": 1116, "win_rate": 46, "rank": 87 }, { "arena": "models", "category": "gamedev", "elo": 1123, "win_rate": 47.3, "rank": 82 }, { "arena": "models", "category": "svg", "elo": 1000, "win_rate": 37.3, "rank": 78 }, { "arena": "models", "category": "uicomponent", "elo": 1112, "win_rate": 47.3, "rank": 81 }, { "arena": "models", "category": "website", "elo": 1134, "win_rate": 47.9, "rank": 81 } ] }, "reasoning": { "mandatory": false } }, { "id": "mistralai/mistral-medium-3.1", "canonical_slug": "mistralai/mistral-medium-3.1", "hugging_face_id": "", "name": "Mistral: Mistral Medium 3.1", "created": 1755095639, "description": "Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...", "context_length": 131072, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000004", "completion": "0.000002", "input_cache_read": "0.00000004" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-medium-3.1/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1131, "win_rate": 44.6, "rank": 76 }, { "arena": "models", "category": "asciiart", "elo": 1030, "win_rate": 30.8, "rank": 57 }, { "arena": "models", "category": "codecategories", "elo": 1139, "win_rate": 45, "rank": 79 }, { "arena": "models", "category": "dataviz", "elo": 1164, "win_rate": 47.3, "rank": 67 }, { "arena": "models", "category": "gamedev", "elo": 1113, "win_rate": 40.8, "rank": 86 }, { "arena": "models", "category": "svg", "elo": 1026, "win_rate": 37.5, "rank": 74 }, { "arena": "models", "category": "uicomponent", "elo": 1126, "win_rate": 43.5, "rank": 78 }, { "arena": "models", "category": "website", "elo": 1144, "win_rate": 45.9, "rank": 79 } ], "artificial_analysis": { "intelligence_index": 14.7, "coding_index": 20.5, "agentic_index": 6.1 } } }, { "id": "z-ai/glm-4.5v", "canonical_slug": "z-ai/glm-4.5v", "hugging_face_id": "zai-org/GLM-4.5V", "name": "Z.ai: GLM 4.5V", "created": 1754922288, "description": "GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...", "context_length": 65536, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.0000018", "input_cache_read": "0.00000011" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 0.75, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-4.5v/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "ai21/jamba-large-1.7", "canonical_slug": "ai21/jamba-large-1.7", "hugging_face_id": "ai21labs/AI21-Jamba-Large-1.7", "name": "AI21: Jamba Large 1.7", "created": 1754669020, "description": "Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000008" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 4096, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "stop", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/ai21/jamba-large-1.7/endpoints" } }, { "id": "openai/gpt-5", "canonical_slug": "openai/gpt-5-2025-08-07", "hugging_face_id": "", "name": "OpenAI: GPT-5", "created": 1754587413, "description": "GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-2025-08-07/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1106, "win_rate": 41.5, "rank": 86 }, { "arena": "models", "category": "asciiart", "elo": 1173, "win_rate": 49, "rank": 33 }, { "arena": "models", "category": "codecategories", "elo": 1187, "win_rate": 54.4, "rank": 59 }, { "arena": "models", "category": "dataviz", "elo": 1241, "win_rate": 60.6, "rank": 36 }, { "arena": "models", "category": "gamedev", "elo": 1224, "win_rate": 59.4, "rank": 43 }, { "arena": "models", "category": "svg", "elo": 1219, "win_rate": 62.3, "rank": 20 }, { "arena": "models", "category": "uicomponent", "elo": 1207, "win_rate": 57.3, "rank": 48 }, { "arena": "models", "category": "website", "elo": 1195, "win_rate": 53.5, "rank": 57 } ], "artificial_analysis": { "intelligence_index": 35.3, "coding_index": 37.8, "agentic_index": 26.5 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "openai/gpt-5:batch", "canonical_slug": "openai/gpt-5-2025-08-07", "hugging_face_id": "", "name": "OpenAI: GPT-5 (batch)", "created": 1754587413, "description": "GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000000625" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-2025-08-07/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1106, "win_rate": 41.5, "rank": 86 }, { "arena": "models", "category": "asciiart", "elo": 1173, "win_rate": 49, "rank": 33 }, { "arena": "models", "category": "codecategories", "elo": 1187, "win_rate": 54.4, "rank": 59 }, { "arena": "models", "category": "dataviz", "elo": 1241, "win_rate": 60.6, "rank": 36 }, { "arena": "models", "category": "gamedev", "elo": 1224, "win_rate": 59.4, "rank": 43 }, { "arena": "models", "category": "svg", "elo": 1219, "win_rate": 62.3, "rank": 20 }, { "arena": "models", "category": "uicomponent", "elo": 1207, "win_rate": 57.3, "rank": 48 }, { "arena": "models", "category": "website", "elo": 1195, "win_rate": 53.5, "rank": 57 } ], "artificial_analysis": { "intelligence_index": 35.3, "coding_index": 37.8, "agentic_index": 26.5 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "openai/gpt-5-mini", "canonical_slug": "openai/gpt-5-mini-2025-08-07", "hugging_face_id": "", "name": "OpenAI: GPT-5 Mini", "created": 1754587407, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "web_search": "0.01", "input_cache_read": "0.000000025" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-05-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-mini-2025-08-07/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1088, "win_rate": 36.9, "rank": 88 }, { "arena": "models", "category": "asciiart", "elo": 1154, "win_rate": 44.5, "rank": 43 }, { "arena": "models", "category": "codecategories", "elo": 1134, "win_rate": 43.4, "rank": 82 }, { "arena": "models", "category": "dataviz", "elo": 1150, "win_rate": 43.8, "rank": 72 }, { "arena": "models", "category": "gamedev", "elo": 1168, "win_rate": 46.5, "rank": 67 }, { "arena": "models", "category": "svg", "elo": 1122, "win_rate": 45, "rank": 51 }, { "arena": "models", "category": "uicomponent", "elo": 1134, "win_rate": 42, "rank": 72 }, { "arena": "models", "category": "website", "elo": 1136, "win_rate": 44.2, "rank": 80 } ], "artificial_analysis": { "intelligence_index": 25.8, "coding_index": 15.6, "agentic_index": 19.6 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "openai/gpt-5-mini:batch", "canonical_slug": "openai/gpt-5-mini-2025-08-07", "hugging_face_id": "", "name": "OpenAI: GPT-5 Mini (batch)", "created": 1754587407, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000125", "completion": "0.000001", "web_search": "0.01", "input_cache_read": "0.0000000125" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-05-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-mini-2025-08-07/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1088, "win_rate": 36.9, "rank": 88 }, { "arena": "models", "category": "asciiart", "elo": 1154, "win_rate": 44.5, "rank": 43 }, { "arena": "models", "category": "codecategories", "elo": 1134, "win_rate": 43.4, "rank": 82 }, { "arena": "models", "category": "dataviz", "elo": 1150, "win_rate": 43.8, "rank": 72 }, { "arena": "models", "category": "gamedev", "elo": 1168, "win_rate": 46.5, "rank": 67 }, { "arena": "models", "category": "svg", "elo": 1122, "win_rate": 45, "rank": 51 }, { "arena": "models", "category": "uicomponent", "elo": 1134, "win_rate": 42, "rank": 72 }, { "arena": "models", "category": "website", "elo": 1136, "win_rate": 44.2, "rank": 80 } ], "artificial_analysis": { "intelligence_index": 25.8, "coding_index": 15.6, "agentic_index": 19.6 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "openai/gpt-5-nano", "canonical_slug": "openai/gpt-5-nano-2025-08-07", "hugging_face_id": "", "name": "OpenAI: GPT-5 Nano", "created": 1754587402, "description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000005", "completion": "0.0000004", "web_search": "0.01", "input_cache_read": "0.000000005" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-05-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-nano-2025-08-07/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1016, "win_rate": 36.3, "rank": 101 }, { "arena": "models", "category": "codecategories", "elo": 1103, "win_rate": 48, "rank": 89 }, { "arena": "models", "category": "dataviz", "elo": 1076, "win_rate": 46.2, "rank": 92 }, { "arena": "models", "category": "gamedev", "elo": 1084, "win_rate": 46.5, "rank": 92 }, { "arena": "models", "category": "uicomponent", "elo": 1094, "win_rate": 51.9, "rank": 86 }, { "arena": "models", "category": "website", "elo": 1112, "win_rate": 48.9, "rank": 90 } ] }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "openai/gpt-5-nano:batch", "canonical_slug": "openai/gpt-5-nano-2025-08-07", "hugging_face_id": "", "name": "OpenAI: GPT-5 Nano (batch)", "created": 1754587402, "description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...", "context_length": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000025", "completion": "0.0000002", "web_search": "0.01", "input_cache_read": "0.0000000025" }, "top_provider": { "context_length": 400000, "max_completion_tokens": 128000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-05-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-5-nano-2025-08-07/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1016, "win_rate": 36.3, "rank": 101 }, { "arena": "models", "category": "codecategories", "elo": 1103, "win_rate": 48, "rank": 89 }, { "arena": "models", "category": "dataviz", "elo": 1076, "win_rate": 46.2, "rank": 92 }, { "arena": "models", "category": "gamedev", "elo": 1084, "win_rate": 46.5, "rank": 92 }, { "arena": "models", "category": "uicomponent", "elo": 1094, "win_rate": 51.9, "rank": 86 }, { "arena": "models", "category": "website", "elo": 1112, "win_rate": 48.9, "rank": 90 } ] }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low", "minimal" ], "default_effort": "medium" } }, { "id": "openai/gpt-oss-120b", "canonical_slug": "openai/gpt-oss-120b", "hugging_face_id": "openai/gpt-oss-120b", "name": "OpenAI: gpt-oss-120b", "created": 1754414231, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000003", "completion": "0.00000017", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-oss-120b/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 951, "win_rate": 29.4, "rank": 104 }, { "arena": "models", "category": "codecategories", "elo": 982, "win_rate": 33.4, "rank": 112 }, { "arena": "models", "category": "dataviz", "elo": 1007, "win_rate": 43.6, "rank": 103 }, { "arena": "models", "category": "gamedev", "elo": 1031, "win_rate": 40.5, "rank": 102 }, { "arena": "models", "category": "uicomponent", "elo": 954, "win_rate": 35.7, "rank": 106 }, { "arena": "models", "category": "website", "elo": 979, "win_rate": 32.5, "rank": 114 } ], "artificial_analysis": { "intelligence_index": 24.1, "coding_index": 30.4, "agentic_index": 13.4 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "openai/gpt-oss-20b", "canonical_slug": "openai/gpt-oss-20b", "hugging_face_id": "openai/gpt-oss-20b", "name": "OpenAI: gpt-oss-20b", "created": 1754414229, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000003", "completion": "0.00000013", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-oss-20b/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "dataviz", "elo": 952, "win_rate": 39.7, "rank": 107 }, { "arena": "models", "category": "website", "elo": 864, "win_rate": 27.9, "rank": 122 } ], "artificial_analysis": { "intelligence_index": 15.2, "coding_index": 20.7, "agentic_index": 3.1 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "openai/gpt-oss-20b:free", "canonical_slug": "openai/gpt-oss-20b", "hugging_face_id": "openai/gpt-oss-20b", "name": "OpenAI: gpt-oss-20b (free)", "created": 1754414229, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0", "completion": "0" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-oss-20b/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "dataviz", "elo": 952, "win_rate": 39.7, "rank": 107 }, { "arena": "models", "category": "website", "elo": 864, "win_rate": 27.9, "rank": 122 } ], "artificial_analysis": { "intelligence_index": 15.2, "coding_index": 20.7, "agentic_index": 3.1 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high", "medium", "low" ], "default_effort": "medium" } }, { "id": "anthropic/claude-opus-4.1", "canonical_slug": "anthropic/claude-4.1-opus-20250805", "hugging_face_id": "", "name": "Anthropic: Claude Opus 4.1", "created": 1754411591, "description": "Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000015", "completion": "0.000075", "web_search": "0.01", "input_cache_read": "0.0000015", "input_cache_write": "0.00001875", "input_cache_write_1h": "0.00003" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 32000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.1-opus-20250805/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1204, "win_rate": 52, "rank": 49 }, { "arena": "models", "category": "asciiart", "elo": 1198, "win_rate": 51.4, "rank": 22 }, { "arena": "models", "category": "codecategories", "elo": 1190, "win_rate": 55.7, "rank": 55 }, { "arena": "models", "category": "dataviz", "elo": 1186, "win_rate": 57.1, "rank": 57 }, { "arena": "models", "category": "gamedev", "elo": 1212, "win_rate": 58.8, "rank": 46 }, { "arena": "models", "category": "svg", "elo": 1186, "win_rate": 58.8, "rank": 32 }, { "arena": "models", "category": "uicomponent", "elo": 1191, "win_rate": 57.7, "rank": 59 }, { "arena": "models", "category": "website", "elo": 1188, "win_rate": 55.1, "rank": 60 } ] }, "reasoning": { "mandatory": false } }, { "id": "anthropic/claude-opus-4.1:batch", "canonical_slug": "anthropic/claude-4.1-opus-20250805", "hugging_face_id": "", "name": "Anthropic: Claude Opus 4.1 (batch)", "created": 1754411591, "description": "Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.0000075", "completion": "0.0000375", "web_search": "0.01", "input_cache_read": "0.00000075", "input_cache_write": "0.000009375", "input_cache_write_1h": "0.000015" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 32000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4.1-opus-20250805/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1204, "win_rate": 52, "rank": 49 }, { "arena": "models", "category": "asciiart", "elo": 1198, "win_rate": 51.4, "rank": 22 }, { "arena": "models", "category": "codecategories", "elo": 1190, "win_rate": 55.7, "rank": 55 }, { "arena": "models", "category": "dataviz", "elo": 1186, "win_rate": 57.1, "rank": 57 }, { "arena": "models", "category": "gamedev", "elo": 1212, "win_rate": 58.8, "rank": 46 }, { "arena": "models", "category": "svg", "elo": 1186, "win_rate": 58.8, "rank": 32 }, { "arena": "models", "category": "uicomponent", "elo": 1191, "win_rate": 57.7, "rank": 59 }, { "arena": "models", "category": "website", "elo": 1188, "win_rate": 55.1, "rank": 60 } ] }, "reasoning": { "mandatory": false } }, { "id": "mistralai/codestral-2508", "canonical_slug": "mistralai/codestral-2508", "hugging_face_id": "", "name": "Mistral: Codestral 2508", "created": 1754079630, "description": "Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)", "context_length": 256000, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000009", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 256000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/codestral-2508/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 1026, "win_rate": 38.5, "rank": 105 }, { "arena": "models", "category": "dataviz", "elo": 1037, "win_rate": 41.1, "rank": 98 }, { "arena": "models", "category": "gamedev", "elo": 1006, "win_rate": 36.3, "rank": 108 }, { "arena": "models", "category": "uicomponent", "elo": 1044, "win_rate": 46.7, "rank": 93 }, { "arena": "models", "category": "website", "elo": 1024, "win_rate": 37.8, "rank": 108 }, { "arena": "models", "category": "3d", "elo": 1071, "win_rate": 45.4, "rank": 91 } ] } }, { "id": "qwen/qwen3-coder-30b-a3b-instruct", "canonical_slug": "qwen/qwen3-coder-30b-a3b-instruct", "hugging_face_id": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "name": "Qwen: Qwen3 Coder 30B A3B Instruct", "created": 1753972379, "description": "Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000007", "completion": "0.00000028" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 262144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-coder-30b-a3b-instruct/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "dataviz", "elo": 1103, "win_rate": 54.7, "rank": 89 }, { "arena": "models", "category": "uicomponent", "elo": 1074, "win_rate": 54.1, "rank": 88 }, { "arena": "models", "category": "website", "elo": 1098, "win_rate": 57.1, "rank": 93 } ] } }, { "id": "qwen/qwen3-30b-a3b-instruct-2507", "canonical_slug": "qwen/qwen3-30b-a3b-instruct-2507", "hugging_face_id": "Qwen/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen: Qwen3 30B A3B Instruct 2507", "created": 1753806965, "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000004815", "completion": "0.00000019305" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-30b-a3b-instruct-2507/endpoints" } }, { "id": "z-ai/glm-4.5", "canonical_slug": "z-ai/glm-4.5", "hugging_face_id": "zai-org/GLM-4.5", "name": "Z.ai: GLM 4.5", "created": 1753471347, "description": "GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000006", "completion": "0.0000022", "input_cache_read": "0.00000011" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 98304, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 0.75, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-12-31", "expiration_date": "2026-12-31", "links": { "details": "/api/v1/models/z-ai/glm-4.5/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1221, "win_rate": 59.9, "rank": 43 }, { "arena": "models", "category": "codecategories", "elo": 1184, "win_rate": 54.4, "rank": 62 }, { "arena": "models", "category": "dataviz", "elo": 1179, "win_rate": 53.3, "rank": 62 }, { "arena": "models", "category": "gamedev", "elo": 1185, "win_rate": 54.7, "rank": 57 }, { "arena": "models", "category": "svg", "elo": 1128, "win_rate": 49.4, "rank": 50 }, { "arena": "models", "category": "uicomponent", "elo": 1172, "win_rate": 55.1, "rank": 64 }, { "arena": "models", "category": "website", "elo": 1181, "win_rate": 53.6, "rank": 63 } ] }, "reasoning": { "mandatory": false } }, { "id": "z-ai/glm-4.5-air", "canonical_slug": "z-ai/glm-4.5-air", "hugging_face_id": "zai-org/GLM-4.5-Air", "name": "Z.ai: GLM 4.5 Air", "created": 1753471258, "description": "GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000013", "completion": "0.00000085", "input_cache_read": "0.000000025" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 98304, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 0.75, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/z-ai/glm-4.5-air/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1173, "win_rate": 54.3, "rank": 60 }, { "arena": "models", "category": "codecategories", "elo": 1156, "win_rate": 51.4, "rank": 73 }, { "arena": "models", "category": "dataviz", "elo": 1206, "win_rate": 58.5, "rank": 47 }, { "arena": "models", "category": "gamedev", "elo": 1134, "win_rate": 48.7, "rank": 77 }, { "arena": "models", "category": "svg", "elo": 1106, "win_rate": 49.4, "rank": 56 }, { "arena": "models", "category": "uicomponent", "elo": 1150, "win_rate": 54.5, "rank": 70 }, { "arena": "models", "category": "website", "elo": 1158, "win_rate": 51.2, "rank": 73 } ] }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3-235b-a22b-thinking-2507", "canonical_slug": "qwen/qwen3-235b-a22b-thinking-2507", "hugging_face_id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen: Qwen3 235B A22B Thinking 2507", "created": 1753449557, "description": "Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "pricing": { "prompt": "0.00000023", "completion": "0.0000023" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-235b-a22b-thinking-2507/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1048, "win_rate": 40.4, "rank": 94 }, { "arena": "models", "category": "codecategories", "elo": 1052, "win_rate": 40.8, "rank": 100 }, { "arena": "models", "category": "dataviz", "elo": 969, "win_rate": 32.6, "rank": 106 }, { "arena": "models", "category": "gamedev", "elo": 996, "win_rate": 34.2, "rank": 109 }, { "arena": "models", "category": "uicomponent", "elo": 972, "win_rate": 34, "rank": 105 }, { "arena": "models", "category": "website", "elo": 1064, "win_rate": 42, "rank": 99 } ], "artificial_analysis": { "intelligence_index": 19.9, "coding_index": 22.1, "agentic_index": 3.8 } }, "reasoning": { "mandatory": true } }, { "id": "qwen/qwen3-coder", "canonical_slug": "qwen/qwen3-coder-480b-a35b-07-25", "hugging_face_id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen: Qwen3 Coder 480B A35B", "created": 1753230546, "description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.000001", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-coder-480b-a35b-07-25/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 1162, "win_rate": 61.2, "rank": 69 }, { "arena": "models", "category": "dataviz", "elo": 1102, "win_rate": 54.9, "rank": 90 }, { "arena": "models", "category": "gamedev", "elo": 1138, "win_rate": 58.7, "rank": 74 }, { "arena": "models", "category": "uicomponent", "elo": 1140, "win_rate": 61.4, "rank": 71 }, { "arena": "models", "category": "website", "elo": 1170, "win_rate": 61.7, "rank": 69 } ] } }, { "id": "bytedance/ui-tars-1.5-7b", "canonical_slug": "bytedance/ui-tars-1.5-7b", "hugging_face_id": "ByteDance-Seed/UI-TARS-1.5-7B", "name": "ByteDance: UI-TARS 7B ", "created": 1753205056, "description": "UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...", "context_length": 128000, "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000002", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 2048, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/bytedance/ui-tars-1.5-7b/endpoints" } }, { "id": "google/gemini-2.5-flash-lite", "canonical_slug": "google/gemini-2.5-flash-lite", "hugging_face_id": "", "name": "Google: Gemini 2.5 Flash Lite", "created": 1753200276, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "image": "0.0000001", "audio": "0.0000003", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000004", "input_cache_read": "0.00000001", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65535, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-flash-lite/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "google/gemini-2.5-flash-lite:batch", "canonical_slug": "google/gemini-2.5-flash-lite", "hugging_face_id": "", "name": "Google: Gemini 2.5 Flash Lite (batch)", "created": 1753200276, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "image": "0.00000005", "audio": "0.00000015", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000002", "input_cache_read": "0.00000001" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65535, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-flash-lite/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3-235b-a22b-2507", "canonical_slug": "qwen/qwen3-235b-a22b-07-25", "hugging_face_id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "created": 1753119555, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "pricing": { "prompt": "0.00000009", "completion": "0.00000055" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-235b-a22b-07-25/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1045, "win_rate": 41.1, "rank": 95 }, { "arena": "models", "category": "codecategories", "elo": 1057, "win_rate": 42.6, "rank": 98 }, { "arena": "models", "category": "dataviz", "elo": 1089, "win_rate": 49, "rank": 91 }, { "arena": "models", "category": "gamedev", "elo": 990, "win_rate": 35, "rank": 110 }, { "arena": "models", "category": "uicomponent", "elo": 992, "win_rate": 38.4, "rank": 101 }, { "arena": "models", "category": "website", "elo": 1069, "win_rate": 43.6, "rank": 98 } ] } }, { "id": "moonshotai/kimi-k2", "canonical_slug": "moonshotai/kimi-k2", "hugging_face_id": "moonshotai/Kimi-K2-Instruct", "name": "MoonshotAI: Kimi K2 0711", "created": 1752263252, "description": "Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000057", "completion": "0.0000023" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 100352, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "repetition_penalty", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/moonshotai/kimi-k2/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 1052, "win_rate": 51.7, "rank": 99 }, { "arena": "models", "category": "dataviz", "elo": 1036, "win_rate": 49.4, "rank": 99 }, { "arena": "models", "category": "gamedev", "elo": 1013, "win_rate": 46.4, "rank": 105 }, { "arena": "models", "category": "uicomponent", "elo": 1058, "win_rate": 55.1, "rank": 90 }, { "arena": "models", "category": "website", "elo": 1061, "win_rate": 53.1, "rank": 101 } ] } }, { "id": "cognitivecomputations/dolphin-mistral-24b-venice-edition", "canonical_slug": "venice/uncensored", "hugging_face_id": "cognitivecomputations/Dolphin-Mistral-24B-Venice-Edition", "name": "Venice: Uncensored", "created": 1752094966, "description": "Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000009" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "stop", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-04-30", "expiration_date": null, "links": { "details": "/api/v1/models/venice/uncensored/endpoints" } }, { "id": "tencent/hunyuan-a13b-instruct", "canonical_slug": "tencent/hunyuan-a13b-instruct", "hugging_face_id": "tencent/Hunyuan-A13B-Instruct", "name": "Tencent: Hunyuan A13B Instruct", "created": 1751987664, "description": "Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000014", "completion": "0.00000057" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "reasoning", "response_format", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/tencent/hunyuan-a13b-instruct/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "morph/morph-v3-large", "canonical_slug": "morph/morph-v3-large", "hugging_face_id": "", "name": "Morph: Morph V3 Large", "created": 1751910858, "description": "Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code}...", "context_length": 262144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000009", "completion": "0.0000019" }, "top_provider": { "context_length": 262144, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "logprobs", "max_tokens", "response_format", "stop", "structured_outputs", "temperature", "top_logprobs" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/morph/morph-v3-large/endpoints" } }, { "id": "morph/morph-v3-fast", "canonical_slug": "morph/morph-v3-fast", "hugging_face_id": "", "name": "Morph: Morph V3 Fast", "created": 1751910002, "description": "Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code} {edit_snippet}...", "context_length": 81920, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000008", "completion": "0.0000012" }, "top_provider": { "context_length": 81920, "max_completion_tokens": 38000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "stop", "temperature" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/morph/morph-v3-fast/endpoints" } }, { "id": "baidu/ernie-4.5-vl-424b-a47b", "canonical_slug": "baidu/ernie-4.5-vl-424b-a47b", "hugging_face_id": "baidu/ERNIE-4.5-VL-424B-A47B-PT", "name": "Baidu: ERNIE 4.5 VL 424B A47B ", "created": 1751300903, "description": "ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...", "context_length": 123000, "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000042", "completion": "0.00000125" }, "top_provider": { "context_length": 123000, "max_completion_tokens": 16000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "seed", "stop", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/baidu/ernie-4.5-vl-424b-a47b/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "mistralai/mistral-small-3.2-24b-instruct", "canonical_slug": "mistralai/mistral-small-3.2-24b-instruct-2506", "hugging_face_id": "mistralai/Mistral-Small-3.2-24B-Instruct-2506", "name": "Mistral: Mistral Small 3.2 24B", "created": 1750443016, "description": "Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...", "context_length": 256000, "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.00000009375", "completion": "0.00000025" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-small-3.2-24b-instruct-2506/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 926, "win_rate": 39.8, "rank": 116 }, { "arena": "models", "category": "dataviz", "elo": 947, "win_rate": 43.3, "rank": 109 }, { "arena": "models", "category": "gamedev", "elo": 927, "win_rate": 39.4, "rank": 117 }, { "arena": "models", "category": "uicomponent", "elo": 934, "win_rate": 40.5, "rank": 109 }, { "arena": "models", "category": "website", "elo": 907, "win_rate": 38.3, "rank": 119 } ] } }, { "id": "minimax/minimax-m1", "canonical_slug": "minimax/minimax-m1", "hugging_face_id": "", "name": "MiniMax: MiniMax M1", "created": 1750200414, "description": "MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000055", "completion": "0.0000022" }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 40000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-m1/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "google/gemini-2.5-flash", "canonical_slug": "google/gemini-2.5-flash", "hugging_face_id": "", "name": "Google: Gemini 2.5 Flash", "created": 1750172488, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65535, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-flash/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1122, "win_rate": 47.8, "rank": 83 }, { "arena": "models", "category": "codecategories", "elo": 1123, "win_rate": 46.9, "rank": 86 }, { "arena": "models", "category": "dataviz", "elo": 1154, "win_rate": 49.1, "rank": 71 }, { "arena": "models", "category": "gamedev", "elo": 1104, "win_rate": 44.5, "rank": 88 }, { "arena": "models", "category": "uicomponent", "elo": 1120, "win_rate": 48.8, "rank": 80 }, { "arena": "models", "category": "website", "elo": 1127, "win_rate": 47.1, "rank": 86 }, { "arena": "models", "category": "svg", "elo": 1053, "win_rate": 42, "rank": 67 } ] }, "reasoning": { "mandatory": false } }, { "id": "google/gemini-2.5-flash:batch", "canonical_slug": "google/gemini-2.5-flash", "hugging_face_id": "", "name": "Google: Gemini 2.5 Flash (batch)", "created": 1750172488, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "audio": "0.0000005", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.00000003" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65535, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-flash/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1122, "win_rate": 47.8, "rank": 83 }, { "arena": "models", "category": "codecategories", "elo": 1123, "win_rate": 46.9, "rank": 86 }, { "arena": "models", "category": "dataviz", "elo": 1154, "win_rate": 49.1, "rank": 71 }, { "arena": "models", "category": "gamedev", "elo": 1104, "win_rate": 44.5, "rank": 88 }, { "arena": "models", "category": "uicomponent", "elo": 1120, "win_rate": 48.8, "rank": 80 }, { "arena": "models", "category": "website", "elo": 1127, "win_rate": 47.1, "rank": 86 }, { "arena": "models", "category": "svg", "elo": 1053, "win_rate": 42, "rank": 67 } ] }, "reasoning": { "mandatory": false } }, { "id": "google/gemini-2.5-pro", "canonical_slug": "google/gemini-2.5-pro", "hugging_face_id": "", "name": "Google: Gemini 2.5 Pro", "created": 1750169544, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-pro/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1131, "win_rate": 50.6, "rank": 75 }, { "arena": "models", "category": "codecategories", "elo": 1171, "win_rate": 57.5, "rank": 65 }, { "arena": "models", "category": "dataviz", "elo": 1241, "win_rate": 68.2, "rank": 35 }, { "arena": "models", "category": "gamedev", "elo": 1150, "win_rate": 54.2, "rank": 70 }, { "arena": "models", "category": "uicomponent", "elo": 1167, "win_rate": 57.5, "rank": 67 }, { "arena": "models", "category": "website", "elo": 1178, "win_rate": 58.4, "rank": 64 } ], "artificial_analysis": { "intelligence_index": 25.9, "coding_index": 33.3, "agentic_index": 7.2 } }, "reasoning": { "mandatory": true } }, { "id": "google/gemini-2.5-pro:batch", "canonical_slug": "google/gemini-2.5-pro", "hugging_face_id": "", "name": "Google: Gemini 2.5 Pro (batch)", "created": 1750169544, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.000000125", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-pro/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1131, "win_rate": 50.6, "rank": 75 }, { "arena": "models", "category": "codecategories", "elo": 1171, "win_rate": 57.5, "rank": 65 }, { "arena": "models", "category": "dataviz", "elo": 1241, "win_rate": 68.2, "rank": 35 }, { "arena": "models", "category": "gamedev", "elo": 1150, "win_rate": 54.2, "rank": 70 }, { "arena": "models", "category": "uicomponent", "elo": 1167, "win_rate": 57.5, "rank": 67 }, { "arena": "models", "category": "website", "elo": 1178, "win_rate": 58.4, "rank": 64 } ], "artificial_analysis": { "intelligence_index": 25.9, "coding_index": 33.3, "agentic_index": 7.2 } }, "reasoning": { "mandatory": true } }, { "id": "openai/o3-pro", "canonical_slug": "openai/o3-pro-2025-06-10", "hugging_face_id": "", "name": "OpenAI: o3 Pro", "created": 1749598352, "description": "The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "file", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00002", "completion": "0.00008", "web_search": "0.01" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-pro-2025-06-10/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "openai/o3-pro:batch", "canonical_slug": "openai/o3-pro-2025-06-10", "hugging_face_id": "", "name": "OpenAI: o3 Pro (batch)", "created": 1749598352, "description": "The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "file", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00004", "web_search": "0.01" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-pro-2025-06-10/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "google/gemini-2.5-pro-preview", "canonical_slug": "google/gemini-2.5-pro-preview-06-05", "hugging_face_id": "", "name": "Google: Gemini 2.5 Pro Preview 06-05", "created": 1749137257, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-pro-preview-06-05/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "deepseek/deepseek-r1-0528", "canonical_slug": "deepseek/deepseek-r1-0528", "hugging_face_id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek: R1 0528", "created": 1748455170, "description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...", "context_length": 163840, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-r1" }, "pricing": { "prompt": "0.0000005", "completion": "0.00000215", "input_cache_read": "0.00000035" }, "top_provider": { "context_length": 163840, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-r1-0528/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1162, "win_rate": 53.5, "rank": 65 }, { "arena": "models", "category": "codecategories", "elo": 1158, "win_rate": 52.6, "rank": 72 }, { "arena": "models", "category": "dataviz", "elo": 1202, "win_rate": 61, "rank": 48 }, { "arena": "models", "category": "gamedev", "elo": 1137, "win_rate": 49.6, "rank": 75 }, { "arena": "models", "category": "svg", "elo": 1072, "win_rate": 47.2, "rank": 61 }, { "arena": "models", "category": "uicomponent", "elo": 1130, "win_rate": 55.1, "rank": 73 }, { "arena": "models", "category": "website", "elo": 1160, "win_rate": 52.6, "rank": 71 } ] }, "reasoning": { "mandatory": true } }, { "id": "anthropic/claude-opus-4", "canonical_slug": "anthropic/claude-4-opus-20250522", "hugging_face_id": "", "name": "Anthropic: Claude Opus 4", "created": 1747931245, "description": "Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000015", "completion": "0.000075", "web_search": "0.01", "input_cache_read": "0.0000015", "input_cache_write": "0.00001875", "input_cache_write_1h": "0.00003" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "stop", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4-opus-20250522/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1189, "win_rate": 57.7, "rank": 54 }, { "arena": "models", "category": "codecategories", "elo": 1180, "win_rate": 55.6, "rank": 64 }, { "arena": "models", "category": "dataviz", "elo": 1165, "win_rate": 58, "rank": 66 }, { "arena": "models", "category": "gamedev", "elo": 1210, "win_rate": 60.2, "rank": 48 }, { "arena": "models", "category": "svg", "elo": 1161, "win_rate": 56.5, "rank": 44 }, { "arena": "models", "category": "uicomponent", "elo": 1180, "win_rate": 59.5, "rank": 61 }, { "arena": "models", "category": "website", "elo": 1176, "win_rate": 54.5, "rank": 65 } ] }, "reasoning": { "mandatory": false } }, { "id": "anthropic/claude-sonnet-4", "canonical_slug": "anthropic/claude-4-sonnet-20250522", "hugging_face_id": "", "name": "Anthropic: Claude Sonnet 4", "created": 1747930371, "description": "Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...", "context_length": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-4-sonnet-20250522/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1191, "win_rate": 58, "rank": 52 }, { "arena": "models", "category": "codecategories", "elo": 1160, "win_rate": 53.4, "rank": 71 }, { "arena": "models", "category": "dataviz", "elo": 1178, "win_rate": 56.3, "rank": 63 }, { "arena": "models", "category": "gamedev", "elo": 1178, "win_rate": 55.2, "rank": 60 }, { "arena": "models", "category": "svg", "elo": 1114, "win_rate": 50, "rank": 54 }, { "arena": "models", "category": "uicomponent", "elo": 1156, "win_rate": 58.1, "rank": 69 }, { "arena": "models", "category": "website", "elo": 1157, "win_rate": 52.3, "rank": 74 } ], "artificial_analysis": { "intelligence_index": 29.8, "coding_index": 37.6, "agentic_index": 17.6 } }, "reasoning": { "mandatory": false } }, { "id": "google/gemma-3n-e4b-it", "canonical_slug": "google/gemma-3n-e4b-it", "hugging_face_id": "google/gemma-3n-E4B-it", "name": "Google: Gemma 3n 4B", "created": 1747776824, "description": "Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000006", "completion": "0.00000012" }, "top_provider": { "context_length": 32768, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-3n-e4b-it/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 3.2, "agentic_index": null } } }, { "id": "mistralai/mistral-medium-3", "canonical_slug": "mistralai/mistral-medium-3", "hugging_face_id": "", "name": "Mistral: Mistral Medium 3", "created": 1746627341, "description": "Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...", "context_length": 131072, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000004", "completion": "0.000002", "input_cache_read": "0.00000004" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-medium-3/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1134, "win_rate": 54.7, "rank": 72 }, { "arena": "models", "category": "codecategories", "elo": 1088, "win_rate": 48.1, "rank": 93 }, { "arena": "models", "category": "dataviz", "elo": 1053, "win_rate": 46, "rank": 96 }, { "arena": "models", "category": "gamedev", "elo": 1056, "win_rate": 45.2, "rank": 97 }, { "arena": "models", "category": "uicomponent", "elo": 1052, "win_rate": 49.6, "rank": 92 }, { "arena": "models", "category": "website", "elo": 1090, "win_rate": 47.6, "rank": 96 } ] } }, { "id": "google/gemini-2.5-pro-preview-05-06", "canonical_slug": "google/gemini-2.5-pro-preview-03-25", "hugging_face_id": "", "name": "Google: Gemini 2.5 Pro Preview 05-06", "created": 1746578513, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "context_length": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 65535, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemini-2.5-pro-preview-03-25/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "arcee-ai/virtuoso-large", "canonical_slug": "arcee-ai/virtuoso-large", "hugging_face_id": "", "name": "Arcee AI: Virtuoso Large", "created": 1746478885, "description": "Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000075", "completion": "0.0000012" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 64000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/arcee-ai/virtuoso-large/endpoints" } }, { "id": "meta-llama/llama-guard-4-12b", "canonical_slug": "meta-llama/llama-guard-4-12b", "hugging_face_id": "meta-llama/Llama-Guard-4-12B", "name": "Meta: Llama Guard 4 12B", "created": 1745975193, "description": "Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...", "context_length": 1048576, "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000018", "completion": "0.00000018" }, "top_provider": { "context_length": 163840, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-guard-4-12b/endpoints" } }, { "id": "qwen/qwen3-30b-a3b", "canonical_slug": "qwen/qwen3-30b-a3b-04-28", "hugging_face_id": "Qwen/Qwen3-30B-A3B", "name": "Qwen: Qwen3 30B A3B", "created": 1745878604, "description": "Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "pricing": { "prompt": "0.00000013", "completion": "0.00000052" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-30b-a3b-04-28/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 961, "win_rate": 37.5, "rank": 115 }, { "arena": "models", "category": "dataviz", "elo": 985, "win_rate": 39, "rank": 105 }, { "arena": "models", "category": "gamedev", "elo": 937, "win_rate": 33.9, "rank": 115 }, { "arena": "models", "category": "uicomponent", "elo": 973, "win_rate": 42.4, "rank": 104 }, { "arena": "models", "category": "website", "elo": 966, "win_rate": 37.7, "rank": 117 } ] }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "qwen/qwen3-8b", "canonical_slug": "qwen/qwen3-8b-04-28", "hugging_face_id": "Qwen/Qwen3-8B", "name": "Qwen: Qwen3 8B", "created": 1745876632, "description": "Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "pricing": { "prompt": "0.000000117", "completion": "0.000000455" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": 0.6, "top_p": 0.95, "top_k": 20, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-8b-04-28/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 8.3, "coding_index": 9, "agentic_index": 1.6 } }, "reasoning": { "mandatory": false, "default_enabled": true } }, { "id": "qwen/qwen3-14b", "canonical_slug": "qwen/qwen3-14b-04-28", "hugging_face_id": "Qwen/Qwen3-14B", "name": "Qwen: Qwen3 14B", "created": 1745876478, "description": "Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "pricing": { "prompt": "0.00000012", "completion": "0.00000024" }, "top_provider": { "context_length": 40960, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-14b-04-28/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 10.4, "coding_index": 13.8, "agentic_index": 1.9 } }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3-32b", "canonical_slug": "qwen/qwen3-32b-04-28", "hugging_face_id": "Qwen/Qwen3-32B", "name": "Qwen: Qwen3 32B", "created": 1745875945, "description": "Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "pricing": { "prompt": "0.00000008", "completion": "0.00000028" }, "top_provider": { "context_length": 40960, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-32b-04-28/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 11.4, "coding_index": 15.3, "agentic_index": 1.8 } }, "reasoning": { "mandatory": false } }, { "id": "qwen/qwen3-235b-a22b", "canonical_slug": "qwen/qwen3-235b-a22b-04-28", "hugging_face_id": "Qwen/Qwen3-235B-A22B", "name": "Qwen: Qwen3 235B A22B", "created": 1745875757, "description": "Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "pricing": { "prompt": "0.000000455", "completion": "0.00000182" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "response_format", "seed", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen3-235b-a22b-04-28/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 909, "win_rate": 24.5, "rank": 107 }, { "arena": "models", "category": "codecategories", "elo": 1021, "win_rate": 38.2, "rank": 107 }, { "arena": "models", "category": "dataviz", "elo": 1015, "win_rate": 40, "rank": 101 }, { "arena": "models", "category": "gamedev", "elo": 966, "win_rate": 32.9, "rank": 112 }, { "arena": "models", "category": "uicomponent", "elo": 986, "win_rate": 38.6, "rank": 103 }, { "arena": "models", "category": "website", "elo": 1042, "win_rate": 40.4, "rank": 105 } ] }, "reasoning": { "mandatory": false } }, { "id": "openai/o4-mini-high", "canonical_slug": "openai/o4-mini-high-2025-04-16", "hugging_face_id": "", "name": "OpenAI: o4 Mini High", "created": 1744824212, "description": "OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.000000275" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o4-mini-high-2025-04-16/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "high" ], "default_effort": "high" } }, { "id": "openai/o4-mini-high:batch", "canonical_slug": "openai/o4-mini-high-2025-04-16", "hugging_face_id": "", "name": "OpenAI: o4 Mini High (batch)", "created": 1744824212, "description": "OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.0000001375" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o4-mini-high-2025-04-16/endpoints" }, "reasoning": { "mandatory": true, "supported_efforts": [ "high" ], "default_effort": "high" } }, { "id": "openai/o3", "canonical_slug": "openai/o3-2025-04-16", "hugging_face_id": "", "name": "OpenAI: o3", "created": 1744823457, "description": "o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.01", "input_cache_read": "0.0000005" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-2025-04-16/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 1039, "win_rate": 51.9, "rank": 103 }, { "arena": "models", "category": "dataviz", "elo": 1200, "win_rate": 48.1, "rank": 50 }, { "arena": "models", "category": "gamedev", "elo": 1073, "win_rate": 56.9, "rank": 94 }, { "arena": "models", "category": "uicomponent", "elo": 1044, "win_rate": 53.3, "rank": 94 }, { "arena": "models", "category": "website", "elo": 1047, "win_rate": 53.8, "rank": 104 } ] }, "reasoning": { "mandatory": false } }, { "id": "openai/o3:batch", "canonical_slug": "openai/o3-2025-04-16", "hugging_face_id": "", "name": "OpenAI: o3 (batch)", "created": 1744823457, "description": "o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000004", "web_search": "0.01", "input_cache_read": "0.00000025" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-2025-04-16/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 1039, "win_rate": 51.9, "rank": 103 }, { "arena": "models", "category": "dataviz", "elo": 1200, "win_rate": 48.1, "rank": 50 }, { "arena": "models", "category": "gamedev", "elo": 1073, "win_rate": 56.9, "rank": 94 }, { "arena": "models", "category": "uicomponent", "elo": 1044, "win_rate": 53.3, "rank": 94 }, { "arena": "models", "category": "website", "elo": 1047, "win_rate": 53.8, "rank": 104 } ] }, "reasoning": { "mandatory": false } }, { "id": "openai/o4-mini", "canonical_slug": "openai/o4-mini-2025-04-16", "hugging_face_id": "", "name": "OpenAI: o4 Mini", "created": 1744820942, "description": "OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.000000275" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o4-mini-2025-04-16/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 904, "win_rate": 34, "rank": 108 }, { "arena": "models", "category": "codecategories", "elo": 995, "win_rate": 46.4, "rank": 110 }, { "arena": "models", "category": "dataviz", "elo": 1010, "win_rate": 50, "rank": 102 }, { "arena": "models", "category": "gamedev", "elo": 1042, "win_rate": 50, "rank": 100 }, { "arena": "models", "category": "uicomponent", "elo": 1011, "win_rate": 46.9, "rank": 99 }, { "arena": "models", "category": "website", "elo": 996, "win_rate": 47.1, "rank": 112 } ] }, "reasoning": { "mandatory": false } }, { "id": "openai/o4-mini:batch", "canonical_slug": "openai/o4-mini-2025-04-16", "hugging_face_id": "", "name": "OpenAI: o4 Mini (batch)", "created": 1744820942, "description": "OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.0000001375" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o4-mini-2025-04-16/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 904, "win_rate": 34, "rank": 108 }, { "arena": "models", "category": "codecategories", "elo": 995, "win_rate": 46.4, "rank": 110 }, { "arena": "models", "category": "dataviz", "elo": 1010, "win_rate": 50, "rank": 102 }, { "arena": "models", "category": "gamedev", "elo": 1042, "win_rate": 50, "rank": 100 }, { "arena": "models", "category": "uicomponent", "elo": 1011, "win_rate": 46.9, "rank": 99 }, { "arena": "models", "category": "website", "elo": 996, "win_rate": 47.1, "rank": 112 } ] }, "reasoning": { "mandatory": false } }, { "id": "openai/gpt-4.1", "canonical_slug": "openai/gpt-4.1-2025-04-14", "hugging_face_id": "", "name": "OpenAI: GPT-4.1", "created": 1744651385, "description": "GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...", "context_length": 1047576, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.01", "input_cache_read": "0.0000005" }, "top_provider": { "context_length": 1047576, "max_completion_tokens": 32768, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_completion_tokens", "max_tokens", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4.1-2025-04-14/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 900, "win_rate": 30.9, "rank": 109 }, { "arena": "models", "category": "codecategories", "elo": 1044, "win_rate": 50.9, "rank": 102 }, { "arena": "models", "category": "dataviz", "elo": 1122, "win_rate": 59.5, "rank": 82 }, { "arena": "models", "category": "gamedev", "elo": 1118, "win_rate": 59.1, "rank": 84 }, { "arena": "models", "category": "uicomponent", "elo": 1028, "win_rate": 49.7, "rank": 98 }, { "arena": "models", "category": "website", "elo": 1050, "win_rate": 52.3, "rank": 103 } ] } }, { "id": "openai/gpt-4.1:batch", "canonical_slug": "openai/gpt-4.1-2025-04-14", "hugging_face_id": "", "name": "OpenAI: GPT-4.1 (batch)", "created": 1744651385, "description": "GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...", "context_length": 1047576, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000004", "web_search": "0.01", "input_cache_read": "0.00000025" }, "top_provider": { "context_length": 1047576, "max_completion_tokens": 32768, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4.1-2025-04-14/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 900, "win_rate": 30.9, "rank": 109 }, { "arena": "models", "category": "codecategories", "elo": 1044, "win_rate": 50.9, "rank": 102 }, { "arena": "models", "category": "dataviz", "elo": 1122, "win_rate": 59.5, "rank": 82 }, { "arena": "models", "category": "gamedev", "elo": 1118, "win_rate": 59.1, "rank": 84 }, { "arena": "models", "category": "uicomponent", "elo": 1028, "win_rate": 49.7, "rank": 98 }, { "arena": "models", "category": "website", "elo": 1050, "win_rate": 52.3, "rank": 103 } ] } }, { "id": "openai/gpt-4.1-mini", "canonical_slug": "openai/gpt-4.1-mini-2025-04-14", "hugging_face_id": "", "name": "OpenAI: GPT-4.1 Mini", "created": 1744651381, "description": "GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...", "context_length": 1047576, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000004", "completion": "0.0000016", "web_search": "0.01", "input_cache_read": "0.0000001" }, "top_provider": { "context_length": 1047576, "max_completion_tokens": 32768, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_completion_tokens", "max_tokens", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4.1-mini-2025-04-14/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 887, "win_rate": 30.5, "rank": 110 }, { "arena": "models", "category": "codecategories", "elo": 1012, "win_rate": 47.5, "rank": 109 }, { "arena": "models", "category": "dataviz", "elo": 1052, "win_rate": 49.2, "rank": 97 }, { "arena": "models", "category": "gamedev", "elo": 1109, "win_rate": 58.5, "rank": 87 }, { "arena": "models", "category": "uicomponent", "elo": 989, "win_rate": 45.4, "rank": 102 }, { "arena": "models", "category": "website", "elo": 1009, "win_rate": 47.8, "rank": 111 } ], "artificial_analysis": { "intelligence_index": 14.8, "coding_index": 20.2, "agentic_index": 1.8 } } }, { "id": "openai/gpt-4.1-mini:batch", "canonical_slug": "openai/gpt-4.1-mini-2025-04-14", "hugging_face_id": "", "name": "OpenAI: GPT-4.1 Mini (batch)", "created": 1744651381, "description": "GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...", "context_length": 1047576, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000008", "web_search": "0.01", "input_cache_read": "0.00000005" }, "top_provider": { "context_length": 1047576, "max_completion_tokens": 32768, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4.1-mini-2025-04-14/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 887, "win_rate": 30.5, "rank": 110 }, { "arena": "models", "category": "codecategories", "elo": 1012, "win_rate": 47.5, "rank": 109 }, { "arena": "models", "category": "dataviz", "elo": 1052, "win_rate": 49.2, "rank": 97 }, { "arena": "models", "category": "gamedev", "elo": 1109, "win_rate": 58.5, "rank": 87 }, { "arena": "models", "category": "uicomponent", "elo": 989, "win_rate": 45.4, "rank": 102 }, { "arena": "models", "category": "website", "elo": 1009, "win_rate": 47.8, "rank": 111 } ], "artificial_analysis": { "intelligence_index": 14.8, "coding_index": 20.2, "agentic_index": 1.8 } } }, { "id": "openai/gpt-4.1-nano", "canonical_slug": "openai/gpt-4.1-nano-2025-04-14", "hugging_face_id": "", "name": "OpenAI: GPT-4.1 Nano", "created": 1744651369, "description": "For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...", "context_length": 1047576, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "web_search": "0.01", "input_cache_read": "0.000000025" }, "top_provider": { "context_length": 1047576, "max_completion_tokens": 32768, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_completion_tokens", "max_tokens", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4.1-nano-2025-04-14/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 975, "win_rate": 46, "rank": 103 }, { "arena": "models", "category": "codecategories", "elo": 982, "win_rate": 47.3, "rank": 111 }, { "arena": "models", "category": "dataviz", "elo": 910, "win_rate": 41.1, "rank": 114 }, { "arena": "models", "category": "gamedev", "elo": 1009, "win_rate": 49.6, "rank": 106 }, { "arena": "models", "category": "uicomponent", "elo": 943, "win_rate": 43.9, "rank": 107 }, { "arena": "models", "category": "website", "elo": 984, "win_rate": 48.1, "rank": 113 } ], "artificial_analysis": { "intelligence_index": 9.6, "coding_index": 11.1, "agentic_index": 1.2 } } }, { "id": "openai/gpt-4.1-nano:batch", "canonical_slug": "openai/gpt-4.1-nano-2025-04-14", "hugging_face_id": "", "name": "OpenAI: GPT-4.1 Nano (batch)", "created": 1744651369, "description": "For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...", "context_length": 1047576, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "web_search": "0.01", "input_cache_read": "0.0000000125" }, "top_provider": { "context_length": 1047576, "max_completion_tokens": 32768, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4.1-nano-2025-04-14/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 975, "win_rate": 46, "rank": 103 }, { "arena": "models", "category": "codecategories", "elo": 982, "win_rate": 47.3, "rank": 111 }, { "arena": "models", "category": "dataviz", "elo": 910, "win_rate": 41.1, "rank": 114 }, { "arena": "models", "category": "gamedev", "elo": 1009, "win_rate": 49.6, "rank": 106 }, { "arena": "models", "category": "uicomponent", "elo": 943, "win_rate": 43.9, "rank": 107 }, { "arena": "models", "category": "website", "elo": 984, "win_rate": 48.1, "rank": 113 } ], "artificial_analysis": { "intelligence_index": 9.6, "coding_index": 11.1, "agentic_index": 1.2 } } }, { "id": "meta-llama/llama-4-maverick", "canonical_slug": "meta-llama/llama-4-maverick-17b-128e-instruct", "hugging_face_id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct", "name": "Meta: Llama 4 Maverick", "created": 1743881822, "description": "Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...", "context_length": 1048576, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000008" }, "top_provider": { "context_length": 1048576, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-4-maverick-17b-128e-instruct/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 950, "win_rate": 40.2, "rank": 105 }, { "arena": "models", "category": "codecategories", "elo": 899, "win_rate": 35.8, "rank": 118 }, { "arena": "models", "category": "dataviz", "elo": 900, "win_rate": 38.4, "rank": 115 }, { "arena": "models", "category": "gamedev", "elo": 876, "win_rate": 33.7, "rank": 118 }, { "arena": "models", "category": "uicomponent", "elo": 927, "win_rate": 40.8, "rank": 110 }, { "arena": "models", "category": "website", "elo": 882, "win_rate": 34.4, "rank": 121 } ], "artificial_analysis": { "intelligence_index": 14.5, "coding_index": 16.3, "agentic_index": 1.2 } } }, { "id": "meta-llama/llama-4-scout", "canonical_slug": "meta-llama/llama-4-scout-17b-16e-instruct", "hugging_face_id": "meta-llama/Llama-4-Scout-17B-16E-Instruct", "name": "Meta: Llama 4 Scout", "created": 1743881519, "description": "Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...", "context_length": 1310720, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000003" }, "top_provider": { "context_length": 327680, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-4-scout-17b-16e-instruct/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "codecategories", "elo": 809, "win_rate": 26.6, "rank": 121 }, { "arena": "models", "category": "dataviz", "elo": 914, "win_rate": 39.3, "rank": 113 }, { "arena": "models", "category": "gamedev", "elo": 812, "win_rate": 27.4, "rank": 120 }, { "arena": "models", "category": "uicomponent", "elo": 795, "win_rate": 25.5, "rank": 115 }, { "arena": "models", "category": "website", "elo": 761, "win_rate": 22.7, "rank": 127 } ], "artificial_analysis": { "intelligence_index": 10.3, "coding_index": 8.2, "agentic_index": 1.1 } } }, { "id": "deepseek/deepseek-chat-v3-0324", "canonical_slug": "deepseek/deepseek-chat-v3-0324", "hugging_face_id": "deepseek-ai/DeepSeek-V3-0324", "name": "DeepSeek: DeepSeek V3 0324", "created": 1742824755, "description": "DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...", "context_length": 163840, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "pricing": { "prompt": "0.00000027", "completion": "0.00000112", "input_cache_read": "0.000000135" }, "top_provider": { "context_length": 163840, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-07-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-chat-v3-0324/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 15.2, "coding_index": 21.2, "agentic_index": 1.6 } } }, { "id": "openai/o1-pro", "canonical_slug": "openai/o1-pro", "hugging_face_id": "", "name": "OpenAI: o1-pro", "created": 1742423211, "description": "The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00015", "completion": "0.0006", "web_search": "0.01" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o1-pro/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "openai/o1-pro:batch", "canonical_slug": "openai/o1-pro", "hugging_face_id": "", "name": "OpenAI: o1-pro (batch)", "created": 1742423211, "description": "The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000075", "completion": "0.0003", "web_search": "0.01" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o1-pro/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "mistralai/mistral-small-3.1-24b-instruct", "canonical_slug": "mistralai/mistral-small-3.1-24b-instruct-2503", "hugging_face_id": "mistralai/Mistral-Small-3.1-24B-Instruct-2503", "name": "Mistral: Mistral Small 3.1 24B", "created": 1742238937, "description": "Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...", "context_length": 128000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.000000351", "completion": "0.000000555" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "seed", "stop", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-small-3.1-24b-instruct-2503/endpoints" } }, { "id": "google/gemma-3-4b-it", "canonical_slug": "google/gemma-3-4b-it", "hugging_face_id": "google/gemma-3-4b-it", "name": "Google: Gemma 3 4B", "created": 1741905510, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "context_length": 131072, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "pricing": { "prompt": "0.00000005", "completion": "0.0000001" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-3-4b-it/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 2.7, "agentic_index": null } } }, { "id": "google/gemma-3-12b-it", "canonical_slug": "google/gemma-3-12b-it", "hugging_face_id": "google/gemma-3-12b-it", "name": "Google: Gemma 3 12B", "created": 1741902625, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "context_length": 131072, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "pricing": { "prompt": "0.00000005", "completion": "0.00000015" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-3-12b-it/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 5.5, "coding_index": 5.8, "agentic_index": 0.3 } } }, { "id": "cohere/command-a", "canonical_slug": "cohere/command-a-03-2025", "hugging_face_id": "CohereForAI/c4ai-command-a-03-2025", "name": "Cohere: Command A", "created": 1741894342, "description": "Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...", "context_length": 256000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.00001" }, "top_provider": { "context_length": 256000, "max_completion_tokens": 8192, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/cohere/command-a-03-2025/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 22.8, "coding_index": 27.8, "agentic_index": 9.2 } } }, { "id": "rekaai/reka-flash-3", "canonical_slug": "rekaai/reka-flash-3", "hugging_face_id": "RekaAI/reka-flash-3", "name": "Reka Flash 3", "created": 1741812813, "description": "Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...", "context_length": 65536, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000001", "completion": "0.0000002" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_tokens", "presence_penalty", "reasoning", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2025-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/rekaai/reka-flash-3/endpoints" }, "reasoning": { "mandatory": true } }, { "id": "google/gemma-3-27b-it", "canonical_slug": "google/gemma-3-27b-it", "hugging_face_id": "google/gemma-3-27b-it", "name": "Google: Gemma 3 27B", "created": 1741756359, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "context_length": 262144, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "pricing": { "prompt": "0.00000008", "completion": "0.00000045", "input_cache_read": "0.00000004" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-3-27b-it/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 7.4, "coding_index": 10.1, "agentic_index": 0.3 } } }, { "id": "thedrummer/skyfall-36b-v2", "canonical_slug": "thedrummer/skyfall-36b-v2", "hugging_face_id": "TheDrummer/Skyfall-36B-v2", "name": "TheDrummer: Skyfall 36B V2", "created": 1741636566, "description": "Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000055", "completion": "0.0000008", "input_cache_read": "0.00000025" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/thedrummer/skyfall-36b-v2/endpoints" } }, { "id": "perplexity/sonar-reasoning-pro", "canonical_slug": "perplexity/sonar-reasoning-pro", "hugging_face_id": "", "name": "Perplexity: Sonar Reasoning Pro", "created": 1741313308, "description": "Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...", "context_length": 128000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": "deepseek-r1" }, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.005" }, "top_provider": { "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "temperature", "top_k", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/perplexity/sonar-reasoning-pro/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "perplexity/sonar-pro", "canonical_slug": "perplexity/sonar-pro", "hugging_face_id": "", "name": "Perplexity: Sonar Pro", "created": 1741312423, "description": "Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...", "context_length": 200000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.005" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 8000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "temperature", "top_k", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/perplexity/sonar-pro/endpoints" } }, { "id": "perplexity/sonar-deep-research", "canonical_slug": "perplexity/sonar-deep-research", "hugging_face_id": "", "name": "Perplexity: Sonar Deep Research", "created": 1741311246, "description": "Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": "deepseek-r1" }, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.005", "internal_reasoning": "0.000003" }, "top_provider": { "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "temperature", "top_k", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/perplexity/sonar-deep-research/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "mistralai/mistral-saba", "canonical_slug": "mistralai/mistral-saba-2502", "hugging_face_id": "", "name": "Mistral: Saba", "created": 1739803239, "description": "Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...", "context_length": 32768, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000006", "input_cache_read": "0.00000002" }, "top_provider": { "context_length": 32768, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2024-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-saba-2502/endpoints" } }, { "id": "openai/o3-mini-high", "canonical_slug": "openai/o3-mini-high-2025-01-31", "hugging_face_id": "", "name": "OpenAI: o3 Mini High", "created": 1739372611, "description": "OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...", "context_length": 200000, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.00000055" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-mini-high-2025-01-31/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 15.7, "coding_index": 16.3, "agentic_index": 1.7 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high" ], "default_effort": "high" } }, { "id": "openai/o3-mini-high:batch", "canonical_slug": "openai/o3-mini-high-2025-01-31", "hugging_face_id": "", "name": "OpenAI: o3 Mini High (batch)", "created": 1739372611, "description": "OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...", "context_length": 200000, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.000000275" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-mini-high-2025-01-31/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 15.7, "coding_index": 16.3, "agentic_index": 1.7 } }, "reasoning": { "mandatory": true, "supported_efforts": [ "high" ], "default_effort": "high" } }, { "id": "aion-labs/aion-rp-llama-3.1-8b", "canonical_slug": "aion-labs/aion-rp-llama-3.1-8b", "hugging_face_id": "", "name": "AionLabs: Aion-RP 1.0 (8B)", "created": 1738696718, "description": "Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000008", "completion": "0.0000016" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/aion-labs/aion-rp-llama-3.1-8b/endpoints" } }, { "id": "qwen/qwen2.5-vl-72b-instruct", "canonical_slug": "qwen/qwen2.5-vl-72b-instruct", "hugging_face_id": "Qwen/Qwen2.5-VL-72B-Instruct", "name": "Qwen: Qwen2.5 VL 72B Instruct", "created": 1738410311, "description": "Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.", "context_length": 128000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.0000008", "completion": "0.000001", "input_cache_read": "0.0000004" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 128000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen2.5-vl-72b-instruct/endpoints" } }, { "id": "qwen/qwen-plus", "canonical_slug": "qwen/qwen-plus-2025-01-25", "hugging_face_id": "", "name": "Qwen: Qwen-Plus", "created": 1738409840, "description": "Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.", "context_length": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "pricing": { "prompt": "0.00000026", "completion": "0.00000078", "input_cache_read": "0.000000052", "input_cache_write": "0.000000325", "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000078", "completion": "0.00000234", "input_cache_read": "0.000000156", "input_cache_write": "0.000000975" } ] }, "top_provider": { "context_length": 1000000, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2025-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen-plus-2025-01-25/endpoints" } }, { "id": "openai/o3-mini", "canonical_slug": "openai/o3-mini-2025-01-31", "hugging_face_id": "", "name": "OpenAI: o3 Mini", "created": 1738351721, "description": "OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...", "context_length": 200000, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.00000055" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-mini-2025-01-31/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "openai/o3-mini:batch", "canonical_slug": "openai/o3-mini-2025-01-31", "hugging_face_id": "", "name": "OpenAI: o3 Mini (batch)", "created": 1738351721, "description": "OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...", "context_length": 200000, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.000000275" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o3-mini-2025-01-31/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "mistralai/mistral-small-24b-instruct-2501", "canonical_slug": "mistralai/mistral-small-24b-instruct-2501", "hugging_face_id": "mistralai/Mistral-Small-24B-Instruct-2501", "name": "Mistral: Mistral Small 3", "created": 1738255409, "description": "Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.00000005", "completion": "0.00000008" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": { "temperature": 0.3, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-small-24b-instruct-2501/endpoints" } }, { "id": "perplexity/sonar", "canonical_slug": "perplexity/sonar", "hugging_face_id": "", "name": "Perplexity: Sonar", "created": 1738013808, "description": "Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...", "context_length": 127072, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000001", "web_search": "0.005" }, "top_provider": { "context_length": 127072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "temperature", "top_k", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/perplexity/sonar/endpoints" } }, { "id": "deepseek/deepseek-r1-distill-llama-70b", "canonical_slug": "deepseek/deepseek-r1-distill-llama-70b", "hugging_face_id": "deepseek-ai/DeepSeek-R1-Distill-Llama-70B", "name": "DeepSeek: R1 Distill Llama 70B", "created": 1737663169, "description": "DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...", "context_length": 8192, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "deepseek-r1" }, "pricing": { "prompt": "0.0000008", "completion": "0.0000008" }, "top_provider": { "context_length": 8192, "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "seed", "stop", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-07-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-r1-distill-llama-70b/endpoints" }, "reasoning": { "mandatory": false } }, { "id": "deepseek/deepseek-r1", "canonical_slug": "deepseek/deepseek-r1", "hugging_face_id": "deepseek-ai/DeepSeek-R1", "name": "DeepSeek: R1", "created": 1737381095, "description": "DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....", "context_length": 64000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-r1" }, "pricing": { "prompt": "0.0000007", "completion": "0.0000025" }, "top_provider": { "context_length": 64000, "max_completion_tokens": 16000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-07-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-r1/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": 18.6, "coding_index": 24.6, "agentic_index": 3.1 } }, "reasoning": { "mandatory": true } }, { "id": "minimax/minimax-01", "canonical_slug": "minimax/minimax-01", "hugging_face_id": "MiniMaxAI/MiniMax-Text-01", "name": "MiniMax: MiniMax-01", "created": 1736915462, "description": "MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...", "context_length": 1000192, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.0000002", "completion": "0.0000011" }, "top_provider": { "context_length": 1000192, "max_completion_tokens": 1000192, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/minimax/minimax-01/endpoints" } }, { "id": "microsoft/phi-4", "canonical_slug": "microsoft/phi-4", "hugging_face_id": "microsoft/phi-4", "name": "Microsoft: Phi 4", "created": 1736489872, "description": "[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...", "context_length": 16384, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "pricing": { "prompt": "0.00000007", "completion": "0.00000014" }, "top_provider": { "context_length": 16384, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/microsoft/phi-4/endpoints" } }, { "id": "deepseek/deepseek-chat", "canonical_slug": "deepseek/deepseek-chat-v3", "hugging_face_id": "deepseek-ai/DeepSeek-V3", "name": "DeepSeek: DeepSeek V3", "created": 1735241320, "description": "DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...", "context_length": 163840, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "pricing": { "prompt": "0.0000002574", "completion": "0.0000010287" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-07-31", "expiration_date": null, "links": { "details": "/api/v1/models/deepseek/deepseek-chat-v3/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 1138, "win_rate": 50.8, "rank": 71 }, { "arena": "models", "category": "codecategories", "elo": 1128, "win_rate": 48.4, "rank": 85 }, { "arena": "models", "category": "dataviz", "elo": 1108, "win_rate": 51.4, "rank": 88 }, { "arena": "models", "category": "gamedev", "elo": 1092, "win_rate": 43.9, "rank": 89 }, { "arena": "models", "category": "svg", "elo": 1009, "win_rate": 37.6, "rank": 76 }, { "arena": "models", "category": "uicomponent", "elo": 1121, "win_rate": 52.7, "rank": 79 }, { "arena": "models", "category": "website", "elo": 1132, "win_rate": 48.5, "rank": 83 } ] } }, { "id": "sao10k/l3.3-euryale-70b", "canonical_slug": "sao10k/l3.3-euryale-70b-v2.3", "hugging_face_id": "Sao10K/L3.3-70B-Euryale-v2.3", "name": "Sao10K: Llama 3.3 Euryale 70B", "created": 1734535928, "description": "Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.00000065", "completion": "0.00000075" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/sao10k/l3.3-euryale-70b-v2.3/endpoints" } }, { "id": "openai/o1", "canonical_slug": "openai/o1-2024-12-17", "hugging_face_id": "", "name": "OpenAI: o1", "created": 1734459999, "description": "The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000015", "completion": "0.00006", "web_search": "0.01", "input_cache_read": "0.0000075" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o1-2024-12-17/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 39.7, "agentic_index": null } }, "reasoning": { "mandatory": false } }, { "id": "openai/o1:batch", "canonical_slug": "openai/o1-2024-12-17", "hugging_face_id": "", "name": "OpenAI: o1 (batch)", "created": 1734459999, "description": "The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...", "context_length": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000075", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.00000375" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 100000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "include_reasoning", "max_tokens", "reasoning", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/o1-2024-12-17/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 39.7, "agentic_index": null } }, "reasoning": { "mandatory": false } }, { "id": "cohere/command-r7b-12-2024", "canonical_slug": "cohere/command-r7b-12-2024", "hugging_face_id": "", "name": "Cohere: Command R7B (12-2024)", "created": 1734158152, "description": "Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "pricing": { "prompt": "0.0000000375", "completion": "0.00000015" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 4000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/cohere/command-r7b-12-2024/endpoints" } }, { "id": "meta-llama/llama-3.3-70b-instruct", "canonical_slug": "meta-llama/llama-3.3-70b-instruct", "hugging_face_id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Meta: Llama 3.3 70B Instruct", "created": 1733506137, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.0000001", "completion": "0.00000032" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-3.3-70b-instruct/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 11.9, "agentic_index": null } } }, { "id": "amazon/nova-lite-v1", "canonical_slug": "amazon/nova-lite-v1", "hugging_face_id": "", "name": "Amazon: Nova Lite 1.0", "created": 1733437363, "description": "Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...", "context_length": 300000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "pricing": { "prompt": "0.00000006", "completion": "0.00000024" }, "top_provider": { "context_length": 300000, "max_completion_tokens": 5120, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "stop", "temperature", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/amazon/nova-lite-v1/endpoints" } }, { "id": "amazon/nova-micro-v1", "canonical_slug": "amazon/nova-micro-v1", "hugging_face_id": "", "name": "Amazon: Nova Micro 1.0", "created": 1733437237, "description": "Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "pricing": { "prompt": "0.000000035", "completion": "0.00000014" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 5120, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "stop", "temperature", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/amazon/nova-micro-v1/endpoints" } }, { "id": "amazon/nova-pro-v1", "canonical_slug": "amazon/nova-pro-v1", "hugging_face_id": "", "name": "Amazon: Nova Pro 1.0", "created": 1733436303, "description": "Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...", "context_length": 300000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "pricing": { "prompt": "0.0000008", "completion": "0.0000032" }, "top_provider": { "context_length": 300000, "max_completion_tokens": 5120, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "stop", "temperature", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/amazon/nova-pro-v1/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "website", "elo": 807, "win_rate": 21.4, "rank": 126 } ] } }, { "id": "openai/gpt-4o-2024-11-20", "canonical_slug": "openai/gpt-4o-2024-11-20", "hugging_face_id": "", "name": "OpenAI: GPT-4o (2024-11-20)", "created": 1732127594, "description": "The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "input_cache_read": "0.00000125" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o-2024-11-20/endpoints" } }, { "id": "mistralai/mistral-large-2407", "canonical_slug": "mistralai/mistral-large-2407", "hugging_face_id": "", "name": "Mistral Large 2407", "created": 1731978415, "description": "This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....", "context_length": 131072, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002" }, "top_provider": { "context_length": 131072, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2024-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-large-2407/endpoints" } }, { "id": "qwen/qwen-2.5-coder-32b-instruct", "canonical_slug": "qwen/qwen-2.5-coder-32b-instruct", "hugging_face_id": "Qwen/Qwen2.5-Coder-32B-Instruct", "name": "Qwen2.5 Coder 32B Instruct", "created": 1731368400, "description": "Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "pricing": { "prompt": "0.00000066", "completion": "0.000001" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "seed", "stop", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen-2.5-coder-32b-instruct/endpoints" } }, { "id": "thedrummer/unslopnemo-12b", "canonical_slug": "thedrummer/unslopnemo-12b", "hugging_face_id": "TheDrummer/UnslopNemo-12B-v4.1", "name": "TheDrummer: UnslopNemo 12B", "created": 1731103448, "description": "UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.", "context_length": 1024000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "pricing": { "prompt": "0.0000004", "completion": "0.0000004" }, "top_provider": { "context_length": 1024000, "max_completion_tokens": 1024000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-04-30", "expiration_date": null, "links": { "details": "/api/v1/models/thedrummer/unslopnemo-12b/endpoints" } }, { "id": "anthracite-org/magnum-v4-72b", "canonical_slug": "anthracite-org/magnum-v4-72b", "hugging_face_id": "anthracite-org/magnum-v4-72b", "name": "Magnum v4 72B", "created": 1729555200, "description": "This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "pricing": { "prompt": "0.000003", "completion": "0.000005" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 4096, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/anthracite-org/magnum-v4-72b/endpoints" } }, { "id": "qwen/qwen-2.5-7b-instruct", "canonical_slug": "qwen/qwen-2.5-7b-instruct", "hugging_face_id": "Qwen/Qwen2.5-7B-Instruct", "name": "Qwen: Qwen2.5 7B Instruct", "created": 1729036800, "description": "Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "pricing": { "prompt": "0.0000001", "completion": "0.0000002" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 32768, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": { "temperature": null, "top_p": null, "frequency_penalty": null }, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen-2.5-7b-instruct/endpoints" } }, { "id": "thedrummer/rocinante-12b", "canonical_slug": "thedrummer/rocinante-12b", "hugging_face_id": "TheDrummer/Rocinante-12B-v1.1", "name": "TheDrummer: Rocinante 12B", "created": 1727654400, "description": "Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...", "context_length": 65536, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "pricing": { "prompt": "0.00000025", "completion": "0.0000005" }, "top_provider": { "context_length": 65536, "max_completion_tokens": 65536, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-04-30", "expiration_date": null, "links": { "details": "/api/v1/models/thedrummer/rocinante-12b/endpoints" } }, { "id": "meta-llama/llama-3.2-1b-instruct", "canonical_slug": "meta-llama/llama-3.2-1b-instruct", "hugging_face_id": "meta-llama/Llama-3.2-1B-Instruct", "name": "Meta: Llama 3.2 1B Instruct", "created": 1727222400, "description": "Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...", "context_length": 60000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.000000027", "completion": "0.000000201" }, "top_provider": { "context_length": 60000, "max_completion_tokens": 60000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "seed", "stop", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-3.2-1b-instruct/endpoints" } }, { "id": "meta-llama/llama-3.2-3b-instruct", "canonical_slug": "meta-llama/llama-3.2-3b-instruct", "hugging_face_id": "meta-llama/Llama-3.2-3B-Instruct", "name": "Meta: Llama 3.2 3B Instruct", "created": 1727222400, "description": "Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.00000005", "completion": "0.00000033" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-3.2-3b-instruct/endpoints" } }, { "id": "qwen/qwen-2.5-72b-instruct", "canonical_slug": "qwen/qwen-2.5-72b-instruct", "hugging_face_id": "Qwen/Qwen2.5-72B-Instruct", "name": "Qwen2.5 72B Instruct", "created": 1726704000, "description": "Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...", "context_length": 32768, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "pricing": { "prompt": "0.00000036", "completion": "0.0000004" }, "top_provider": { "context_length": 32768, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/qwen/qwen-2.5-72b-instruct/endpoints" } }, { "id": "cohere/command-r-08-2024", "canonical_slug": "cohere/command-r-08-2024", "hugging_face_id": null, "name": "Cohere: Command R (08-2024)", "created": 1724976000, "description": "command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000006" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 4000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/cohere/command-r-08-2024/endpoints" } }, { "id": "cohere/command-r-plus-08-2024", "canonical_slug": "cohere/command-r-plus-08-2024", "hugging_face_id": null, "name": "Cohere: Command R+ (08-2024)", "created": 1724976000, "description": "command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.00001" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 4000, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-03-31", "expiration_date": null, "links": { "details": "/api/v1/models/cohere/command-r-plus-08-2024/endpoints" } }, { "id": "sao10k/l3.1-euryale-70b", "canonical_slug": "sao10k/l3.1-euryale-70b", "hugging_face_id": "Sao10K/L3.1-70B-Euryale-v2.2", "name": "Sao10K: Llama 3.1 Euryale 70B v2.2", "created": 1724803200, "description": "Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.00000085", "completion": "0.00000085" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/sao10k/l3.1-euryale-70b/endpoints" } }, { "id": "nousresearch/hermes-3-llama-3.1-70b", "canonical_slug": "nousresearch/hermes-3-llama-3.1-70b", "hugging_face_id": "NousResearch/Hermes-3-Llama-3.1-70B", "name": "Nous: Hermes 3 70B Instruct", "created": 1723939200, "description": "Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "chatml" }, "pricing": { "prompt": "0.0000007", "completion": "0.0000007" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/nousresearch/hermes-3-llama-3.1-70b/endpoints" } }, { "id": "nousresearch/hermes-3-llama-3.1-405b", "canonical_slug": "nousresearch/hermes-3-llama-3.1-405b", "hugging_face_id": "NousResearch/Hermes-3-Llama-3.1-405B", "name": "Nous: Hermes 3 405B Instruct", "created": 1723766400, "description": "Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "chatml" }, "pricing": { "prompt": "0.000001", "completion": "0.000001" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/nousresearch/hermes-3-llama-3.1-405b/endpoints" } }, { "id": "sao10k/l3-lunaris-8b", "canonical_slug": "sao10k/l3-lunaris-8b", "hugging_face_id": "Sao10K/L3-8B-Lunaris-v1", "name": "Sao10K: Llama 3 8B Lunaris", "created": 1723507200, "description": "Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....", "context_length": 8192, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.00000004", "completion": "0.00000005" }, "top_provider": { "context_length": 8192, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/sao10k/l3-lunaris-8b/endpoints" } }, { "id": "openai/gpt-4o-2024-08-06", "canonical_slug": "openai/gpt-4o-2024-08-06", "hugging_face_id": null, "name": "OpenAI: GPT-4o (2024-08-06)", "created": 1722902400, "description": "The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "input_cache_read": "0.00000125" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o-2024-08-06/endpoints" } }, { "id": "meta-llama/llama-3.1-70b-instruct", "canonical_slug": "meta-llama/llama-3.1-70b-instruct", "hugging_face_id": "meta-llama/Meta-Llama-3.1-70B-Instruct", "name": "Meta: Llama 3.1 70B Instruct", "created": 1721692800, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.0000004", "completion": "0.0000004" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-3.1-70b-instruct/endpoints" } }, { "id": "meta-llama/llama-3.1-8b-instruct", "canonical_slug": "meta-llama/llama-3.1-8b-instruct", "hugging_face_id": "meta-llama/Meta-Llama-3.1-8B-Instruct", "name": "Meta: Llama 3.1 8B Instruct", "created": 1721692800, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "pricing": { "prompt": "0.00000005", "completion": "0.00000008", "input_cache_read": "0.000000025" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 131072, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/meta-llama/llama-3.1-8b-instruct/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 5.4, "agentic_index": null } } }, { "id": "mistralai/mistral-nemo", "canonical_slug": "mistralai/mistral-nemo", "hugging_face_id": "mistralai/Mistral-Nemo-Instruct-2407", "name": "Mistral: Mistral Nemo", "created": 1721347200, "description": "A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...", "context_length": 131072, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "pricing": { "prompt": "0.000000019", "completion": "0.00000003" }, "top_provider": { "context_length": 131072, "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2024-04-30", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-nemo/endpoints" } }, { "id": "openai/gpt-4o-mini", "canonical_slug": "openai/gpt-4o-mini", "hugging_face_id": null, "name": "OpenAI: GPT-4o-mini", "created": 1721260800, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000075" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o-mini/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 11.4, "agentic_index": 1 } } }, { "id": "openai/gpt-4o-mini-2024-07-18", "canonical_slug": "openai/gpt-4o-mini-2024-07-18", "hugging_face_id": null, "name": "OpenAI: GPT-4o-mini (2024-07-18)", "created": 1721260800, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000075" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o-mini-2024-07-18/endpoints" } }, { "id": "openai/gpt-4o-mini:batch", "canonical_slug": "openai/gpt-4o-mini", "hugging_face_id": null, "name": "OpenAI: GPT-4o-mini (batch)", "created": 1721260800, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000000075", "completion": "0.0000003", "web_search": "0.01", "input_cache_read": "0.0000000375" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o-mini/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 11.4, "agentic_index": 1 } } }, { "id": "google/gemma-2-27b-it", "canonical_slug": "google/gemma-2-27b-it", "hugging_face_id": "google/gemma-2-27b-it", "name": "Google: Gemma 2 27B", "created": 1720828800, "description": "Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...", "context_length": 8192, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "pricing": { "prompt": "0.00000065", "completion": "0.00000065" }, "top_provider": { "context_length": 8192, "max_completion_tokens": 2048, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/google/gemma-2-27b-it/endpoints" } }, { "id": "openai/gpt-4o", "canonical_slug": "openai/gpt-4o", "hugging_face_id": null, "name": "OpenAI: GPT-4o", "created": 1715558400, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "input_cache_read": "0.00000125" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 919, "win_rate": 39.2, "rank": 106 }, { "arena": "models", "category": "codecategories", "elo": 880, "win_rate": 34.8, "rank": 119 }, { "arena": "models", "category": "dataviz", "elo": 875, "win_rate": 36, "rank": 116 }, { "arena": "models", "category": "gamedev", "elo": 945, "win_rate": 42.3, "rank": 114 }, { "arena": "models", "category": "uicomponent", "elo": 913, "win_rate": 38.1, "rank": 111 }, { "arena": "models", "category": "website", "elo": 842, "win_rate": 31.5, "rank": 125 } ] } }, { "id": "openai/gpt-4o-2024-05-13", "canonical_slug": "openai/gpt-4o-2024-05-13", "hugging_face_id": null, "name": "OpenAI: GPT-4o (2024-05-13)", "created": 1715558400, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000015" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 4096, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o-2024-05-13/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 24.2, "agentic_index": null } } }, { "id": "openai/gpt-4o:batch", "canonical_slug": "openai/gpt-4o", "hugging_face_id": null, "name": "OpenAI: GPT-4o (batch)", "created": 1715558400, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "context_length": 128000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000125", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.000000625" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "prediction", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-10-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4o/endpoints" }, "benchmarks": { "design_arena": [ { "arena": "models", "category": "3d", "elo": 919, "win_rate": 39.2, "rank": 106 }, { "arena": "models", "category": "codecategories", "elo": 880, "win_rate": 34.8, "rank": 119 }, { "arena": "models", "category": "dataviz", "elo": 875, "win_rate": 36, "rank": 116 }, { "arena": "models", "category": "gamedev", "elo": 945, "win_rate": 42.3, "rank": 114 }, { "arena": "models", "category": "uicomponent", "elo": 913, "win_rate": 38.1, "rank": 111 }, { "arena": "models", "category": "website", "elo": 842, "win_rate": 31.5, "rank": 125 } ] } }, { "id": "mistralai/mixtral-8x22b-instruct", "canonical_slug": "mistralai/mixtral-8x22b-instruct", "hugging_face_id": "mistralai/Mixtral-8x22B-Instruct-v0.1", "name": "Mistral: Mixtral 8x22B Instruct", "created": 1713312000, "description": "Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...", "context_length": 65536, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002" }, "top_provider": { "context_length": 65536, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2024-01-31", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mixtral-8x22b-instruct/endpoints" } }, { "id": "microsoft/wizardlm-2-8x22b", "canonical_slug": "microsoft/wizardlm-2-8x22b", "hugging_face_id": "microsoft/WizardLM-2-8x22B", "name": "WizardLM-2 8x22B", "created": 1713225600, "description": "WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...", "context_length": 65535, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "vicuna" }, "pricing": { "prompt": "0.00000062", "completion": "0.00000062" }, "top_provider": { "context_length": 65535, "max_completion_tokens": 8000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "temperature", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2024-04-30", "expiration_date": null, "links": { "details": "/api/v1/models/microsoft/wizardlm-2-8x22b/endpoints" } }, { "id": "openai/gpt-4-turbo", "canonical_slug": "openai/gpt-4-turbo", "hugging_face_id": null, "name": "OpenAI: GPT-4 Turbo", "created": 1712620800, "description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.", "context_length": 128000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00003" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 4096, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4-turbo/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 21.5, "agentic_index": null } } }, { "id": "openai/gpt-4-turbo:batch", "canonical_slug": "openai/gpt-4-turbo", "hugging_face_id": null, "name": "OpenAI: GPT-4 Turbo (batch)", "created": 1712620800, "description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.", "context_length": 128000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000005", "completion": "0.000015", "web_search": "0.01" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 4096, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4-turbo/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 21.5, "agentic_index": null } } }, { "id": "anthropic/claude-3-haiku", "canonical_slug": "anthropic/claude-3-haiku", "hugging_face_id": null, "name": "Anthropic: Claude 3 Haiku", "created": 1710288000, "description": "Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal", "context_length": 200000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.00000125", "web_search": "0.01", "input_cache_read": "0.00000003", "input_cache_write": "0.0000003", "input_cache_write_1h": "0.0000005" }, "top_provider": { "context_length": 200000, "max_completion_tokens": 4096, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "max_tokens", "stop", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-08-31", "expiration_date": null, "links": { "details": "/api/v1/models/anthropic/claude-3-haiku/endpoints" } }, { "id": "mistralai/mistral-large", "canonical_slug": "mistralai/mistral-large", "hugging_face_id": null, "name": "Mistral Large", "created": 1708905600, "description": "This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....", "context_length": 128000, "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002" }, "top_provider": { "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "default_parameters": { "temperature": 0.3 }, "supported_voices": null, "knowledge_cutoff": "2024-11-30", "expiration_date": null, "links": { "details": "/api/v1/models/mistralai/mistral-large/endpoints" } }, { "id": "openai/gpt-3.5-turbo-0613", "canonical_slug": "openai/gpt-3.5-turbo-0613", "hugging_face_id": null, "name": "OpenAI: GPT-3.5 Turbo (older v0613)", "created": 1706140800, "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.", "context_length": 4095, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000001", "completion": "0.000002" }, "top_provider": { "context_length": 4095, "max_completion_tokens": 4096, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_completion_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2021-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-3.5-turbo-0613/endpoints" } }, { "id": "openai/gpt-4-turbo-preview", "canonical_slug": "openai/gpt-4-turbo-preview", "hugging_face_id": null, "name": "OpenAI: GPT-4 Turbo Preview", "created": 1706140800, "description": "The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...", "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00001", "completion": "0.00003" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 4096, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-12-31", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4-turbo-preview/endpoints" } }, { "id": "openrouter/auto", "canonical_slug": "openrouter/auto", "hugging_face_id": null, "name": "Auto Router", "created": 1699401600, "description": "Your prompt will be processed by a meta-model and routed to one of dozens of models (see below), optimizing for the best possible output. To see which model was used,...", "context_length": 2000000, "architecture": { "modality": "text+image+file+audio+video->text+image", "input_modalities": [ "text", "image", "audio", "file", "video" ], "output_modalities": [ "text", "image" ], "tokenizer": "Router", "instruct_type": null }, "pricing": { "prompt": "-1", "completion": "-1" }, "top_provider": { "context_length": null, "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "prediction", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p", "web_search_options" ], "default_parameters": { "temperature": null, "top_p": null, "top_k": null, "frequency_penalty": null, "presence_penalty": null, "repetition_penalty": null }, "supported_voices": null, "knowledge_cutoff": null, "expiration_date": null, "links": { "details": "/api/v1/models/openrouter/auto/endpoints" } }, { "id": "openai/gpt-3.5-turbo-instruct", "canonical_slug": "openai/gpt-3.5-turbo-instruct", "hugging_face_id": null, "name": "OpenAI: GPT-3.5 Turbo Instruct", "created": 1695859200, "description": "This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.", "context_length": 4095, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": "chatml" }, "pricing": { "prompt": "0.0000015", "completion": "0.000002" }, "top_provider": { "context_length": 4095, "max_completion_tokens": 4096, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2021-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-3.5-turbo-instruct/endpoints" } }, { "id": "openai/gpt-3.5-turbo-16k", "canonical_slug": "openai/gpt-3.5-turbo-16k", "hugging_face_id": null, "name": "OpenAI: GPT-3.5 Turbo 16k", "created": 1693180800, "description": "This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...", "context_length": 16385, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.000003", "completion": "0.000004" }, "top_provider": { "context_length": 16385, "max_completion_tokens": 4096, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2021-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-3.5-turbo-16k/endpoints" } }, { "id": "mancer/weaver", "canonical_slug": "mancer/weaver", "hugging_face_id": null, "name": "Mancer: Weaver (alpha)", "created": 1690934400, "description": "An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.", "context_length": 8000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "pricing": { "prompt": "0.0000005", "completion": "0.00000075" }, "top_provider": { "context_length": 8000, "max_completion_tokens": 6000, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/mancer/weaver/endpoints" } }, { "id": "undi95/remm-slerp-l2-13b", "canonical_slug": "undi95/remm-slerp-l2-13b", "hugging_face_id": "Undi95/ReMM-SLERP-L2-13B", "name": "ReMM SLERP 13B", "created": 1689984000, "description": "A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge", "context_length": 6144, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "pricing": { "prompt": "0.00000045", "completion": "0.00000065" }, "top_provider": { "context_length": 6144, "max_completion_tokens": 6144, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/undi95/remm-slerp-l2-13b/endpoints" } }, { "id": "gryphe/mythomax-l2-13b", "canonical_slug": "gryphe/mythomax-l2-13b", "hugging_face_id": "Gryphe/MythoMax-L2-13b", "name": "MythoMax 13B", "created": 1688256000, "description": "One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge", "context_length": 8192, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "pricing": { "prompt": "0.00000006", "completion": "0.00000006" }, "top_provider": { "context_length": 4096, "max_completion_tokens": 4096, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_a", "top_k", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2023-06-30", "expiration_date": null, "links": { "details": "/api/v1/models/gryphe/mythomax-l2-13b/endpoints" } }, { "id": "openai/gpt-3.5-turbo", "canonical_slug": "openai/gpt-3.5-turbo", "hugging_face_id": null, "name": "OpenAI: GPT-3.5 Turbo", "created": 1685232000, "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.", "context_length": 16385, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.0000005", "completion": "0.0000015" }, "top_provider": { "context_length": 16385, "max_completion_tokens": 4096, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2021-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-3.5-turbo/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 10.7, "agentic_index": null } } }, { "id": "openai/gpt-3.5-turbo:batch", "canonical_slug": "openai/gpt-3.5-turbo", "hugging_face_id": null, "name": "OpenAI: GPT-3.5 Turbo (batch)", "created": 1685232000, "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.", "context_length": 16385, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00000025", "completion": "0.00000075", "web_search": "0.01" }, "top_provider": { "context_length": 16385, "max_completion_tokens": 4096, "is_moderated": true }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2021-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-3.5-turbo/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 10.7, "agentic_index": null } } }, { "id": "openai/gpt-4", "canonical_slug": "openai/gpt-4", "hugging_face_id": null, "name": "OpenAI: GPT-4", "created": 1685232000, "description": "OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...", "context_length": 8191, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "pricing": { "prompt": "0.00003", "completion": "0.00006" }, "top_provider": { "context_length": 8191, "max_completion_tokens": 4096, "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ "frequency_penalty", "logit_bias", "logprobs", "max_completion_tokens", "max_tokens", "presence_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "default_parameters": {}, "supported_voices": null, "knowledge_cutoff": "2021-09-30", "expiration_date": null, "links": { "details": "/api/v1/models/openai/gpt-4/endpoints" }, "benchmarks": { "design_arena": [], "artificial_analysis": { "intelligence_index": null, "coding_index": 13.1, "agentic_index": null } } } ] }