diff --git a/assets/models-dev/catalog.json b/assets/models-dev/catalog.json index ea5fd04624..92aee15a1b 100644 --- a/assets/models-dev/catalog.json +++ b/assets/models-dev/catalog.json @@ -6937,6 +6937,56 @@ "temperature": true, "tool_call": true }, + "claude-fable-5": { + "attachment": true, + "cost": { + "cache_read": 1.1, + "cache_write": 13.75, + "input": 11, + "output": 55 + }, + "description": "Claude model for creative writing, analysis, and controlled agent workflows", + "family": "claude-fable", + "id": "claude-fable-5", + "interleaved": true, + "knowledge": "2026-01-31", + "last_updated": "2026-06-09", + "limit": { + "context": 1000000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Claude Fable 5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + ], + "release_date": "2026-06-09", + "structured_output": true, + "temperature": false, + "tool_call": true + }, "claude-opus-4-6": { "attachment": true, "cost": { @@ -7469,6 +7519,56 @@ "temperature": true, "tool_call": true }, + "claude-sonnet-5": { + "attachment": true, + "cost": { + "cache_read": 0.2, + "cache_write": 2.5, + "input": 2, + "output": 10 + }, + "description": "Everyday Claude agent model for coding, planning, browsing, and general work", + "family": "claude-sonnet", + "id": "claude-sonnet-5", + "interleaved": true, + "knowledge": "2026-01-31", + "last_updated": "2026-06-30", + "limit": { + "context": 1000000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Claude Sonnet 5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + ], + "release_date": "2026-06-30", + "structured_output": true, + "temperature": false, + "tool_call": true + }, "coding-glm-5.1": { "attachment": false, "cost": { @@ -8479,6 +8579,52 @@ "temperature": true, "tool_call": true }, + "gemini-3.5-flash": { + "attachment": true, + "cost": { + "cache_read": 1.5, + "input": 1.5, + "output": 9 + }, + "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", + "family": "gemini-flash", + "id": "gemini-3.5-flash", + "knowledge": "2025-01", + "last_updated": "2026-05-19", + "limit": { + "context": 1000000, + "output": 64000 + }, + "modalities": { + "input": [ + "text", + "image", + "audio", + "video" + ], + "output": [ + "text" + ] + }, + "name": "Gemini 3.5 Flash", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "minimal", + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-05-19", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "glm-5.2": { "attachment": false, "cost": { @@ -9054,6 +9200,150 @@ "temperature": false, "tool_call": true }, + "gpt-5.6-luna": { + "attachment": true, + "cost": { + "cache_read": 0.1, + "cache_write": 1.25, + "input": 1, + "output": 6 + }, + "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", + "family": "gpt-luna", + "id": "gpt-5.6-luna", + "knowledge": "2026-02-16", + "last_updated": "2026-07-09", + "limit": { + "context": 1050000, + "input": 922000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "GPT-5.6 Luna", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "none", + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + ], + "release_date": "2026-07-09", + "structured_output": true, + "temperature": false, + "tool_call": true + }, + "gpt-5.6-sol": { + "attachment": true, + "cost": { + "cache_read": 0.5, + "cache_write": 6.25, + "input": 5, + "output": 30 + }, + "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", + "family": "gpt-sol", + "id": "gpt-5.6-sol", + "knowledge": "2026-02-16", + "last_updated": "2026-07-09", + "limit": { + "context": 1050000, + "input": 922000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "GPT-5.6 Sol", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "none", + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + ], + "release_date": "2026-07-09", + "structured_output": true, + "temperature": false, + "tool_call": true + }, + "gpt-5.6-terra": { + "attachment": true, + "cost": { + "cache_read": 0.25, + "cache_write": 3.125, + "input": 2.5, + "output": 15 + }, + "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", + "family": "gpt-terra", + "id": "gpt-5.6-terra", + "knowledge": "2026-02-16", + "last_updated": "2026-07-09", + "limit": { + "context": 1050000, + "input": 922000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "GPT-5.6 Terra", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "none", + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + ], + "release_date": "2026-07-09", + "structured_output": true, + "temperature": false, + "tool_call": true + }, "grok-4.3": { "attachment": true, "cost": { @@ -9113,6 +9403,123 @@ "temperature": true, "tool_call": true }, + "grok-4.5": { + "attachment": true, + "cost": { + "cache_read": 0.5, + "input": 2, + "output": 6 + }, + "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", + "family": "grok", + "id": "grok-4.5", + "last_updated": "2026-07-08", + "limit": { + "context": 1000000, + "output": 1000000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Grok 4.5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "none", + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-07-08", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "grok-build-0.1": { + "attachment": true, + "cost": { + "cache_read": 0.2, + "input": 1, + "output": 2 + }, + "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", + "family": "grok-build", + "id": "grok-build-0.1", + "last_updated": "2026-04-16", + "limit": { + "context": 256000, + "output": 256000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Grok Build 0.1", + "open_weights": false, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-04-16", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "hy3-preview": { + "attachment": false, + "cost": { + "cache_read": 0.051, + "input": 0.17, + "output": 0.566661 + }, + "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", + "family": "Hy", + "id": "hy3-preview", + "last_updated": "2026-04-20", + "limit": { + "context": 256000, + "output": 128000 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "Hy3 Preview", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "none", + "low", + "high" + ] + } + ], + "release_date": "2026-04-20", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "kimi-k2.5": { "attachment": true, "cost": { @@ -9197,6 +9604,90 @@ "temperature": false, "tool_call": true }, + "kimi-k2.7-code": { + "attachment": true, + "cost": { + "cache_read": 0.160835, + "input": 0.95, + "output": 3.9995 + }, + "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", + "family": "kimi-k2", + "id": "kimi-k2.7-code", + "interleaved": { + "field": "reasoning_content" + }, + "knowledge": "2025-01", + "last_updated": "2026-06-12", + "limit": { + "context": 262144, + "output": 32768 + }, + "modalities": { + "input": [ + "text", + "image", + "video" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K2.7 Code", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-06-12", + "structured_output": true, + "temperature": false, + "tool_call": true + }, + "kimi-k2.7-code-highspeed": { + "attachment": true, + "cost": { + "cache_read": 0.32167, + "input": 1.9, + "output": 7.999 + }, + "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", + "family": "kimi-k2", + "id": "kimi-k2.7-code-highspeed", + "interleaved": { + "field": "reasoning_content" + }, + "knowledge": "2025-01", + "last_updated": "2026-06-12", + "limit": { + "context": 262144, + "output": 32768 + }, + "modalities": { + "input": [ + "text", + "image", + "video" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K2.7 Code Highspeed", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-06-12", + "structured_output": true, + "temperature": false, + "tool_call": true + }, "minimax-m2.7": { "attachment": false, "cost": { @@ -21280,10 +21771,10 @@ "moonshotai/kimi-k2.7-code": { "attachment": true, "cost": { - "cache_read": 0.149, + "cache_read": 0.16, "cache_write": 0, - "input": 0.719, - "output": 3.49 + "input": 0.75, + "output": 3.5 }, "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", @@ -21316,10 +21807,10 @@ "stepfun/step-3.7-flash": { "attachment": true, "cost": { - "cache_read": 0.04, + "cache_read": 0.03, "cache_write": 0, - "input": 0.2, - "output": 1.15 + "input": 0.19, + "output": 1.14 }, "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "id": "stepfun/step-3.7-flash", @@ -32648,8 +33139,8 @@ }, "last_updated": "2026-06-13", "limit": { - "context": 1048576, - "output": 1048576 + "context": 256000, + "output": 256000 }, "modalities": { "input": [ @@ -53150,6 +53641,49 @@ "reasoning": false, "release_date": "2024-10-01", "tool_call": false + }, + "zai-org/GLM-5.2": { + "attachment": false, + "cost": { + "input": 1.4375, + "output": 5.75 + }, + "description": "Open flagship GLM for long-horizon coding agents and million-token context work", + "family": "glm", + "id": "zai-org/GLM-5.2", + "last_updated": "2026-06-13", + "limit": { + "context": 1048576, + "output": 131072 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "GLM-5.2", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "none", + "high", + "max" + ] + } + ], + "release_date": "2026-06-13", + "structured_output": true, + "temperature": true, + "tool_call": true } }, "name": "evroc", @@ -62529,16 +63063,16 @@ "gemini-flash-latest": { "attachment": true, "cost": { - "cache_read": 0.075, - "input": 0.3, - "input_audio": 1, - "output": 2.5 + "cache_read": 0.15, + "input": 1.5, + "input_audio": 1.5, + "output": 9 }, "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "id": "gemini-flash-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-19", "limit": { "context": 1048576, "output": 65536 @@ -62547,8 +63081,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -62560,15 +63094,16 @@ "reasoning": true, "reasoning_options": [ { - "type": "toggle" - }, - { - "max": 24576, - "min": 0, - "type": "budget_tokens" + "type": "effort", + "values": [ + "minimal", + "low", + "medium", + "high" + ] } ], - "release_date": "2025-09-25", + "release_date": "2026-05-19", "structured_output": true, "temperature": true, "tool_call": true @@ -62577,14 +63112,15 @@ "attachment": true, "cost": { "cache_read": 0.025, - "input": 0.1, - "output": 0.4 + "input": 0.25, + "input_audio": 0.5, + "output": 1.5 }, "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "id": "gemini-flash-lite-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-07", "limit": { "context": 1048576, "output": 65536 @@ -62593,8 +63129,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -62606,15 +63142,16 @@ "reasoning": true, "reasoning_options": [ { - "type": "toggle" - }, - { - "max": 24576, - "min": 512, - "type": "budget_tokens" + "type": "effort", + "values": [ + "minimal", + "low", + "medium", + "high" + ] } ], - "release_date": "2025-09-25", + "release_date": "2026-05-07", "structured_output": true, "temperature": true, "tool_call": true @@ -64102,16 +64639,16 @@ "gemini-flash-latest": { "attachment": true, "cost": { - "cache_read": 0.075, - "cache_write": 0.383, - "input": 0.3, - "output": 2.5 + "cache_read": 0.15, + "input": 1.5, + "input_audio": 1.5, + "output": 9 }, "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "id": "gemini-flash-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-19", "limit": { "context": 1048576, "output": 65536 @@ -64120,8 +64657,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -64142,7 +64679,8 @@ ] } ], - "release_date": "2025-09-25", + "release_date": "2026-05-19", + "structured_output": true, "temperature": true, "tool_call": true }, @@ -64150,14 +64688,15 @@ "attachment": true, "cost": { "cache_read": 0.025, - "input": 0.1, - "output": 0.4 + "input": 0.25, + "input_audio": 0.5, + "output": 1.5 }, "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "id": "gemini-flash-lite-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-07", "limit": { "context": 1048576, "output": 65536 @@ -64166,8 +64705,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -64188,7 +64727,7 @@ ] } ], - "release_date": "2025-09-25", + "release_date": "2026-05-07", "temperature": true, "tool_call": true }, @@ -86774,9 +87313,13 @@ "high", "max" ] + }, + { + "min": 1024, + "type": "budget_tokens" } ], - "release_date": "2026-04-27", + "release_date": "2025-10-15", "temperature": true, "tool_call": true }, @@ -86837,6 +87380,7 @@ }, "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "id": "~anthropic/claude-sonnet-latest", + "knowledge": "2025-08-31", "last_updated": "2026-05-01", "limit": { "context": 1000000, @@ -86867,23 +87411,27 @@ "high", "max" ] + }, + { + "min": 1024, + "type": "budget_tokens" } ], - "release_date": "2026-04-27", + "release_date": "2026-02-17", "temperature": true, "tool_call": true }, "~google/gemini-flash-latest": { "attachment": true, "cost": { - "cache_read": 0.05, + "cache_read": 0.15, "cache_write": 0.08333333333333334, - "input": 0.5, - "output": 3 + "input": 1.5, + "output": 9 }, "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "id": "~google/gemini-flash-latest", - "last_updated": "2026-05-01", + "last_updated": "2026-05-19", "limit": { "context": 1048576, "output": 65536 @@ -86914,7 +87462,7 @@ ] } ], - "release_date": "2026-04-27", + "release_date": "2026-05-19", "temperature": true, "tool_call": true }, @@ -87025,12 +87573,10 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ + "none", "low", "medium", "high", @@ -87038,7 +87584,7 @@ ] } ], - "release_date": "2026-04-27", + "release_date": "2026-04-24", "temperature": false, "tool_call": true }, @@ -87070,12 +87616,10 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ + "none", "low", "medium", "high", @@ -87083,7 +87627,7 @@ ] } ], - "release_date": "2026-04-27", + "release_date": "2026-03-17", "temperature": false, "tool_call": true } @@ -87215,6 +87759,87 @@ "temperature": false, "tool_call": true }, + "k3": { + "attachment": false, + "cost": { + "cache_read": 0, + "cache_write": 0, + "input": 0, + "output": 0 + }, + "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", + "family": "kimi-k3", + "id": "k3", + "last_updated": "2026-07-16", + "limit": { + "context": 1048576, + "output": 131072 + }, + "modalities": { + "input": [ + "text", + "image", + "video" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K3", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "max" + ] + } + ], + "release_date": "2026-07-16", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "kimi-for-coding-highspeed": { + "attachment": true, + "cost": { + "cache_read": 0, + "cache_write": 0, + "input": 0, + "output": 0 + }, + "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", + "family": "kimi-k2", + "id": "kimi-for-coding-highspeed", + "knowledge": "2025-01", + "last_updated": "2026-06-12", + "limit": { + "context": 262144, + "output": 32768 + }, + "modalities": { + "input": [ + "text", + "image", + "video" + ], + "output": [ + "text" + ] + }, + "name": "Kimi For Coding HighSpeed", + "open_weights": true, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-06-12", + "structured_output": true, + "temperature": false, + "tool_call": true + }, "kimi-k2-thinking": { "attachment": false, "cost": { @@ -88424,49 +89049,6 @@ "temperature": true, "tool_call": true }, - "deepseek-v3.1": { - "attachment": true, - "cost": { - "cache_read": 0.112, - "input": 0.56, - "output": 1.68 - }, - "description": "DeepSeek chat model for instruction following, coding, and analysis", - "family": "deepseek", - "id": "deepseek-v3.1", - "last_updated": "2025-08-21", - "limit": { - "context": 128000, - "output": 32768 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "DeepSeek V3.1", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "effort", - "values": [ - "minimal", - "low", - "medium", - "high" - ] - } - ], - "release_date": "2025-08-21", - "structured_output": true, - "temperature": true, - "tool_call": true - }, "deepseek-v3.2": { "attachment": true, "cost": { @@ -88607,37 +89189,6 @@ "temperature": true, "tool_call": true }, - "devstral-small-2507": { - "attachment": false, - "cost": { - "input": 0.1, - "output": 0.3 - }, - "description": "Mistral coding agent model for repository tasks and software engineering workflows", - "family": "devstral", - "id": "devstral-small-2507", - "knowledge": "2025-05", - "last_updated": "2025-07-10", - "limit": { - "context": 131072, - "output": 128000 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "Devstral Small", - "open_weights": true, - "reasoning": false, - "release_date": "2025-07-10", - "status": "deprecated", - "temperature": true, - "tool_call": true - }, "fugu-ultra": { "attachment": true, "cost": { @@ -91896,36 +92447,6 @@ "temperature": true, "tool_call": false }, - "llama-3-8b-instruct": { - "attachment": false, - "cost": { - "input": 0.04, - "output": 0.04 - }, - "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", - "family": "llama", - "id": "llama-3-8b-instruct", - "last_updated": "2025-04-03", - "limit": { - "context": 8192, - "output": 8192 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "Llama 3 8B Instruct", - "open_weights": true, - "reasoning": false, - "release_date": "2025-04-03", - "structured_output": true, - "temperature": true, - "tool_call": false - }, "llama-3.1-70b-instruct": { "attachment": false, "cost": { @@ -92475,16 +92996,16 @@ "minimax-m3": { "attachment": true, "cost": { - "cache_read": 0.12, - "input": 0.6, - "output": 2.4 + "cache_read": 0.06, + "input": 0.3, + "output": 1.2 }, "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "id": "minimax-m3", "last_updated": "2026-06-01", "limit": { - "context": 512000, + "context": 524288, "output": 128000 }, "modalities": { @@ -92959,37 +93480,6 @@ "temperature": false, "tool_call": true }, - "pixtral-large-latest": { - "attachment": true, - "cost": { - "input": 4, - "output": 12 - }, - "description": "Mistral's larger vision model for document-heavy image understanding and chat", - "family": "pixtral", - "id": "pixtral-large-latest", - "knowledge": "2024-11", - "last_updated": "2024-11-04", - "limit": { - "context": 128000, - "output": 128000 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "Pixtral Large (latest)", - "open_weights": true, - "reasoning": false, - "release_date": "2024-11-01", - "temperature": true, - "tool_call": true - }, "qwen-coder-plus": { "attachment": false, "cost": { @@ -93431,37 +93921,6 @@ "temperature": true, "tool_call": true }, - "qwen3-4b-fp8": { - "attachment": false, - "cost": { - "input": 0.03, - "output": 0.03 - }, - "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", - "family": "qwen", - "id": "qwen3-4b-fp8", - "last_updated": "2025-04-28", - "limit": { - "context": 128000, - "output": 8192 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "Qwen3 4B FP8", - "open_weights": true, - "reasoning": true, - "reasoning_options": [], - "release_date": "2025-04-28", - "structured_output": true, - "temperature": true, - "tool_call": true - }, "qwen3-coder-30b-a3b-instruct": { "attachment": false, "cost": { @@ -93805,69 +94264,6 @@ "temperature": true, "tool_call": true }, - "qwen3-vl-30b-a3b-thinking": { - "attachment": true, - "cost": { - "input": 0.2, - "output": 1 - }, - "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", - "family": "qwen", - "id": "qwen3-vl-30b-a3b-thinking", - "last_updated": "2025-10-02", - "limit": { - "context": 131072, - "output": 8192 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "Qwen3 VL 30B A3B Thinking", - "open_weights": true, - "reasoning": true, - "reasoning_options": [], - "release_date": "2025-10-02", - "structured_output": true, - "temperature": true, - "tool_call": true - }, - "qwen3-vl-8b-instruct": { - "attachment": true, - "cost": { - "input": 0.08, - "output": 0.5 - }, - "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", - "family": "qwen", - "id": "qwen3-vl-8b-instruct", - "last_updated": "2025-08-19", - "limit": { - "context": 131072, - "output": 8192 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "Qwen3 VL 8B Instruct", - "open_weights": true, - "reasoning": false, - "release_date": "2025-08-19", - "structured_output": false, - "temperature": true, - "tool_call": false - }, "qwen3-vl-flash": { "attachment": true, "cost": { @@ -96974,16 +97370,16 @@ "google/gemini-flash-latest": { "attachment": true, "cost": { - "cache_read": 0.075, - "input": 0.3, - "input_audio": 1, - "output": 2.5 + "cache_read": 0.15, + "input": 1.5, + "input_audio": 1.5, + "output": 9 }, "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "id": "google/gemini-flash-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-19", "limit": { "context": 1048576, "output": 65536 @@ -96992,8 +97388,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -97014,7 +97410,7 @@ ] } ], - "release_date": "2025-09-25", + "release_date": "2026-05-19", "structured_output": true, "temperature": true, "tool_call": true @@ -97023,14 +97419,15 @@ "attachment": true, "cost": { "cache_read": 0.025, - "input": 0.1, - "output": 0.4 + "input": 0.25, + "input_audio": 0.5, + "output": 1.5 }, "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "id": "google/gemini-flash-lite-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-07", "limit": { "context": 1048576, "output": 65536 @@ -97039,8 +97436,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -97061,7 +97458,7 @@ ] } ], - "release_date": "2025-09-25", + "release_date": "2026-05-07", "structured_output": true, "temperature": true, "tool_call": true @@ -102925,6 +103322,48 @@ "structured_output": true, "temperature": false, "tool_call": true + }, + "kimi-k3": { + "attachment": true, + "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", + "family": "kimi-k3", + "id": "kimi-k3", + "interleaved": { + "field": "reasoning_content" + }, + "last_updated": "2026-07-16", + "limit": { + "context": 1048576, + "output": 131072 + }, + "modalities": { + "input": [ + "text", + "image", + "video" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K3", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "max" + ] + } + ], + "release_date": "2026-07-16", + "structured_output": true, + "temperature": true, + "tool_call": true } }, "name": "Moonshot AI", @@ -103260,6 +103699,48 @@ "structured_output": true, "temperature": false, "tool_call": true + }, + "kimi-k3": { + "attachment": true, + "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", + "family": "kimi-k3", + "id": "kimi-k3", + "interleaved": { + "field": "reasoning_content" + }, + "last_updated": "2026-07-16", + "limit": { + "context": 1048576, + "output": 131072 + }, + "modalities": { + "input": [ + "text", + "image", + "video" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K3", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "max" + ] + } + ], + "release_date": "2026-07-16", + "structured_output": true, + "temperature": true, + "tool_call": true } }, "name": "Moonshot AI (China)", @@ -113941,10 +114422,10 @@ }, "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "id": "google/gemini-flash-latest", - "last_updated": "2026-03-29", + "last_updated": "2026-05-19", "limit": { - "context": 1048756, - "input": 1048756, + "context": 1048576, + "input": 1048576, "output": 65536 }, "modalities": { @@ -113970,9 +114451,9 @@ ] } ], - "release_date": "2026-03-29", - "structured_output": false, - "tool_call": false + "release_date": "2026-05-19", + "structured_output": true, + "tool_call": true }, "google/gemini-flash-lite-latest": { "attachment": true, @@ -113983,7 +114464,7 @@ }, "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "id": "google/gemini-flash-lite-latest", - "last_updated": "2026-03-29", + "last_updated": "2026-05-07", "limit": { "context": 1048576, "input": 1048576, @@ -114013,7 +114494,7 @@ ] } ], - "release_date": "2026-03-29", + "release_date": "2026-05-07", "structured_output": true, "tool_call": true }, @@ -134713,94 +135194,6 @@ ], "id": "ollama-cloud", "models": { - "cogito-2.1:671b": { - "attachment": false, - "description": "Legacy model retained for compatibility with older integrations", - "family": "cogito", - "id": "cogito-2.1:671b", - "last_updated": "2026-01-19", - "limit": { - "context": 163840, - "output": 32000 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "cogito-2.1:671b", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-11-19", - "status": "deprecated", - "tool_call": true - }, - "deepseek-v3.1:671b": { - "attachment": false, - "description": "DeepSeek chat model for instruction following, coding, and analysis", - "family": "deepseek", - "id": "deepseek-v3.1:671b", - "last_updated": "2026-01-19", - "limit": { - "context": 163840, - "output": 163840 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "deepseek-v3.1:671b", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-08-21", - "tool_call": true - }, - "deepseek-v3.2": { - "attachment": false, - "description": "DeepSeek chat model for instruction following, coding, and analysis", - "family": "deepseek", - "id": "deepseek-v3.2", - "last_updated": "2026-01-19", - "limit": { - "context": 163840, - "output": 65536 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "deepseek-v3.2", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-06-15", - "tool_call": true - }, "deepseek-v4-flash": { "attachment": false, "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", @@ -134873,161 +135266,6 @@ "release_date": "2026-04-24", "tool_call": true }, - "devstral-2:123b": { - "attachment": false, - "description": "Mistral coding agent model for repository tasks and software engineering workflows", - "family": "devstral", - "id": "devstral-2:123b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 262144 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "devstral-2:123b", - "open_weights": true, - "reasoning": false, - "release_date": "2025-12-09", - "tool_call": true - }, - "devstral-small-2:24b": { - "attachment": true, - "description": "Mistral coding agent model for repository tasks and software engineering workflows", - "family": "devstral", - "id": "devstral-small-2:24b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 262144 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "devstral-small-2:24b", - "open_weights": true, - "reasoning": false, - "release_date": "2025-12-09", - "tool_call": true - }, - "gemini-3-flash-preview": { - "attachment": true, - "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", - "family": "gemini-flash", - "id": "gemini-3-flash-preview", - "knowledge": "2025-01", - "last_updated": "2026-04-08", - "limit": { - "context": 1048576, - "output": 65536 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "gemini-3-flash-preview", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-12-17", - "tool_call": true - }, - "gemma3:12b": { - "attachment": true, - "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", - "family": "gemma", - "id": "gemma3:12b", - "last_updated": "2026-01-19", - "limit": { - "context": 131072, - "output": 131072 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "gemma3:12b", - "open_weights": true, - "reasoning": false, - "release_date": "2024-12-01", - "tool_call": false - }, - "gemma3:27b": { - "attachment": true, - "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", - "family": "gemma", - "id": "gemma3:27b", - "last_updated": "2026-01-19", - "limit": { - "context": 131072, - "output": 131072 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "gemma3:27b", - "open_weights": true, - "reasoning": false, - "release_date": "2025-07-27", - "tool_call": false - }, - "gemma3:4b": { - "attachment": true, - "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", - "family": "gemma", - "id": "gemma3:4b", - "last_updated": "2026-01-19", - "limit": { - "context": 131072, - "output": 131072 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "gemma3:4b", - "open_weights": true, - "reasoning": false, - "release_date": "2024-12-01", - "tool_call": false - }, "gemma4:31b": { "attachment": true, "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", @@ -135059,97 +135297,6 @@ "release_date": "2026-04-02", "tool_call": true }, - "glm-4.6": { - "attachment": false, - "description": "Legacy model retained for compatibility with older integrations", - "family": "glm", - "id": "glm-4.6", - "last_updated": "2026-01-19", - "limit": { - "context": 202752, - "output": 131072 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "glm-4.6", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-09-29", - "status": "deprecated", - "tool_call": true - }, - "glm-4.7": { - "attachment": false, - "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", - "family": "glm", - "id": "glm-4.7", - "last_updated": "2026-01-19", - "limit": { - "context": 202752, - "output": 131072 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "glm-4.7", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-12-22", - "tool_call": true - }, - "glm-5": { - "attachment": false, - "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", - "family": "glm", - "id": "glm-5", - "interleaved": { - "field": "reasoning_content" - }, - "last_updated": "2026-02-11", - "limit": { - "context": 202752, - "output": 131072 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "glm-5", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2026-02-11", - "tool_call": true - }, "glm-5.1": { "attachment": false, "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", @@ -135288,33 +135435,6 @@ "release_date": "2025-08-05", "tool_call": true }, - "kimi-k2-thinking": { - "attachment": false, - "description": "Legacy model retained for compatibility with older integrations", - "family": "kimi-thinking", - "id": "kimi-k2-thinking", - "knowledge": "2024-08", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 262144 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "kimi-k2-thinking", - "open_weights": true, - "reasoning": true, - "reasoning_options": [], - "release_date": "2025-11-06", - "status": "deprecated", - "tool_call": true - }, "kimi-k2.5": { "attachment": true, "description": "Kimi multimodal agent model for visual understanding, coding, and planning", @@ -135408,83 +135528,6 @@ "temperature": false, "tool_call": true }, - "kimi-k2:1t": { - "attachment": false, - "description": "Legacy model retained for compatibility with older integrations", - "family": "kimi-k2", - "id": "kimi-k2:1t", - "knowledge": "2024-10", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 262144 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "kimi-k2:1t", - "open_weights": true, - "reasoning": false, - "release_date": "2025-07-11", - "status": "deprecated", - "tool_call": true - }, - "minimax-m2": { - "attachment": false, - "description": "Legacy model retained for compatibility with older integrations", - "family": "minimax", - "id": "minimax-m2", - "last_updated": "2026-01-19", - "limit": { - "context": 204800, - "output": 128000 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "minimax-m2", - "open_weights": true, - "reasoning": true, - "reasoning_options": [], - "release_date": "2025-10-23", - "status": "deprecated", - "tool_call": true - }, - "minimax-m2.1": { - "attachment": false, - "description": "MiniMax model for chat, coding, office work, and agentic tasks", - "family": "minimax", - "id": "minimax-m2.1", - "last_updated": "2026-01-19", - "limit": { - "context": 204800, - "output": 131072 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "minimax-m2.1", - "open_weights": true, - "reasoning": true, - "reasoning_options": [], - "release_date": "2025-12-23", - "tool_call": true - }, "minimax-m2.5": { "attachment": false, "description": "MiniMax model for chat, coding, office work, and agentic tasks", @@ -135582,81 +135625,6 @@ "temperature": true, "tool_call": true }, - "ministral-3:14b": { - "attachment": true, - "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", - "family": "ministral", - "id": "ministral-3:14b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 128000 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "ministral-3:14b", - "open_weights": true, - "reasoning": false, - "release_date": "2024-12-01", - "tool_call": true - }, - "ministral-3:3b": { - "attachment": true, - "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", - "family": "ministral", - "id": "ministral-3:3b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 128000 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "ministral-3:3b", - "open_weights": true, - "reasoning": false, - "release_date": "2024-10-22", - "tool_call": true - }, - "ministral-3:8b": { - "attachment": true, - "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", - "family": "ministral", - "id": "ministral-3:8b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 128000 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "ministral-3:8b", - "open_weights": true, - "reasoning": false, - "release_date": "2024-12-01", - "tool_call": true - }, "mistral-large-3:675b": { "attachment": true, "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", @@ -135772,133 +135740,6 @@ "temperature": true, "tool_call": true }, - "qwen3-coder-next": { - "attachment": false, - "description": "Qwen coding model for software agents, repository edits, and code reasoning", - "family": "qwen", - "id": "qwen3-coder-next", - "last_updated": "2026-02-08", - "limit": { - "context": 262144, - "output": 65536 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "qwen3-coder-next", - "open_weights": true, - "reasoning": false, - "release_date": "2026-02-02", - "tool_call": true - }, - "qwen3-coder:480b": { - "attachment": false, - "description": "Qwen coding model for software agents, repository edits, and code reasoning", - "family": "qwen", - "id": "qwen3-coder:480b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 65536 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "qwen3-coder:480b", - "open_weights": true, - "reasoning": false, - "release_date": "2025-07-22", - "tool_call": true - }, - "qwen3-next:80b": { - "attachment": false, - "description": "Legacy model retained for compatibility with older integrations", - "family": "qwen", - "id": "qwen3-next:80b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 32768 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "qwen3-next:80b", - "open_weights": true, - "reasoning": true, - "reasoning_options": [], - "release_date": "2025-09-15", - "status": "deprecated", - "tool_call": true - }, - "qwen3-vl:235b": { - "attachment": true, - "description": "Legacy model retained for compatibility with older integrations", - "family": "qwen", - "id": "qwen3-vl:235b", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 32768 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "qwen3-vl:235b", - "open_weights": true, - "reasoning": true, - "reasoning_options": [], - "release_date": "2025-09-22", - "status": "deprecated", - "tool_call": true - }, - "qwen3-vl:235b-instruct": { - "attachment": true, - "description": "Legacy model retained for compatibility with older integrations", - "family": "qwen", - "id": "qwen3-vl:235b-instruct", - "last_updated": "2026-01-19", - "limit": { - "context": 262144, - "output": 131072 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "qwen3-vl:235b-instruct", - "open_weights": true, - "reasoning": false, - "release_date": "2025-09-22", - "status": "deprecated", - "tool_call": true - }, "qwen3.5:397b": { "attachment": true, "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", @@ -135931,30 +135772,6 @@ ], "release_date": "2026-02-15", "tool_call": true - }, - "rnj-1:8b": { - "attachment": false, - "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", - "family": "rnj", - "id": "rnj-1:8b", - "last_updated": "2026-01-19", - "limit": { - "context": 32768, - "output": 4096 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "rnj-1:8b", - "open_weights": true, - "reasoning": false, - "release_date": "2025-12-06", - "tool_call": true } }, "name": "Ollama Cloud", @@ -143590,9 +143407,6 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -143603,8 +143417,6 @@ ] }, { - "max": 127999, - "min": 1024, "type": "budget_tokens" } ], @@ -143662,9 +143474,6 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -143692,7 +143501,8 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.7-fast", - "last_updated": "2026-05-12", + "knowledge": "2026-01-31", + "last_updated": "2026-04-16", "limit": { "context": 1000000, "output": 128000 @@ -143711,9 +143521,6 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -143725,7 +143532,7 @@ ] } ], - "release_date": "2026-05-12", + "release_date": "2026-04-16", "structured_output": true, "temperature": false, "tool_call": true @@ -143761,9 +143568,6 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -143791,7 +143595,8 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.8-fast", - "last_updated": "2026-05-27", + "knowledge": "2026-01", + "last_updated": "2026-05-28", "limit": { "context": 1000000, "output": 128000 @@ -143810,9 +143615,6 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -143824,7 +143626,7 @@ ] } ], - "release_date": "2026-05-27", + "release_date": "2026-05-28", "structured_output": true, "temperature": false, "tool_call": true @@ -144004,9 +143806,6 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -144015,11 +143814,6 @@ "high", "max" ] - }, - { - "max": 127999, - "min": 1024, - "type": "budget_tokens" } ], "release_date": "2026-02-17", @@ -144074,36 +143868,6 @@ "temperature": false, "tool_call": true }, - "arcee-ai/coder-large": { - "attachment": false, - "cost": { - "input": 0.5, - "output": 0.8 - }, - "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", - "id": "arcee-ai/coder-large", - "knowledge": "2025-03-31", - "last_updated": "2025-05-05", - "limit": { - "context": 32768, - "output": 32768 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "Coder Large", - "open_weights": false, - "reasoning": false, - "release_date": "2025-05-05", - "structured_output": false, - "temperature": true, - "tool_call": false - }, "arcee-ai/trinity-large-thinking": { "attachment": false, "cost": { @@ -144666,8 +144430,8 @@ "attachment": false, "cost": { "cache_read": 0.135, - "input": 0.24, - "output": 0.9 + "input": 0.27, + "output": 1.12 }, "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", @@ -144676,7 +144440,7 @@ "last_updated": "2025-03-24", "limit": { "context": 163840, - "output": 16384 + "output": 65536 }, "modalities": { "input": [ @@ -144831,9 +144595,9 @@ "deepseek/deepseek-v3.1-terminus": { "attachment": false, "cost": { - "cache_read": 0.13, + "cache_read": 0.135, "input": 0.27, - "output": 0.95 + "output": 1 }, "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", @@ -144841,7 +144605,7 @@ "knowledge": "2025-03-31", "last_updated": "2025-09-22", "limit": { - "context": 163840, + "context": 131072, "output": 32768 }, "modalities": { @@ -144868,9 +144632,9 @@ "deepseek/deepseek-v3.2": { "attachment": false, "cost": { - "cache_read": 0.02145, - "input": 0.2145, - "output": 0.32175 + "cache_read": 0.1345, + "input": 0.269, + "output": 0.4 }, "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", @@ -144878,8 +144642,8 @@ "knowledge": "2024-07", "last_updated": "2025-12-01", "limit": { - "context": 131072, - "output": 64000 + "context": 163840, + "output": 65536 }, "modalities": { "input": [ @@ -144941,9 +144705,9 @@ "deepseek/deepseek-v4-flash": { "attachment": false, "cost": { - "cache_read": 0.018, - "input": 0.09, - "output": 0.18 + "cache_read": 0.02, + "input": 0.098, + "output": 0.196 }, "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", @@ -144969,9 +144733,6 @@ "open_weights": true, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -145016,9 +144777,6 @@ "open_weights": true, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -145095,7 +144853,7 @@ "last_updated": "2025-08-26", "limit": { "context": 32768, - "output": 32768 + "output": 8192 }, "modalities": { "input": [ @@ -145569,7 +145327,7 @@ "last_updated": "2026-06-30", "limit": { "context": 65536, - "output": 66000 + "output": 65536 }, "modalities": { "input": [ @@ -145896,8 +145654,9 @@ "google/gemma-3-27b-it": { "attachment": true, "cost": { + "cache_read": 0.04, "input": 0.08, - "output": 0.16 + "output": 0.45 }, "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", @@ -145906,7 +145665,7 @@ "last_updated": "2025-03-12", "limit": { "context": 131072, - "output": 16384 + "output": 131072 }, "modalities": { "input": [ @@ -145991,8 +145750,8 @@ "google/gemma-4-26b-a4b-it": { "attachment": true, "cost": { - "input": 0.06, - "output": 0.33 + "input": 0.1, + "output": 0.3 }, "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", @@ -146000,7 +145759,7 @@ "last_updated": "2026-04-02", "limit": { "context": 262144, - "output": 262144 + "output": 256000 }, "modalities": { "input": [ @@ -146065,8 +145824,9 @@ "google/gemma-4-31b-it": { "attachment": true, "cost": { - "input": 0.06, - "output": 0.35 + "cache_read": 0.12, + "input": 0.22, + "output": 0.55 }, "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", @@ -146074,7 +145834,7 @@ "last_updated": "2026-04-02", "limit": { "context": 262144, - "output": 8192 + "output": 262144 }, "modalities": { "input": [ @@ -146653,8 +146413,9 @@ "meta-llama/llama-3.1-8b-instruct": { "attachment": false, "cost": { - "input": 0.02, - "output": 0.03 + "cache_read": 0.025, + "input": 0.05, + "output": 0.08 }, "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", @@ -146663,7 +146424,7 @@ "last_updated": "2024-07-23", "limit": { "context": 131072, - "output": 16384 + "output": 131072 }, "modalities": { "input": [ @@ -146809,8 +146570,8 @@ "meta-llama/llama-3.3-70b-instruct": { "attachment": false, "cost": { - "input": 0.1, - "output": 0.32 + "input": 0.13, + "output": 0.4 }, "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", @@ -146819,7 +146580,7 @@ "last_updated": "2024-12-06", "limit": { "context": 131072, - "output": 16384 + "output": 128000 }, "modalities": { "input": [ @@ -146964,6 +146725,53 @@ "temperature": true, "tool_call": false }, + "meta/muse-spark-1.1": { + "attachment": true, + "cost": { + "cache_read": 0.15, + "input": 1.25, + "output": 4.25 + }, + "description": "Open Llama multimodal model for image understanding and text reasoning", + "family": "muse", + "id": "meta/muse-spark-1.1", + "last_updated": "2026-07-09", + "limit": { + "context": 1048576, + "output": 1048576 + }, + "modalities": { + "input": [ + "text", + "image", + "video", + "pdf", + "audio" + ], + "output": [ + "text" + ] + }, + "name": "Muse Spark 1.1", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + ], + "release_date": "2026-04-08", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "microsoft/phi-4": { "attachment": false, "cost": { @@ -147060,7 +146868,7 @@ "minimax/minimax-m1": { "attachment": false, "cost": { - "input": 0.4, + "input": 0.55, "output": 2.2 }, "description": "MiniMax model for chat, coding, office work, and agentic tasks", @@ -147227,8 +147035,9 @@ "minimax/minimax-m2.7": { "attachment": false, "cost": { - "input": 0.24, - "output": 0.96 + "cache_read": 0.06, + "input": 0.3, + "output": 1.2 }, "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", @@ -147236,7 +147045,7 @@ "last_updated": "2026-03-18", "limit": { "context": 204800, - "output": 196608 + "output": 131072 }, "modalities": { "input": [ @@ -147268,7 +147077,7 @@ "last_updated": "2026-06-01", "limit": { "context": 1048576, - "output": 131072 + "output": 512000 }, "modalities": { "input": [ @@ -147665,7 +147474,7 @@ "attachment": false, "cost": { "input": 0.02, - "output": 0.03 + "output": 0.04 }, "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", @@ -147674,7 +147483,7 @@ "last_updated": "2024-07-01", "limit": { "context": 131072, - "output": 131072 + "output": 16384 }, "modalities": { "input": [ @@ -147833,8 +147642,9 @@ "mistralai/mistral-small-3.2-24b-instruct": { "attachment": true, "cost": { - "input": 0.075, - "output": 0.2 + "cache_read": 0.01, + "input": 0.1, + "output": 0.3 }, "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", @@ -147842,7 +147652,7 @@ "knowledge": "2023-10-31", "last_updated": "2025-06-20", "limit": { - "context": 128000, + "context": 131072, "output": 16384 }, "modalities": { @@ -147993,7 +147803,6 @@ "moonshotai/kimi-k2-thinking": { "attachment": false, "cost": { - "cache_read": 0.15, "input": 0.6, "output": 2.5 }, @@ -148007,7 +147816,7 @@ "last_updated": "2025-11-06", "limit": { "context": 262144, - "output": 100352 + "output": 262144 }, "modalities": { "input": [ @@ -148029,9 +147838,9 @@ "moonshotai/kimi-k2.5": { "attachment": true, "cost": { - "cache_read": 0.203, - "input": 0.375, - "output": 2.025 + "cache_read": 0.095, + "input": 0.57, + "output": 2.85 }, "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", @@ -148043,7 +147852,7 @@ "last_updated": "2026-01", "limit": { "context": 262144, - "output": 256000 + "output": 262144 }, "modalities": { "input": [ @@ -148066,9 +147875,9 @@ "moonshotai/kimi-k2.6": { "attachment": true, "cost": { - "cache_read": 0.15, - "input": 0.66, - "output": 3.41 + "cache_read": 0.16, + "input": 0.95, + "output": 4 }, "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", @@ -148103,9 +147912,9 @@ "moonshotai/kimi-k2.7-code": { "attachment": true, "cost": { - "cache_read": 0.149, - "input": 0.719, - "output": 3.49 + "cache_read": 0.16, + "input": 0.75, + "output": 3.5 }, "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", @@ -148134,6 +147943,46 @@ "temperature": true, "tool_call": true }, + "moonshotai/kimi-k3": { + "attachment": true, + "cost": { + "cache_read": 0.3, + "input": 3, + "output": 15 + }, + "description": "Kimi multimodal agent model for visual understanding, coding, and planning", + "family": "kimi-k3", + "id": "moonshotai/kimi-k3", + "last_updated": "2026-07-16", + "limit": { + "context": 1048576, + "output": 1048576 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K3", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "max" + ] + } + ], + "release_date": "2026-07-16", + "structured_output": true, + "temperature": false, + "tool_call": true + }, "morph/morph-v3-fast": { "attachment": false, "cost": { @@ -148559,8 +148408,9 @@ "nvidia/nemotron-3-super-120b-a12b": { "attachment": false, "cost": { - "input": 0.08, - "output": 0.45 + "cache_read": 0.06, + "input": 0.21, + "output": 0.455 }, "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", @@ -148643,9 +148493,9 @@ "nvidia/nemotron-3-ultra-550b-a55b": { "attachment": false, "cost": { - "cache_read": 0.1, - "input": 0.5, - "output": 2.2 + "cache_read": 0.2, + "input": 0.6, + "output": 3.6 }, "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", @@ -149144,6 +148994,7 @@ "openai/gpt-4o": { "attachment": true, "cost": { + "cache_read": 1.25, "input": 2.5, "output": 10 }, @@ -149783,7 +149634,7 @@ "openai/gpt-5.1-chat": { "attachment": true, "cost": { - "cache_read": 0.13, + "cache_read": 0.125, "input": 1.25, "output": 10 }, @@ -149794,7 +149645,7 @@ "last_updated": "2025-11-13", "limit": { "context": 128000, - "output": 32000 + "output": 16384 }, "modalities": { "input": [ @@ -149817,7 +149668,7 @@ "openai/gpt-5.1-codex": { "attachment": true, "cost": { - "cache_read": 0.13, + "cache_read": 0.125, "input": 1.25, "output": 10 }, @@ -150184,6 +150035,7 @@ { "type": "effort", "values": [ + "none", "low", "medium", "high", @@ -150912,8 +150764,8 @@ "openai/gpt-oss-120b": { "attachment": false, "cost": { - "input": 0.03, - "output": 0.15 + "input": 0.037, + "output": 0.17 }, "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", @@ -150952,8 +150804,9 @@ "openai/gpt-oss-20b": { "attachment": false, "cost": { - "input": 0.029, - "output": 0.14 + "cache_read": 0.03, + "input": 0.03, + "output": 0.13 }, "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", @@ -152132,8 +151985,9 @@ "qwen/qwen2.5-vl-72b-instruct": { "attachment": true, "cost": { - "input": 0.25, - "output": 0.75 + "cache_read": 0.4, + "input": 0.8, + "output": 1 }, "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", @@ -152340,8 +152194,8 @@ "qwen/qwen3-30b-a3b-instruct-2507": { "attachment": false, "cost": { - "input": 0.04815, - "output": 0.19305 + "input": 0.1, + "output": 0.3 }, "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", @@ -152349,7 +152203,7 @@ "knowledge": "2025-06-30", "last_updated": "2025-07-29", "limit": { - "context": 131072, + "context": 262144, "output": 32000 }, "modalities": { @@ -152475,8 +152329,9 @@ "qwen/qwen3-coder": { "attachment": false, "cost": { - "input": 0.22, - "output": 1.8 + "cache_read": 0.1, + "input": 0.3, + "output": 1 }, "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", @@ -152735,7 +152590,8 @@ "qwen/qwen3-next-80b-a3b-instruct": { "attachment": false, "cost": { - "input": 0.09, + "cache_read": 0.07, + "input": 0.1, "output": 1.1 }, "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", @@ -152745,7 +152601,7 @@ "last_updated": "2025-09", "limit": { "context": 262144, - "output": 16384 + "output": 262144 }, "modalities": { "input": [ @@ -152829,9 +152685,9 @@ "qwen/qwen3-vl-235b-a22b-instruct": { "attachment": true, "cost": { - "cache_read": 0.11, - "input": 0.2, - "output": 0.88 + "cache_read": 0.1, + "input": 0.21, + "output": 1.9 }, "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", @@ -152839,8 +152695,8 @@ "knowledge": "2025-03-31", "last_updated": "2025-09-23", "limit": { - "context": 262144, - "output": 16384 + "context": 131072, + "output": 32768 }, "modalities": { "input": [ @@ -153063,7 +152919,7 @@ "last_updated": "2026-02-23", "limit": { "context": 262144, - "output": 262144 + "output": 65536 }, "modalities": { "input": [ @@ -153128,7 +152984,6 @@ "qwen/qwen3.5-35b-a3b": { "attachment": true, "cost": { - "cache_read": 0.05, "input": 0.14, "output": 1 }, @@ -153138,7 +152993,7 @@ "last_updated": "2026-02-23", "limit": { "context": 262144, - "output": 81920 + "output": 262144 }, "modalities": { "input": [ @@ -153166,17 +153021,17 @@ "qwen/qwen3.5-397b-a17b": { "attachment": true, "cost": { - "cache_read": 0.111, - "input": 0.385, - "output": 2.45 + "cache_read": 0.225, + "input": 0.45, + "output": 3 }, "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "id": "qwen/qwen3.5-397b-a17b", "last_updated": "2026-02-15", "limit": { - "context": 256000, - "output": 64000 + "context": 262144, + "output": 65536 }, "modalities": { "input": [ @@ -153369,8 +153224,8 @@ "qwen/qwen3.6-27b": { "attachment": true, "cost": { - "input": 0.289, - "output": 2.4 + "input": 0.45, + "output": 2.7 }, "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", @@ -153378,7 +153233,7 @@ "last_updated": "2026-04-22", "limit": { "context": 262144, - "output": 131072 + "output": 65536 }, "modalities": { "input": [ @@ -153572,10 +153427,10 @@ "qwen/qwen3.7-max": { "attachment": false, "cost": { - "cache_read": 0.25, - "cache_write": 1.5625, - "input": 1.25, - "output": 3.75 + "cache_read": 0.295, + "cache_write": 1.84375, + "input": 1.475, + "output": 4.425 }, "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", @@ -154022,9 +153877,9 @@ "tencent/hy3": { "attachment": false, "cost": { - "cache_read": 0.035, - "input": 0.14, - "output": 0.58 + "cache_read": 0.05, + "input": 0.2, + "output": 0.8 }, "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hy3", @@ -154424,7 +154279,9 @@ "type": "effort", "values": [ "low", - "high" + "medium", + "high", + "xhigh" ] } ], @@ -154589,8 +154446,8 @@ "xiaomi/mimo-v2.5": { "attachment": true, "cost": { - "cache_read": 0.028, - "input": 0.105, + "cache_read": 0.0028, + "input": 0.14, "output": 0.28 }, "description": "Open MiMo model for multimodal coding agents and long-context automation", @@ -154784,9 +154641,9 @@ "z-ai/glm-4.6": { "attachment": false, "cost": { - "cache_read": 0.08, - "input": 0.43, - "output": 1.75 + "cache_read": 0.1, + "input": 0.5, + "output": 2 }, "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", @@ -154794,8 +154651,8 @@ "knowledge": "2025-04", "last_updated": "2025-09-30", "limit": { - "context": 200000, - "output": 16384 + "context": 202752, + "output": 131072 }, "modalities": { "input": [ @@ -154888,8 +154745,7 @@ "z-ai/glm-4.7-flash": { "attachment": false, "cost": { - "cache_read": 0.01, - "input": 0.06, + "input": 0.0605, "output": 0.4 }, "description": "Budget GLM lane for fast coding help, routing, and everyday automation", @@ -154901,8 +154757,8 @@ "knowledge": "2025-04", "last_updated": "2026-01-19", "limit": { - "context": 202752, - "output": 16384 + "context": 200000, + "output": 131072 }, "modalities": { "input": [ @@ -154924,9 +154780,9 @@ "z-ai/glm-5": { "attachment": false, "cost": { - "cache_read": 0.12, - "input": 0.6, - "output": 1.92 + "cache_read": 0.19, + "input": 0.95, + "output": 3.15 }, "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", @@ -154937,7 +154793,7 @@ "last_updated": "2026-02-12", "limit": { "context": 202752, - "output": 128000 + "output": 202752 }, "modalities": { "input": [ @@ -154971,7 +154827,7 @@ }, "last_updated": "2026-03-16", "limit": { - "context": 262144, + "context": 202752, "output": 131072 }, "modalities": { @@ -155029,9 +154885,9 @@ "z-ai/glm-5.2": { "attachment": false, "cost": { - "cache_read": 0.169, - "input": 0.91, - "output": 2.86 + "cache_read": 0.17342, + "input": 0.9338, + "output": 2.9348 }, "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", @@ -155042,7 +154898,7 @@ "last_updated": "2026-06-13", "limit": { "context": 1048576, - "output": 128000 + "output": 131072 }, "modalities": { "input": [ @@ -155223,9 +155079,6 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ @@ -155253,6 +155106,7 @@ "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "id": "~anthropic/claude-sonnet-latest", + "knowledge": "2026-01-31", "last_updated": "2026-04-27", "limit": { "context": 1000000, @@ -155272,22 +155126,15 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ - { - "type": "toggle" - }, { "type": "effort", "values": [ "low", "medium", "high", + "xhigh", "max" ] - }, - { - "max": 127999, - "min": 1024, - "type": "budget_tokens" } ], "release_date": "2026-04-27", @@ -155349,13 +155196,30 @@ "cost": { "cache_read": 0.2, "cache_write": 0.375, + "context_over_200k": { + "cache_read": 0.4, + "input": 4, + "output": 18 + }, "input": 2, "output": 12, - "reasoning": 12 + "reasoning": 12, + "tiers": [ + { + "cache_read": 0.4, + "input": 4, + "output": 18, + "tier": { + "size": 200000, + "type": "context" + } + } + ] }, "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "id": "~google/gemini-pro-latest", + "knowledge": "2025-01", "last_updated": "2026-04-27", "limit": { "context": 1048576, @@ -155394,16 +155258,16 @@ "~moonshotai/kimi-latest": { "attachment": true, "cost": { - "cache_read": 0.15, - "input": 0.66, - "output": 3.41 + "cache_read": 0.3, + "input": 3, + "output": 15 }, "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "id": "~moonshotai/kimi-latest", "last_updated": "2026-04-27", "limit": { - "context": 262144, + "context": 1048576, "output": 262144 }, "modalities": { @@ -155418,10 +155282,17 @@ "name": "MoonshotAI Kimi Latest", "open_weights": false, "reasoning": true, - "reasoning_options": [], + "reasoning_options": [ + { + "type": "effort", + "values": [ + "max" + ] + } + ], "release_date": "2026-04-27", "structured_output": true, - "temperature": true, + "temperature": false, "tool_call": true }, "~openai/gpt-latest": { @@ -155462,7 +155333,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ] } ], @@ -155549,7 +155421,6 @@ { "type": "effort", "values": [ - "none", "low", "medium", "high" @@ -156532,16 +156403,16 @@ "google/gemini-flash-latest": { "attachment": true, "cost": { - "cache_read": 0.075, - "input": 0.5, - "input_audio": 1, - "output": 3 + "cache_read": 0.15, + "input": 1.5, + "input_audio": 1.5, + "output": 9 }, "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "id": "google/gemini-flash-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-19", "limit": { "context": 1048576, "output": 65536 @@ -156550,8 +156421,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -156561,8 +156432,18 @@ "name": "Gemini Flash Latest", "open_weights": false, "reasoning": true, - "reasoning_options": [], - "release_date": "2025-09-25", + "reasoning_options": [ + { + "type": "effort", + "values": [ + "minimal", + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-05-19", "structured_output": true, "temperature": true, "tool_call": true @@ -156572,13 +156453,14 @@ "cost": { "cache_read": 0.025, "input": 0.25, + "input_audio": 0.5, "output": 1.5 }, "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "id": "google/gemini-flash-lite-latest", "knowledge": "2025-01", - "last_updated": "2025-09-25", + "last_updated": "2026-05-07", "limit": { "context": 1048576, "output": 65536 @@ -156587,8 +156469,8 @@ "input": [ "text", "image", - "audio", "video", + "audio", "pdf" ], "output": [ @@ -156598,8 +156480,18 @@ "name": "Gemini Flash-Lite Latest", "open_weights": false, "reasoning": true, - "reasoning_options": [], - "release_date": "2025-09-25", + "reasoning_options": [ + { + "type": "effort", + "values": [ + "minimal", + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-05-07", "structured_output": true, "temperature": true, "tool_call": true @@ -160256,7 +160148,7 @@ "last_updated": "2025-06-30", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -160300,7 +160192,7 @@ "last_updated": "2026-02-25", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -160433,7 +160325,7 @@ "last_updated": "2025-03-31", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -160522,7 +160414,7 @@ "last_updated": "2025-03-31", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -160565,7 +160457,7 @@ "last_updated": "2025-07-31", "limit": { "context": 262144, - "output": 4096 + "output": 131072 }, "modalities": { "input": [ @@ -160611,7 +160503,7 @@ "last_updated": "2025-04-28", "limit": { "context": 131072, - "output": 4096 + "output": 131072 }, "modalities": { "input": [ @@ -160656,7 +160548,7 @@ "last_updated": "2026-02-23", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -160702,7 +160594,7 @@ "last_updated": "2026-04-22", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -161691,7 +161583,7 @@ "last_updated": "2026-05-31", "limit": { "context": 262144, - "output": 4096 + "output": 131072 }, "modalities": { "input": [ @@ -161735,7 +161627,7 @@ "last_updated": "2025-02-28", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -161778,7 +161670,7 @@ "last_updated": "2026-05-31", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -161822,7 +161714,7 @@ "last_updated": "2026-04-02", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -161868,7 +161760,7 @@ "last_updated": "2026-04-02", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -161915,7 +161807,7 @@ "last_updated": "2026-04-02", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -162728,7 +162620,7 @@ "last_updated": "2024-12-06", "limit": { "context": 131072, - "output": 4096 + "output": 131072 }, "modalities": { "input": [ @@ -162817,7 +162709,7 @@ "last_updated": "2023-04-30", "limit": { "context": 32768, - "output": 4096 + "output": 32768 }, "modalities": { "input": [ @@ -168278,38 +168170,6 @@ ], "id": "privatemode-ai", "models": { - "gemma-3-27b": { - "attachment": true, - "cost": { - "input": 0, - "output": 0 - }, - "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", - "family": "gemma", - "id": "gemma-3-27b", - "knowledge": "2024-08", - "last_updated": "2025-03-12", - "limit": { - "context": 128000, - "output": 8192 - }, - "modalities": { - "input": [ - "text", - "image" - ], - "output": [ - "text" - ] - }, - "name": "Gemma 3 27B", - "open_weights": true, - "reasoning": false, - "release_date": "2025-03-12", - "structured_output": true, - "temperature": true, - "tool_call": true - }, "gpt-oss-120b": { "attachment": false, "cost": { @@ -168351,33 +168211,39 @@ "temperature": true, "tool_call": true }, - "qwen3-coder-30b-a3b": { - "attachment": false, + "kimi-k2.6": { + "attachment": true, "cost": { "input": 0, "output": 0 }, - "description": "Qwen coding model for software agents, repository edits, and code reasoning", - "family": "qwen", - "id": "qwen3-coder-30b-a3b", - "knowledge": "2025-04", - "last_updated": "2025-04", + "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", + "family": "kimi-k2", + "id": "kimi-k2.6", + "knowledge": "2025-01", + "last_updated": "2026-04-21", "limit": { - "context": 128000, - "output": 32768 + "context": 262144, + "output": 262144 }, "modalities": { "input": [ - "text" + "text", + "image" ], "output": [ "text" ] }, - "name": "Qwen3-Coder 30B-A3B", + "name": "Kimi K2.6", "open_weights": true, - "reasoning": false, - "release_date": "2025-04", + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-21", "structured_output": true, "temperature": true, "tool_call": true @@ -168413,6 +168279,36 @@ "temperature": true, "tool_call": false }, + "voxtral-mini-3b": { + "attachment": true, + "cost": { + "input": 0, + "output": 0 + }, + "description": "Speech-to-text model for audio transcription, translation, and audio understanding", + "family": "voxtral", + "id": "voxtral-mini-3b", + "last_updated": "2025-07", + "limit": { + "context": 32000, + "output": 32000 + }, + "modalities": { + "input": [ + "audio" + ], + "output": [ + "text" + ] + }, + "name": "Voxtral Mini 3B", + "open_weights": true, + "reasoning": false, + "release_date": "2025-07", + "structured_output": false, + "temperature": true, + "tool_call": false + }, "whisper-large-v3": { "attachment": true, "cost": { @@ -183395,6 +183291,67 @@ "name": "The Grid AI", "npm": "@ai-sdk/openai-compatible" }, + "thinkingmachines": { + "api": "https://tinker.thinkingmachines.dev/services/tinker-prod/oai/api/v1", + "doc": "https://tinker-docs.thinkingmachines.ai/tinker/compatible-apis/openai/", + "env": [ + "TINKER_API_KEY" + ], + "id": "thinkingmachines", + "models": { + "inkling": { + "attachment": true, + "cost": { + "cache_read": 0.748, + "input": 3.74, + "output": 9.36 + }, + "description": "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio", + "family": "ling", + "id": "inkling", + "interleaved": { + "field": "reasoning_content" + }, + "last_updated": "2026-07-15", + "limit": { + "context": 256000, + "output": 256000 + }, + "modalities": { + "input": [ + "text", + "image", + "audio", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Inkling", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + ], + "release_date": "2026-07-15", + "temperature": true, + "tool_call": true + } + }, + "name": "Thinking Machines", + "npm": "@ai-sdk/openai-compatible" + }, "tinfoil": { "api": "https://inference.tinfoil.sh/v1", "doc": "https://docs.tinfoil.sh", @@ -186556,44 +186513,6 @@ ], "id": "venice", "models": { - "aion-labs-aion-2-0": { - "attachment": false, - "cost": { - "cache_read": 0.25, - "input": 1, - "output": 2 - }, - "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", - "id": "aion-labs-aion-2-0", - "last_updated": "2026-06-11", - "limit": { - "context": 128000, - "output": 32768 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "Aion 2.0", - "open_weights": false, - "reasoning": true, - "reasoning_options": [ - { - "type": "effort", - "values": [ - "low", - "medium", - "high" - ] - } - ], - "release_date": "2026-03-24", - "tool_call": false - }, "aion-labs-aion-3-0": { "attachment": false, "cost": { @@ -186701,7 +186620,18 @@ "name": "Claude Fable 5", "open_weights": false, "reasoning": true, - "reasoning_options": [], + "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + ], "release_date": "2026-06-10", "structured_output": true, "temperature": false, @@ -191656,6 +191586,54 @@ "temperature": true, "tool_call": true }, + "anthropic/claude-opus-4.7-fast": { + "attachment": true, + "cost": { + "cache_read": 3, + "cache_write": 37.5, + "input": 30, + "output": 150 + }, + "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", + "family": "claude-opus", + "id": "anthropic/claude-opus-4.7-fast", + "knowledge": "2026-01-31", + "last_updated": "2026-04-16", + "limit": { + "context": 1000000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Claude Opus 4.7 (Fast)", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "low", + "medium", + "high", + "xhigh" + ] + } + ], + "release_date": "2026-04-16", + "temperature": true, + "tool_call": true + }, "anthropic/claude-opus-4.8": { "attachment": true, "cost": { @@ -191704,6 +191682,54 @@ "temperature": true, "tool_call": true }, + "anthropic/claude-opus-4.8-fast": { + "attachment": true, + "cost": { + "cache_read": 1, + "cache_write": 12.5, + "input": 10, + "output": 50 + }, + "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", + "family": "claude-opus", + "id": "anthropic/claude-opus-4.8-fast", + "knowledge": "2026-01", + "last_updated": "2026-05-28", + "limit": { + "context": 1000000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Claude Opus 4.8 (Fast)", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + }, + { + "type": "effort", + "values": [ + "low", + "medium", + "high", + "xhigh" + ] + } + ], + "release_date": "2026-05-28", + "temperature": true, + "tool_call": true + }, "anthropic/claude-sonnet-4": { "attachment": true, "cost": { @@ -195637,6 +195663,40 @@ "temperature": true, "tool_call": true }, + "moonshotai/kimi-k3": { + "attachment": true, + "cost": { + "cache_read": 0.3, + "input": 3, + "output": 15 + }, + "description": "Kimi multimodal agent model for visual understanding, coding, and planning", + "family": "kimi-k3", + "id": "moonshotai/kimi-k3", + "last_updated": "2026-07-16", + "limit": { + "context": 1000000, + "output": 131072 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K3", + "open_weights": false, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-07-16", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "morph/morph-v3-fast": { "attachment": false, "cost": { @@ -197727,6 +197787,31 @@ "temperature": true, "tool_call": false }, + "openai/gpt-realtime-whisper": { + "attachment": false, + "description": "Streaming speech-to-text model for low-latency transcript deltas from live audio", + "family": "whisper", + "id": "openai/gpt-realtime-whisper", + "last_updated": "2026-05-07", + "limit": { + "context": 0, + "output": 0 + }, + "modalities": { + "input": [ + "audio" + ], + "output": [ + "text" + ] + }, + "name": "gpt-realtime-whisper", + "open_weights": false, + "reasoning": false, + "release_date": "2026-05-07", + "temperature": true, + "tool_call": false + }, "openai/o1": { "attachment": true, "cost": { @@ -198596,6 +198681,39 @@ "temperature": true, "tool_call": true }, + "thinkingmachines/inkling": { + "attachment": true, + "cost": { + "cache_read": 0.17, + "input": 1, + "output": 4.05 + }, + "description": "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio", + "family": "ling", + "id": "thinkingmachines/inkling", + "last_updated": "2026-07-15", + "limit": { + "context": 256000, + "output": 256000 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Inkling", + "open_weights": true, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-07-15", + "temperature": true, + "tool_call": true + }, "voyage/rerank-2.5": { "attachment": false, "description": "Reranking model for improving retrieval quality in search and recommendation systems", @@ -199533,8 +199651,7 @@ }, "modalities": { "input": [ - "text", - "pdf" + "text" ], "output": [ "text"