From 186677e7ed4d3f0771715fc62d1243023c17cfbb Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" Date: Wed, 8 Jul 2026 23:41:20 +0000 Subject: [PATCH] chore: update models.dev catalogue --- assets/models-dev/catalog.json | 1748 ++++++++++++++++++++++++-------- 1 file changed, 1334 insertions(+), 414 deletions(-) diff --git a/assets/models-dev/catalog.json b/assets/models-dev/catalog.json index abb2381cef..f97faf27d1 100644 --- a/assets/models-dev/catalog.json +++ b/assets/models-dev/catalog.json @@ -5974,6 +5974,7 @@ "family": "claude-opus", "id": "claude-opus-4-8", "interleaved": true, + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 200000, @@ -6024,6 +6025,7 @@ "interleaved": { "field": "reasoning_content" }, + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 200000, @@ -15990,6 +15992,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic.claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -16267,6 +16270,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "au.anthropic.claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -16780,6 +16784,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "eu.anthropic.claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -17198,6 +17203,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "global.anthropic.claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -17565,6 +17571,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "jp.anthropic.claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -19281,6 +19288,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "us.anthropic.claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -20114,7 +20122,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-5", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-24", "limit": { "context": 200000, @@ -20163,7 +20171,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-5-20251101", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-01", "limit": { "context": 200000, @@ -20369,6 +20377,7 @@ }, "family": "claude-opus", "id": "claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -32724,6 +32733,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -36519,6 +36529,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -39281,7 +39292,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "databricks-claude-opus-4-5", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-24", "limit": { "context": 200000, @@ -41626,7 +41637,7 @@ "openai/gpt-oss-120b": { "attachment": false, "cost": { - "input": 0.039, + "input": 0.037, "output": 0.17 }, "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", @@ -46096,6 +46107,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -48579,6 +48591,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -50225,7 +50238,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4.5", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-24", "limit": { "context": 200000, @@ -50420,6 +50433,7 @@ }, "family": "claude-opus", "id": "claude-opus-4.8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 200000, @@ -53926,6 +53940,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -55211,6 +55226,38 @@ "temperature": true, "tool_call": true }, + "gemini-omni-flash-preview": { + "attachment": true, + "cost": { + "input": 1.5, + "output": 17.5 + }, + "description": "Video generation and editing model for fast, conversational text- and image-to-video workflows", + "family": "gemini", + "id": "gemini-omni-flash-preview", + "last_updated": "2026-06-30", + "limit": { + "context": 131072, + "output": 65536 + }, + "modalities": { + "input": [ + "text", + "image", + "video" + ], + "output": [ + "video" + ] + }, + "name": "Gemini Omni Flash Preview", + "open_weights": false, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-06-30", + "temperature": true, + "tool_call": false + }, "gemma-4-26b-a4b-it": { "attachment": true, "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", @@ -55424,7 +55471,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-5@20251101", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-01", "limit": { "context": 200000, @@ -55630,6 +55677,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8@default", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -57092,7 +57140,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-5@20251101", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-01", "limit": { "context": 200000, @@ -57289,6 +57337,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8@default", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -66586,6 +66635,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -79941,7 +79991,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-5-20251101", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-01", "limit": { "context": 200000, @@ -80082,6 +80132,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -83366,6 +83417,48 @@ "temperature": true, "tool_call": true }, + "grok-4-5": { + "attachment": true, + "cost": { + "cache_read": 0.5, + "input": 2, + "output": 6 + }, + "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", + "family": "grok", + "id": "grok-4-5", + "last_updated": "2026-07-08", + "limit": { + "context": 500000, + "output": 500000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Grok 4.5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-07-08", + "structured_output": false, + "temperature": true, + "tool_call": true + }, "grok-build-0-1": { "attachment": true, "cost": { @@ -87774,7 +87867,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4-5-20251101", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-01", "limit": { "context": 200000, @@ -87966,6 +88059,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -117452,7 +117546,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-5", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-24", "limit": { "context": 200000, @@ -129300,6 +129394,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -131068,6 +131163,64 @@ "temperature": false, "tool_call": true }, + "grok-4.5": { + "attachment": true, + "cost": { + "cache_read": 0.5, + "context_over_200k": { + "cache_read": 1, + "input": 4, + "output": 12 + }, + "input": 2, + "output": 6, + "tiers": [ + { + "cache_read": 1, + "input": 4, + "output": 12, + "tier": { + "size": 200000, + "type": "context" + } + } + ] + }, + "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", + "family": "grok", + "id": "grok-4.5", + "last_updated": "2026-07-08", + "limit": { + "context": 500000, + "output": 500000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Grok 4.5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-07-08", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "grok-build-0.1": { "attachment": true, "cost": { @@ -133665,7 +133818,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.5", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-24", "limit": { "context": 200000, @@ -133907,6 +134060,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -138339,7 +138493,7 @@ "moonshotai/kimi-k2.6": { "attachment": true, "cost": { - "cache_read": 0.3, + "cache_read": 0.14, "input": 0.65, "output": 3.41 }, @@ -144164,16 +144318,16 @@ "tencent/hy3": { "attachment": false, "cost": { - "cache_read": 0.5, - "input": 0.2, - "output": 0.8 + "cache_read": 0.035, + "input": 0.14, + "output": 0.58 }, "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hy3", "id": "tencent/hy3", "last_updated": "2026-07-06", "limit": { - "context": 202752, + "context": 262144, "output": 131072 }, "modalities": { @@ -144635,6 +144789,65 @@ "temperature": true, "tool_call": true }, + "x-ai/grok-4.5": { + "attachment": true, + "cost": { + "cache_read": 0.5, + "context_over_200k": { + "cache_read": 1, + "input": 4, + "output": 12 + }, + "input": 2, + "output": 6, + "tiers": [ + { + "cache_read": 1, + "input": 4, + "output": 12, + "tier": { + "size": 200000, + "type": "context" + } + } + ] + }, + "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", + "family": "grok", + "id": "x-ai/grok-4.5", + "last_updated": "2026-07-08", + "limit": { + "context": 500000, + "output": 500000 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Grok 4.5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-07-08", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "x-ai/grok-build-0.1": { "attachment": true, "cost": { @@ -145112,9 +145325,9 @@ "z-ai/glm-5.2": { "attachment": false, "cost": { - "cache_read": 0.18, - "input": 0.93, - "output": 3 + "cache_read": 0.078, + "input": 0.42, + "output": 1.32 }, "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", @@ -145124,8 +145337,8 @@ }, "last_updated": "2026-06-13", "limit": { - "context": 1048576, - "output": 32768 + "context": 1024000, + "output": 128000 }, "modalities": { "input": [ @@ -145477,7 +145690,7 @@ "~moonshotai/kimi-latest": { "attachment": true, "cost": { - "cache_read": 0.3, + "cache_read": 0.14, "input": 0.65, "output": 3.41 }, @@ -145602,16 +145815,16 @@ "~x-ai/grok-latest": { "attachment": true, "cost": { - "cache_read": 0.2, - "input": 1.25, - "output": 2.5 + "cache_read": 0.5, + "input": 2, + "output": 6 }, "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "id": "~x-ai/grok-latest", "last_updated": "2026-07-08", "limit": { - "context": 1000000, + "context": 500000, "output": 1000000 }, "modalities": { @@ -145771,7 +145984,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.5", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-24", "limit": { "context": 200000, @@ -150648,6 +150861,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1048576, @@ -161691,6 +161905,14 @@ "open_weights": false, "reasoning": true, "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high" + ] + }, { "min": 1024, "type": "budget_tokens" @@ -161825,7 +162047,8 @@ "values": [ "low", "medium", - "high" + "high", + "max" ] }, { @@ -161873,7 +162096,9 @@ "values": [ "low", "medium", - "high" + "high", + "xhigh", + "max" ] } ], @@ -161881,6 +162106,53 @@ "temperature": false, "tool_call": true }, + "anthropic--claude-4.8-opus": { + "attachment": true, + "cost": { + "cache_read": 0.5, + "cache_write": 6.25, + "input": 5, + "output": 25 + }, + "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", + "family": "claude-opus", + "id": "anthropic--claude-4.8-opus", + "knowledge": "2026-01", + "last_updated": "2026-05-28", + "limit": { + "context": 1000000, + "output": 128000 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "anthropic--claude-4.8-opus", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + ], + "release_date": "2026-05-28", + "structured_output": true, + "temperature": false, + "tool_call": true + }, "gemini-2.5-flash": { "attachment": true, "cost": { @@ -166538,6 +166810,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -171850,7 +172123,7 @@ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-5", - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2026-06-11", "limit": { "context": 198000, @@ -171990,6 +172263,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8", + "knowledge": "2026-01", "last_updated": "2026-06-11", "limit": { "context": 1000000, @@ -172024,6 +172298,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "claude-opus-4-8-fast", + "knowledge": "2026-01", "last_updated": "2026-06-11", "limit": { "context": 1000000, @@ -172728,6 +173003,55 @@ "temperature": true, "tool_call": true }, + "grok-4-5": { + "attachment": true, + "cost": { + "cache_read": 0.57, + "context_over_200k": { + "cache_read": 1.13, + "input": 4.53, + "output": 13.6 + }, + "input": 2.27, + "output": 6.8, + "tiers": [ + { + "cache_read": 1.13, + "input": 4.53, + "output": 13.6, + "tier": { + "size": 200000, + "type": "context" + } + } + ] + }, + "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", + "family": "grok", + "id": "grok-4-5", + "last_updated": "2026-07-08", + "limit": { + "context": 500000, + "output": 32000 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Grok 4.5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-07-07", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "grok-build-0-1": { "attachment": true, "cost": { @@ -176360,7 +176684,7 @@ "family": "claude-opus", "id": "anthropic/claude-opus-4.5", "interleaved": true, - "knowledge": "2025-03-31", + "knowledge": "2025-05", "last_updated": "2025-11-24", "limit": { "context": 200000, @@ -176503,6 +176827,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000, @@ -178435,6 +178760,38 @@ "temperature": true, "tool_call": false }, + "google/gemini-omni-flash-preview": { + "attachment": true, + "cost": { + "input": 1.5, + "output": 9 + }, + "description": "Omni-modal model for text, vision, audio, and multimodal agent tasks", + "family": "gemini", + "id": "google/gemini-omni-flash-preview", + "last_updated": "2026-06-30", + "limit": { + "context": 1000000, + "output": 57920 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Gemini Omni Flash Preview", + "open_weights": false, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-06-30", + "temperature": true, + "tool_call": false + }, "google/gemma-4-26b-a4b-it": { "attachment": true, "cost": { @@ -183875,6 +184232,49 @@ "temperature": true, "tool_call": true }, + "xai/grok-4.5": { + "attachment": true, + "cost": { + "cache_read": 0.5, + "input": 2, + "output": 6 + }, + "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", + "family": "grok", + "id": "xai/grok-4.5", + "last_updated": "2026-07-08", + "limit": { + "context": 500000, + "output": 500000 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Grok 4.5", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high" + ] + } + ], + "release_date": "2026-07-08", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "xai/grok-build-0.1": { "attachment": true, "cost": { @@ -186149,16 +186549,47 @@ ], "id": "wandb", "models": { + "JetBrains/Mellum2-12B-A2.5B-Instruct": { + "attachment": false, + "cost": { + "cache_read": 0.05, + "input": 0.05, + "output": 0.1 + }, + "description": "Mellum2-12B-A2.5B-Instruct is a fast MoE model with 131K context built for coding, tool use, and low-latency AI workflows.", + "id": "JetBrains/Mellum2-12B-A2.5B-Instruct", + "last_updated": "2026-06-01", + "limit": { + "context": 131072, + "output": 131072 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "Mellum2 12B A2.5B", + "open_weights": true, + "reasoning": false, + "release_date": "2026-06-01", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "MiniMaxAI/MiniMax-M2.5": { "attachment": false, "cost": { + "cache_read": 0.3, "input": 0.3, "output": 1.2 }, - "description": "MiniMax model for chat, coding, office work, and agentic tasks", - "family": "minimax", + "description": "MoE model with a highly sparse architecture designed for high-throughput and low latency with strong coding capabilities.", + "family": "minimax-m2.5", "id": "MiniMaxAI/MiniMax-M2.5", - "last_updated": "2026-03-12", + "last_updated": "2026-02-12", "limit": { "context": 196608, "output": 196608 @@ -186183,13 +186614,14 @@ "OpenPipe/Qwen3-14B-Instruct": { "attachment": false, "cost": { + "cache_read": 0.05, "input": 0.05, "output": 0.22 }, - "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", + "description": "An efficient multilingual, dense, instruction-tuned model, optimized by OpenPipe for building agents with finetuning.", "family": "qwen", "id": "OpenPipe/Qwen3-14B-Instruct", - "last_updated": "2026-03-12", + "last_updated": "2025-04-29", "limit": { "context": 32768, "output": 32768 @@ -186202,7 +186634,7 @@ "text" ] }, - "name": "OpenPipe Qwen3 14B Instruct", + "name": "Qwen3 14B Instruct", "open_weights": true, "reasoning": false, "release_date": "2025-04-29", @@ -186213,14 +186645,15 @@ "Qwen/Qwen3-235B-A22B-Instruct-2507": { "attachment": false, "cost": { + "cache_read": 0.1, "input": 0.1, "output": 0.1 }, - "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", + "description": "Efficient multilingual, Mixture-of-Experts, instruction-tuned model, optimized for logical reasoning.", "family": "qwen", "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "knowledge": "2025-04", - "last_updated": "2026-03-12", + "last_updated": "2025-07-22", "limit": { "context": 262144, "output": 262144 @@ -186233,10 +186666,11 @@ "text" ] }, - "name": "Qwen3 235B A22B Instruct 2507", + "name": "Qwen3 235B A22B-2507", "open_weights": true, "reasoning": false, - "release_date": "2025-04-28", + "release_date": "2025-07-22", + "status": "deprecated", "structured_output": true, "temperature": true, "tool_call": true @@ -186244,14 +186678,15 @@ "Qwen/Qwen3-235B-A22B-Thinking-2507": { "attachment": false, "cost": { + "cache_read": 0.1, "input": 0.1, "output": 0.1 }, - "description": "Qwen reasoning model for deliberate problem solving, math, and coding", + "description": "High-performance Mixture-of-Experts model optimized for structured reasoning, math, and long-form generation.", "family": "qwen", "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "knowledge": "2025-04", - "last_updated": "2026-03-12", + "last_updated": "2025-07-25", "limit": { "context": 262144, "output": 262144 @@ -186264,11 +186699,12 @@ "text" ] }, - "name": "Qwen3-235B-A22B-Thinking-2507", + "name": "Qwen3 235B A22B Thinking-2507", "open_weights": true, "reasoning": true, "reasoning_options": [], "release_date": "2025-07-25", + "status": "deprecated", "structured_output": true, "temperature": true, "tool_call": true @@ -186276,13 +186712,14 @@ "Qwen/Qwen3-30B-A3B-Instruct-2507": { "attachment": false, "cost": { + "cache_read": 0.1, "input": 0.1, "output": 0.3 }, - "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", + "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B MoE instruction-tuned model with enhanced reasoning, coding, and long-context understanding.", "family": "qwen", "id": "Qwen/Qwen3-30B-A3B-Instruct-2507", - "last_updated": "2026-03-12", + "last_updated": "2025-07-29", "limit": { "context": 262144, "output": 262144 @@ -186306,14 +186743,15 @@ "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "attachment": false, "cost": { + "cache_read": 1, "input": 1, "output": 1.5 }, - "description": "Qwen coding model for software agents, repository edits, and code reasoning", + "description": "Mixture-of-Experts model optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning.", "family": "qwen", "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "knowledge": "2025-04", - "last_updated": "2026-03-12", + "last_updated": "2025-07-22", "limit": { "context": 262144, "output": 262144 @@ -186326,10 +186764,159 @@ "text" ] }, - "name": "Qwen3-Coder-480B-A35B-Instruct", + "name": "Qwen3 Coder 480B A35B", "open_weights": true, "reasoning": false, - "release_date": "2025-07-23", + "release_date": "2025-07-22", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "Qwen/Qwen3.5-27B": { + "attachment": true, + "cost": { + "cache_read": 0.08, + "input": 0.39, + "output": 3.12 + }, + "description": "Qwen3.5-27B is a dense model from the Qwen3.5 family built for high performance across a large range of benchmarks.", + "family": "qwen3.5", + "id": "Qwen/Qwen3.5-27B", + "last_updated": "2026-02-24", + "limit": { + "context": 262144, + "output": 262144 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Qwen3.5-27B", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-02-24", + "status": "deprecated", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "Qwen/Qwen3.5-35B-A3B": { + "attachment": true, + "cost": { + "cache_read": 0.25, + "input": 0.25, + "output": 1.25 + }, + "description": "Qwen3.5-35B-A3B is an open-weights multimodal MoE model built for efficient, high-throughput inference across chat, reasoning, and agentic tasks.", + "family": "qwen3.5", + "id": "Qwen/Qwen3.5-35B-A3B", + "last_updated": "2026-02-24", + "limit": { + "context": 262144, + "output": 262144 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Qwen3.5-35B-A3B", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-02-24", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "Qwen/Qwen3.6-27B": { + "attachment": true, + "cost": { + "cache_read": 0.12, + "input": 0.6, + "output": 3.6 + }, + "description": "Qwen3.6-27B is a 27B dense multimodal model with 262K context built for flagship-level agentic coding.", + "family": "qwen3.6", + "id": "Qwen/Qwen3.6-27B", + "last_updated": "2026-04-22", + "limit": { + "context": 262144, + "output": 262144 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Qwen3.6 27B", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-22", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "Qwen/Qwen3.6-35B-A3B": { + "attachment": true, + "cost": { + "cache_read": 0.25, + "input": 0.25, + "output": 1.25 + }, + "description": "Qwen3.6-35B-A3B is an MoE multimodal model with 262K context optimized for agentic coding workflows.", + "family": "qwen3.6", + "id": "Qwen/Qwen3.6-35B-A3B", + "last_updated": "2026-04-15", + "limit": { + "context": 262144, + "output": 262144 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Qwen3.6 35B A3B", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-15", "structured_output": true, "temperature": true, "tool_call": true @@ -186337,13 +186924,14 @@ "deepseek-ai/DeepSeek-V3.1": { "attachment": false, "cost": { + "cache_read": 0.55, "input": 0.55, "output": 1.65 }, - "description": "DeepSeek chat model for instruction following, coding, and analysis", + "description": "A large hybrid model that supports both thinking and non-thinking modes via prompt templates.", "family": "deepseek", "id": "deepseek-ai/DeepSeek-V3.1", - "last_updated": "2026-03-12", + "last_updated": "2025-08-21", "limit": { "context": 161000, "output": 161000 @@ -186364,16 +186952,159 @@ "temperature": true, "tool_call": true }, + "deepseek-ai/DeepSeek-V4-Flash": { + "attachment": false, + "cost": { + "cache_read": 0.07, + "input": 0.14, + "output": 0.28 + }, + "description": "DeepSeek V4-Flash is an MoE model with 1M context length great for coding, reasoning, and agentic workloads.", + "family": "deepseek", + "id": "deepseek-ai/DeepSeek-V4-Flash", + "knowledge": "2025-05", + "last_updated": "2026-04-24", + "limit": { + "context": 1048576, + "output": 1048576 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "DeepSeek V4 Flash", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-24", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "deepseek-ai/DeepSeek-V4-Pro": { + "attachment": false, + "cost": { + "cache_read": 0.14, + "input": 1.74, + "output": 3.48 + }, + "description": "DeepSeek V4-Pro is a 1.6T-parameter MoE model with 49B active parameters excelling at advanced reasoning, coding, and complex agentic workloads.", + "family": "deepseek", + "id": "deepseek-ai/DeepSeek-V4-Pro", + "knowledge": "2025-05", + "last_updated": "2026-04-24", + "limit": { + "context": 1048576, + "output": 1048576 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "DeepSeek V4 Pro", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-24", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "google/gemma-4-31B-it": { + "attachment": true, + "cost": { + "cache_read": 0.09, + "input": 0.12, + "output": 0.35 + }, + "description": "Gemma 4 31B Dense is designed for advanced reasoning, agentic workflows, and longer context and is natively trained on 140+ languages.", + "family": "gemma", + "id": "google/gemma-4-31B-it", + "last_updated": "2026-04-02", + "limit": { + "context": 262144, + "output": 262144 + }, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Gemma 4 31B", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-02", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "ibm-granite/granite-4.1-8b": { + "attachment": false, + "cost": { + "cache_read": 0.05, + "input": 0.05, + "output": 0.1 + }, + "description": "Granite 4.1 8B is a long-context instruct model capable of enhanced tool calling, instruction following, and chat capabilities.", + "family": "granite", + "id": "ibm-granite/granite-4.1-8b", + "last_updated": "2026-04-29", + "limit": { + "context": 131072, + "output": 131072 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "Granite 4.1 8B", + "open_weights": true, + "reasoning": false, + "release_date": "2026-04-29", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "meta-llama/Llama-3.1-70B-Instruct": { "attachment": false, "cost": { + "cache_read": 0.8, "input": 0.8, "output": 0.8 }, - "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", + "description": "Efficient conversational model optimized for responsive multilingual chatbot interactions.", "family": "llama", "id": "meta-llama/Llama-3.1-70B-Instruct", - "last_updated": "2026-03-12", + "last_updated": "2024-07-23", "limit": { "context": 128000, "output": 128000 @@ -186397,14 +187128,15 @@ "meta-llama/Llama-3.1-8B-Instruct": { "attachment": false, "cost": { + "cache_read": 0.22, "input": 0.22, "output": 0.22 }, - "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", + "description": "Efficient conversational model optimized for responsive multilingual chatbot interactions.", "family": "llama", "id": "meta-llama/Llama-3.1-8B-Instruct", "knowledge": "2023-12", - "last_updated": "2026-03-12", + "last_updated": "2024-07-23", "limit": { "context": 128000, "output": 128000 @@ -186417,7 +187149,7 @@ "text" ] }, - "name": "Meta-Llama-3.1-8B-Instruct", + "name": "Llama 3.1 8B", "open_weights": true, "reasoning": false, "release_date": "2024-07-23", @@ -186428,14 +187160,15 @@ "meta-llama/Llama-3.3-70B-Instruct": { "attachment": false, "cost": { + "cache_read": 0.71, "input": 0.71, "output": 0.71 }, - "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", + "description": "Multilingual model excelling in conversational tasks, detailed instruction-following, and coding.", "family": "llama", "id": "meta-llama/Llama-3.3-70B-Instruct", "knowledge": "2023-12", - "last_updated": "2026-03-12", + "last_updated": "2024-12-01", "limit": { "context": 128000, "output": 128000 @@ -186448,90 +187181,127 @@ "text" ] }, - "name": "Llama-3.3-70B-Instruct", + "name": "Llama 3.3 70B", "open_weights": true, "reasoning": false, - "release_date": "2024-12-06", + "release_date": "2024-12-01", "structured_output": true, "temperature": true, "tool_call": true }, - "meta-llama/Llama-4-Scout-17B-16E-Instruct": { + "microsoft/Phi-4-mini-instruct": { "attachment": false, "cost": { - "input": 0.17, - "output": 0.66 + "cache_read": 0.08, + "input": 0.08, + "output": 0.35 }, - "description": "Open multimodal Llama model for long-context analysis and efficient agents", - "family": "llama", - "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct", - "knowledge": "2024-12", - "last_updated": "2026-03-12", + "description": "Compact, efficient model ideal for fast responses in resource-constrained environments.", + "family": "phi", + "id": "microsoft/Phi-4-mini-instruct", + "knowledge": "2023-10", + "last_updated": "2025-02-01", "limit": { - "context": 64000, - "output": 64000 + "context": 128000, + "output": 128000 }, "modalities": { "input": [ - "text", - "image" + "text" ], "output": [ "text" ] }, - "name": "Llama 4 Scout 17B 16E Instruct", + "name": "Phi 4 Mini 3.8B", "open_weights": true, "reasoning": false, - "release_date": "2025-01-31", + "release_date": "2025-02-01", + "status": "deprecated", "structured_output": true, "temperature": true, "tool_call": true }, - "microsoft/Phi-4-mini-instruct": { - "attachment": false, + "moonshotai/Kimi-K2.5": { + "attachment": true, "cost": { - "input": 0.08, - "output": 0.35 + "cache_read": 0.1, + "input": 0.6, + "output": 3 }, - "description": "Efficient model for low-latency assistance, extraction, and routine automation", - "family": "phi", - "id": "microsoft/Phi-4-mini-instruct", - "knowledge": "2023-10", - "last_updated": "2026-03-12", + "description": "Kimi K2.5 is a multimodal Mixture-of-Experts language model featuring 32 billion activated parameters and a total of 1 trillion parameters.", + "family": "kimi-k2", + "id": "moonshotai/Kimi-K2.5", + "knowledge": "2025-01", + "last_updated": "2026-02-02", "limit": { - "context": 128000, - "output": 128000 + "context": 262144, + "output": 262144 }, "modalities": { "input": [ - "text" + "text", + "image" ], "output": [ "text" ] }, - "name": "Phi-4-mini-instruct", + "name": "Kimi K2.5", "open_weights": true, - "reasoning": false, - "release_date": "2024-12-11", + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-02-02", "structured_output": true, "temperature": true, "tool_call": true }, - "moonshotai/Kimi-K2.5": { + "moonshotai/Kimi-K2.6": { "attachment": true, "cost": { - "input": 0.5, - "output": 2.85 + "cache_read": 0.16, + "input": 0.95, + "output": 4 }, - "description": "Kimi multimodal agent model for visual understanding, coding, and planning", + "description": "Kimi K2.6 is a multimodal Mixture-of-Experts language model featuring 32 billion activated parameters and a total of 1 trillion parameters.", "family": "kimi-k2", - "id": "moonshotai/Kimi-K2.5", - "interleaved": { - "field": "reasoning_content" + "id": "moonshotai/Kimi-K2.6", + "knowledge": "2025-01", + "last_updated": "2026-04-20", + "limit": { + "context": 262144, + "output": 262144 }, - "last_updated": "2026-03-12", + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "name": "Kimi K2.6", + "open_weights": true, + "reasoning": true, + "reasoning_options": [], + "release_date": "2026-04-20", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "moonshotai/Kimi-K2.7-Code": { + "attachment": true, + "cost": { + "cache_read": 0.19, + "input": 0.94, + "output": 4 + }, + "description": "Kimi K2.7 Code is a 1T-parameter MoE model with 32B active parameters purpose-built for long-horizon agentic coding and software engineering.", + "family": "kimi-k2", + "id": "moonshotai/Kimi-K2.7-Code", + "knowledge": "2025-01", + "last_updated": "2026-06-12", "limit": { "context": 262144, "output": 262144 @@ -186545,11 +187315,11 @@ "text" ] }, - "name": "Kimi K2.5", + "name": "Kimi K2.7 Code", "open_weights": true, "reasoning": true, "reasoning_options": [], - "release_date": "2026-01-27", + "release_date": "2026-06-12", "structured_output": true, "temperature": true, "tool_call": true @@ -186557,13 +187327,14 @@ "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8": { "attachment": false, "cost": { + "cache_read": 0.2, "input": 0.2, "output": 0.8 }, - "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", + "description": "Nemotron 3 is a LatentMoE model designed to deliver strong agentic, reasoning, and conversational capabilities.", "family": "nemotron", "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", - "last_updated": "2026-03-12", + "last_updated": "2026-03-11", "limit": { "context": 262144, "output": 262144 @@ -186576,7 +187347,7 @@ "text" ] }, - "name": "NVIDIA Nemotron 3 Super 120B", + "name": "Nemotron 3 Super", "open_weights": true, "reasoning": true, "reasoning_options": [ @@ -186589,16 +187360,53 @@ "temperature": true, "tool_call": true }, + "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { + "attachment": false, + "cost": { + "cache_read": 0.15, + "input": 0.75, + "output": 2.75 + }, + "description": "Nemotron 3 Ultra is a powerful MoE model designed for long-running agents across coding, deep research, and enterprise automation.", + "family": "nemotron", + "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", + "last_updated": "2026-06-04", + "limit": { + "context": 262144, + "output": 262144 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "Nemotron 3 Ultra", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-06-04", + "structured_output": true, + "temperature": true, + "tool_call": true + }, "openai/gpt-oss-120b": { "attachment": false, "cost": { - "input": 0.15, - "output": 0.6 + "cache_read": 0.04, + "input": 0.04, + "output": 0.14 }, - "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", + "description": "Efficient Mixture-of-Experts model designed for high-reasoning, agentic and general-purpose use cases.", "family": "gpt-oss", "id": "openai/gpt-oss-120b", - "last_updated": "2026-03-12", + "last_updated": "2025-08-05", "limit": { "context": 131072, "output": 131072 @@ -186612,7 +187420,7 @@ ] }, "name": "gpt-oss-120b", - "open_weights": false, + "open_weights": true, "reasoning": true, "reasoning_options": [], "release_date": "2025-08-05", @@ -186623,13 +187431,14 @@ "openai/gpt-oss-20b": { "attachment": false, "cost": { - "input": 0.05, - "output": 0.2 + "cache_read": 0.03, + "input": 0.03, + "output": 0.13 }, - "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", + "description": "Lower latency Mixture-of-Experts model trained on OpenAI's Harmony response format with reasoning capabilities.", "family": "gpt-oss", "id": "openai/gpt-oss-20b", - "last_updated": "2026-03-12", + "last_updated": "2025-08-05", "limit": { "context": 131072, "output": 131072 @@ -186643,7 +187452,7 @@ ] }, "name": "gpt-oss-20b", - "open_weights": false, + "open_weights": true, "reasoning": true, "reasoning_options": [], "release_date": "2025-08-05", @@ -186651,19 +187460,20 @@ "temperature": true, "tool_call": true }, - "zai-org/GLM-5-FP8": { + "zai-org/GLM-5.1": { "attachment": false, "cost": { - "input": 1, - "output": 3.2 + "cache_read": 0.26, + "input": 1.4, + "output": 4.4 }, - "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", + "description": "Powerful MoE model for long-horizon agentic engineering and advanced reasoning.", "family": "glm", - "id": "zai-org/GLM-5-FP8", - "last_updated": "2026-03-12", + "id": "zai-org/GLM-5.1", + "last_updated": "2026-04-07", "limit": { - "context": 200000, - "output": 200000 + "context": 202752, + "output": 202752 }, "modalities": { "input": [ @@ -186673,32 +187483,33 @@ "text" ] }, - "name": "GLM 5", + "name": "GLM 5.1", "open_weights": true, - "reasoning": false, - "release_date": "2026-02-11", + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-07", "structured_output": true, "temperature": true, "tool_call": true }, - "zai-org/GLM-5.1": { + "zai-org/GLM-5.2": { "attachment": false, "cost": { "cache_read": 0.26, - "cache_write": 0, - "input": 1.4, + "input": 1.39, "output": 4.4 }, - "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", + "description": "GLM-5.2 is a Mixture-of-Experts language model featuring 40 billion activated parameters and a total of 744 billion parameters.", "family": "glm", - "id": "zai-org/GLM-5.1", - "interleaved": { - "field": "reasoning_content" - }, - "last_updated": "2026-04-07", + "id": "zai-org/GLM-5.2", + "last_updated": "2026-06-16", "limit": { - "context": 200000, - "output": 131072 + "context": 262144, + "output": 262144 }, "modalities": { "input": [ @@ -186708,7 +187519,7 @@ "text" ] }, - "name": "GLM-5.1", + "name": "GLM 5.2", "open_weights": true, "reasoning": true, "reasoning_options": [ @@ -186716,7 +187527,7 @@ "type": "toggle" } ], - "release_date": "2026-04-07", + "release_date": "2026-06-16", "structured_output": true, "temperature": true, "tool_call": true @@ -186854,10 +187665,70 @@ } ] }, - "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", + "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", + "family": "grok", + "id": "grok-4.20-multi-agent-0309", + "last_updated": "2026-03-09", + "limit": { + "context": 1000000, + "output": 30000 + }, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "Grok 4.20 Multi-Agent", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "low", + "medium", + "high", + "xhigh" + ] + } + ], + "release_date": "2026-03-09", + "structured_output": true, + "temperature": true, + "tool_call": false + }, + "grok-4.3": { + "attachment": true, + "cost": { + "cache_read": 0.2, + "context_over_200k": { + "cache_read": 0.4, + "input": 2.5, + "output": 5 + }, + "input": 1.25, + "output": 2.5, + "tiers": [ + { + "cache_read": 0.4, + "input": 2.5, + "output": 5, + "tier": { + "size": 200000, + "type": "context" + } + } + ] + }, + "description": "xAI's Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", - "id": "grok-4.20-multi-agent-0309", - "last_updated": "2026-03-09", + "id": "grok-4.3", + "last_updated": "2026-04-17", "limit": { "context": 1000000, "output": 30000 @@ -186872,41 +187743,41 @@ "text" ] }, - "name": "Grok 4.20 Multi-Agent", + "name": "Grok 4.3", "open_weights": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": [ + "none", "low", "medium", - "high", - "xhigh" + "high" ] } ], - "release_date": "2026-03-09", + "release_date": "2026-04-17", "structured_output": true, "temperature": true, - "tool_call": false + "tool_call": true }, - "grok-4.3": { + "grok-4.5": { "attachment": true, "cost": { - "cache_read": 0.2, + "cache_read": 0.5, "context_over_200k": { - "cache_read": 0.4, - "input": 2.5, - "output": 5 + "cache_read": 1, + "input": 4, + "output": 12 }, - "input": 1.25, - "output": 2.5, + "input": 2, + "output": 6, "tiers": [ { - "cache_read": 0.4, - "input": 2.5, - "output": 5, + "cache_read": 1, + "input": 4, + "output": 12, "tier": { "size": 200000, "type": "context" @@ -186914,13 +187785,13 @@ } ] }, - "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", + "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", - "id": "grok-4.3", - "last_updated": "2026-04-17", + "id": "grok-4.5", + "last_updated": "2026-07-08", "limit": { - "context": 1000000, - "output": 30000 + "context": 500000, + "output": 500000 }, "modalities": { "input": [ @@ -186932,21 +187803,20 @@ "text" ] }, - "name": "Grok 4.3", + "name": "Grok 4.5", "open_weights": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": [ - "none", "low", "medium", "high" ] } ], - "release_date": "2026-04-17", + "release_date": "2026-07-08", "structured_output": true, "temperature": true, "tool_call": true @@ -188745,10 +189615,265 @@ "glm-5v-turbo": { "attachment": true, "cost": { - "cache_read": 0.24, + "cache_read": 0.24, + "cache_write": 0, + "input": 1.2, + "output": 4 + }, + "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", + "family": "glm", + "id": "glm-5v-turbo", + "interleaved": { + "field": "reasoning_content" + }, + "last_updated": "2026-04-01", + "limit": { + "context": 200000, + "output": 131072 + }, + "modalities": { + "input": [ + "text", + "image", + "video", + "pdf" + ], + "output": [ + "text" + ] + }, + "name": "GLM-5V-Turbo", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-04-01", + "temperature": true, + "tool_call": true + } + }, + "name": "Z.AI", + "npm": "@ai-sdk/openai-compatible" + }, + "zai-coding-plan": { + "api": "https://api.z.ai/api/coding/paas/v4", + "doc": "https://docs.z.ai/devpack/overview", + "env": [ + "ZHIPU_API_KEY" + ], + "id": "zai-coding-plan", + "models": { + "glm-4.5-air": { + "attachment": false, + "cost": { + "cache_read": 0, + "cache_write": 0, + "input": 0, + "output": 0 + }, + "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", + "family": "glm-air", + "id": "glm-4.5-air", + "knowledge": "2025-04", + "last_updated": "2025-07-28", + "limit": { + "context": 131072, + "output": 98304 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "GLM-4.5-Air", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2025-07-28", + "temperature": true, + "tool_call": true + }, + "glm-4.7": { + "attachment": false, + "cost": { + "cache_read": 0, + "cache_write": 0, + "input": 0, + "output": 0 + }, + "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", + "family": "glm", + "id": "glm-4.7", + "interleaved": { + "field": "reasoning_content" + }, + "knowledge": "2025-04", + "last_updated": "2025-12-22", + "limit": { + "context": 204800, + "output": 131072 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "GLM-4.7", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2025-12-22", + "temperature": true, + "tool_call": true + }, + "glm-5-turbo": { + "attachment": false, + "cost": { + "cache_read": 0, + "cache_write": 0, + "input": 0, + "output": 0 + }, + "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", + "family": "glm", + "id": "glm-5-turbo", + "interleaved": { + "field": "reasoning_content" + }, + "last_updated": "2026-03-16", + "limit": { + "context": 200000, + "output": 131072 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "GLM-5-Turbo", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-03-16", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "glm-5.1": { + "attachment": false, + "cost": { + "cache_read": 0, + "cache_write": 0, + "input": 0, + "output": 0 + }, + "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", + "family": "glm", + "id": "glm-5.1", + "interleaved": { + "field": "reasoning_content" + }, + "last_updated": "2026-03-27", + "limit": { + "context": 200000, + "output": 131072 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "GLM-5.1", + "open_weights": false, + "reasoning": true, + "reasoning_options": [ + { + "type": "toggle" + } + ], + "release_date": "2026-03-27", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "glm-5.2": { + "attachment": false, + "cost": { + "cache_read": 0, + "cache_write": 0, + "input": 0, + "output": 0 + }, + "description": "Open flagship GLM for long-horizon coding agents and million-token context work", + "family": "glm", + "id": "glm-5.2", + "interleaved": { + "field": "reasoning_content" + }, + "last_updated": "2026-06-13", + "limit": { + "context": 1000000, + "output": 131072 + }, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "name": "GLM-5.2", + "open_weights": true, + "reasoning": true, + "reasoning_options": [ + { + "type": "effort", + "values": [ + "high", + "max" + ] + } + ], + "release_date": "2026-06-13", + "structured_output": true, + "temperature": true, + "tool_call": true + }, + "glm-5v-turbo": { + "attachment": true, + "cost": { + "cache_read": 0, "cache_write": 0, - "input": 1.2, - "output": 4 + "input": 0, + "output": 0 }, "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", @@ -188785,189 +189910,30 @@ "tool_call": true } }, - "name": "Z.AI", + "name": "Z.AI Coding Plan", "npm": "@ai-sdk/openai-compatible" }, - "zai-coding-plan": { - "api": "https://api.z.ai/api/coding/paas/v4", - "doc": "https://docs.z.ai/devpack/overview", + "zeldoc": { + "api": "https://api.zeldoc.ai/v1", + "doc": "https://docs.zeldoc.ai", "env": [ - "ZHIPU_API_KEY" + "ZELDOC_API_KEY" ], - "id": "zai-coding-plan", + "id": "zeldoc", "models": { - "glm-4.5-air": { - "attachment": false, - "cost": { - "cache_read": 0, - "cache_write": 0, - "input": 0, - "output": 0 - }, - "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", - "family": "glm-air", - "id": "glm-4.5-air", - "knowledge": "2025-04", - "last_updated": "2025-07-28", - "limit": { - "context": 131072, - "output": 98304 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "GLM-4.5-Air", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-07-28", - "temperature": true, - "tool_call": true - }, - "glm-4.7": { - "attachment": false, - "cost": { - "cache_read": 0, - "cache_write": 0, - "input": 0, - "output": 0 - }, - "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", - "family": "glm", - "id": "glm-4.7", - "interleaved": { - "field": "reasoning_content" - }, - "knowledge": "2025-04", - "last_updated": "2025-12-22", - "limit": { - "context": 204800, - "output": 131072 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "GLM-4.7", - "open_weights": true, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2025-12-22", - "temperature": true, - "tool_call": true - }, - "glm-5-turbo": { - "attachment": false, - "cost": { - "cache_read": 0, - "cache_write": 0, - "input": 0, - "output": 0 - }, - "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", - "family": "glm", - "id": "glm-5-turbo", - "interleaved": { - "field": "reasoning_content" - }, - "last_updated": "2026-03-16", - "limit": { - "context": 200000, - "output": 131072 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "GLM-5-Turbo", - "open_weights": false, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2026-03-16", - "structured_output": true, - "temperature": true, - "tool_call": true - }, - "glm-5.1": { - "attachment": false, - "cost": { - "cache_read": 0, - "cache_write": 0, - "input": 0, - "output": 0 - }, - "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", - "family": "glm", - "id": "glm-5.1", - "interleaved": { - "field": "reasoning_content" - }, - "last_updated": "2026-03-27", - "limit": { - "context": 200000, - "output": 131072 - }, - "modalities": { - "input": [ - "text" - ], - "output": [ - "text" - ] - }, - "name": "GLM-5.1", - "open_weights": false, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2026-03-27", - "structured_output": true, - "temperature": true, - "tool_call": true - }, - "glm-5.2": { + "z-code": { "attachment": false, "cost": { - "cache_read": 0, - "cache_write": 0, "input": 0, "output": 0 }, - "description": "Open flagship GLM for long-horizon coding agents and million-token context work", - "family": "glm", - "id": "glm-5.2", + "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", + "id": "z-code", "interleaved": { "field": "reasoning_content" }, - "last_updated": "2026-06-13", + "knowledge": "2025-01", + "last_updated": "2026-04-15", "limit": { "context": 1000000, "output": 131072 @@ -188980,8 +189946,8 @@ "text" ] }, - "name": "GLM-5.2", - "open_weights": true, + "name": "Z-Code", + "open_weights": false, "reasoning": true, "reasoning_options": [ { @@ -188992,109 +189958,62 @@ ] } ], - "release_date": "2026-06-13", + "release_date": "2026-04-15", "structured_output": true, "temperature": true, "tool_call": true - }, - "glm-5v-turbo": { - "attachment": true, - "cost": { - "cache_read": 0, - "cache_write": 0, - "input": 0, - "output": 0 - }, - "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", - "family": "glm", - "id": "glm-5v-turbo", - "interleaved": { - "field": "reasoning_content" - }, - "last_updated": "2026-04-01", - "limit": { - "context": 200000, - "output": 131072 - }, - "modalities": { - "input": [ - "text", - "image", - "video", - "pdf" - ], - "output": [ - "text" - ] - }, - "name": "GLM-5V-Turbo", - "open_weights": false, - "reasoning": true, - "reasoning_options": [ - { - "type": "toggle" - } - ], - "release_date": "2026-04-01", - "temperature": true, - "tool_call": true } }, - "name": "Z.AI Coding Plan", + "name": "Zeldoc", "npm": "@ai-sdk/openai-compatible" }, - "zeldoc": { - "api": "https://api.zeldoc.ai/v1", - "doc": "https://docs.zeldoc.ai", + "zenifra": { + "api": "https://ai.zenifra.com/v1", + "doc": "https://docs.zenifra.com", "env": [ - "ZELDOC_API_KEY" + "ZENIFRA_AI_KEY" ], - "id": "zeldoc", + "id": "zenifra", "models": { - "z-code": { - "attachment": false, + "qwen3.6-35b-a3b": { + "attachment": true, "cost": { - "input": 0, - "output": 0 - }, - "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", - "id": "z-code", - "interleaved": { - "field": "reasoning_content" + "input": 0.19, + "output": 0.48 }, - "knowledge": "2025-01", - "last_updated": "2026-04-15", + "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", + "family": "qwen", + "id": "qwen3.6-35b-a3b", + "last_updated": "2026-04-17", "limit": { - "context": 1000000, - "output": 131072 + "context": 262144, + "output": 65536 }, "modalities": { "input": [ - "text" + "text", + "image", + "video", + "audio" ], "output": [ "text" ] }, - "name": "Z-Code", - "open_weights": false, + "name": "Qwen3.6 35B-A3B", + "open_weights": true, + "provider": { + "shape": "completions" + }, "reasoning": true, - "reasoning_options": [ - { - "type": "effort", - "values": [ - "high", - "max" - ] - } - ], - "release_date": "2026-04-15", + "reasoning_options": [], + "release_date": "2026-04-17", "structured_output": true, "temperature": true, "tool_call": true } }, - "name": "Zeldoc", + "name": "Zenifra", "npm": "@ai-sdk/openai-compatible" }, "zenmux": { @@ -189517,6 +190436,7 @@ "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "id": "anthropic/claude-opus-4.8", + "knowledge": "2026-01", "last_updated": "2026-05-28", "limit": { "context": 1000000,