vtcode_config/models/model_id.rs
1use serde::{Deserialize, Serialize};
2
3mod as_str;
4mod capabilities;
5mod collection;
6mod defaults;
7mod description;
8mod display;
9mod format;
10mod openrouter;
11mod parse;
12mod provider;
13mod table;
14
15pub use capabilities::{
16 ModelCatalogEntry, ModelPricing, catalog_provider_keys, model_catalog_entry, supported_models_for_provider,
17};
18
19/// Centralized enum for all supported model identifiers
20#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
21#[derive(Clone, Debug, Default, PartialEq, Eq, Hash, Serialize, Deserialize)]
22pub enum ModelId {
23 // Gemini models
24 /// Gemini 3.6 Flash - Latest flash model with improved capabilities
25 /// Gemini 3.7 Flash - Flash model with 1M context and tunable thinking levels
26 /// Gemini 3.8 Flash - Most intelligent Flash for long-horizon SWE, agents, and enterprise workflows (1M context, 64k output, low/medium/high thinking)
27 Gemini38Flash,
28
29 // OpenAI models
30 /// GPT-6 Astra - Most capable model for hardest end-to-end work with complex reasoning, coding, computer use, and research
31 GPT6Astra,
32 /// GPT-6 Sol - Cost-efficient high-end model in the GPT-6 series for demanding professional work
33 GPT6Sol,
34 /// GPT-6.1 Sol - Near-Astra performance for complex coding, computer use, and professional work at a lower cost
35 GPT61Sol,
36 /// GPT-6 Luna - Fast cost-efficient model in the GPT-6 series for high-volume latency-sensitive workloads
37 GPT6Luna,
38 /// GPT-5.6 Sol - Frontier model for complex professional work in the GPT-5.6 family
39 GPT56Sol,
40 /// GPT-5.6 Terra - GPT-5.6 model that balances intelligence and cost
41 GPT56Terra,
42 /// GPT-5.6 Luna - GPT-5.6 model optimized for cost-sensitive workloads
43 GPT56Luna,
44 /// GPT-OSS 20B - OpenAI's open-source 20B parameter model using harmony
45 OpenAIGptOss20b,
46 /// GPT-OSS 120B - OpenAI's open-source 120B parameter model using harmony
47 OpenAIGptOss120b,
48
49 // Anthropic models
50 /// Claude Sonnet 5 - The best combination of speed and intelligence with adaptive thinking on by default
51 #[default]
52 ClaudeSonnet5,
53 /// Claude Sonnet 5.5 - Latest Sonnet: best speed/intelligence balance, 1M context, 128K output, `between_tools` as the lowest thinking setting, default effort high
54 ClaudeSonnet55,
55 /// Claude Haiku 5.5 - High-volume latency-sensitive work: classification, routing, extraction, subagents; adaptive thinking, 1M context, 128K output, default effort medium
56 ClaudeHaiku55,
57 /// Claude Fable 5 - Anthropic's most capable widely released model for demanding reasoning and long-horizon agentic work
58 ClaudeFable5,
59 /// Claude Fable 5.1 - successor to Fable 5 for demanding reasoning and long-horizon agentic work, 1M context, adaptive thinking always on, cache reads at 1/4 cost
60 ClaudeFable51,
61 /// Claude Opus 5 - Anthropic's newest Opus-tier model with 1M context, thinking on by default
62 ClaudeOpus5,
63 /// Claude Opus 5.5 - Opus-tier successor for long-running agentic coding, adaptive thinking always on, 1M context, 128K output, default effort medium
64 ClaudeOpus55,
65 /// GitHub Copilot auto model selection
66 CopilotAuto,
67 /// GitHub Copilot GPT-5.2 Codex
68 CopilotGPT52Codex,
69 /// GitHub Copilot GPT-5.1 Codex Max
70 CopilotGPT51CodexMax,
71 /// GitHub Copilot GPT-5.4
72 CopilotGPT54,
73 /// GitHub Copilot GPT-5.4 Mini
74 CopilotGPT54Mini,
75
76 // DeepSeek models
77 /// DeepSeek V4.1 Flash - Latest flash model with improved reasoning and efficiency
78 DeepSeekFlash,
79
80 // Official Meta AI models
81 /// Meta Muse Spark 1.1 - Official Meta AI Standard-tier reasoning model
82 MetaMuseSpark11,
83 /// Meta Muse Spark 1.3 - Official Meta AI flagship Standard-tier reasoning model, tuned for agentic workflows
84 MetaMuseSpark13,
85 /// Meta Muse Spark 1.3 Contributor tier - opt-in variant with Meta's discounted data-contribution terms
86 MetaMuseSpark13Contributor,
87
88 // NVIDIA NIM models
89 /// NVIDIA Nemotron 3 Ultra - NVIDIA's flagship agentic reasoning model via NIM
90 NvidiaNemotron3Ultra550bA55b,
91 /// NVIDIA Nemotron 3 Super - Efficient long-context agentic reasoning model via NIM
92 NvidiaNemotron3Super120bA12b,
93 /// NVIDIA Nemotron 3 Nano - Efficient reasoning and tool-use model via NIM
94 NvidiaNemotron3Nano30bA3b,
95
96 // Merge Gateway routes
97 /// Merge Gateway's default route selected by Merge
98 MergeGatewayDefaultRouting,
99 /// Anthropic Claude Opus 5 through Merge Gateway
100 MergeGatewayAnthropicClaudeOpus5,
101 /// Anthropic Claude Opus 5.5 through Merge Gateway
102 MergeGatewayAnthropicClaudeOpus55,
103 /// Anthropic Claude Sonnet 5 through Merge Gateway
104 MergeGatewayAnthropicClaudeSonnet5,
105 /// Anthropic Claude Sonnet 5.5 through Merge Gateway
106 MergeGatewayAnthropicClaudeSonnet55,
107 /// Google Gemini 3.6 Flash through Merge Gateway
108 /// Google Gemini 3.7 Flash through Merge Gateway
109 /// DeepSeek V4.1 Flash through Merge Gateway
110 MergeGatewayDeepseekFlash,
111 /// xAI Grok 4.6 through Merge Gateway
112 MergeGatewayXaiGrok46,
113 /// xAI Grok 4.7 through Merge Gateway
114 MergeGatewayXaiGrok47,
115 /// MiniMax H3 through Merge Gateway
116 MergeGatewayMinimaxH3,
117 /// Moonshot Kimi K3 through Merge Gateway
118 MergeGatewayMoonshotKimiK3,
119 /// Thinking Machines Inkling through Merge Gateway
120 MergeGatewayThinkingMachinesInkling,
121 /// Z.AI GLM-5.3 Flash through Merge Gateway
122 MergeGatewayZaiGlm53Flash,
123 /// Z.AI GLM-5.3 FlashX through Merge Gateway
124 MergeGatewayZaiGlm53Flashx,
125 /// OpenAI GPT-5.6 Luna through Merge Gateway
126 MergeGatewayOpenAIGpt56Luna,
127 /// OpenAI GPT-5.6 Sol through Merge Gateway
128 MergeGatewayOpenAIGpt56Sol,
129 /// OpenAI GPT-5.6 Terra through Merge Gateway
130 MergeGatewayOpenAIGpt56Terra,
131 /// Google Gemini 3.8 Flash through Merge Gateway
132 MergeGatewayGoogleGemini38Flash,
133 /// Anthropic Claude Haiku 4.5 through Merge Gateway
134 MergeGatewayAnthropicClaudeHaiku4520251001,
135 /// Anthropic Claude Haiku 5.5 through Merge Gateway
136 MergeGatewayAnthropicClaudeHaiku55,
137 /// Anthropic Claude Fable 5.1 through Merge Gateway
138 MergeGatewayAnthropicClaudeFable51,
139 /// OpenAI GPT-6 Astra through Merge Gateway
140 MergeGatewayOpenAIGpt6Astra,
141 /// OpenAI GPT-6 Sol through Merge Gateway
142 MergeGatewayOpenAIGpt6Sol,
143 /// OpenAI GPT-6.1 Sol through Merge Gateway
144 MergeGatewayOpenAIGpt61Sol,
145 /// OpenAI GPT-6 Luna through Merge Gateway
146 MergeGatewayOpenAIGpt6Luna,
147 /// Xiaomi MiMo V2.6 Pro through Merge Gateway
148 MergeGatewayXiaomimimoMimoV26Pro,
149 /// Xiaomi MiMo V2.6 Flash through Merge Gateway
150 MergeGatewayXiaomimimoMimoV26Flash,
151 /// Mistral Large 4 through Merge Gateway
152 MergeGatewayMistralLarge4,
153
154 // Mistral AI models
155 /// Mistral Large 3 - State-of-the-art open-weight general-purpose multimodal model
156 MistralLarge3,
157 /// Mistral Large 4 - Open-weight MoE flagship (49B active / 1.05T total) with 1M context (Public Preview)
158 MistralLarge4,
159 // Hugging Face models
160 /// OpenAI GPT-OSS 20B via Hugging Face router
161 HuggingFaceOpenAIGptOss20b,
162 /// OpenAI GPT-OSS 120B via Hugging Face router
163 HuggingFaceOpenAIGptOss120b,
164 /// Z.AI GLM-5.2 via Novita inference provider on Hugging Face router
165 /// Z.AI GLM-5.3 Flash via Together inference provider on Hugging Face router
166 HuggingFaceGlm53FlashTogether,
167 /// Z.AI GLM-5.3 via Together inference provider on Hugging Face router
168 HuggingFaceGlm53Together,
169 /// Kimi K3 via Together on Hugging Face router
170 HuggingFaceKimiK3Together,
171 /// MiniMax M3 via Novita on Hugging Face router
172 HuggingFaceMinimaxM3Novita,
173
174 // StepFun models
175 /// Step 3.7 Flash - StepFun's flagship multimodal reasoning model with tool calling
176 StepFun37Flash,
177 /// Step 5 Preview - StepFun's frontier model for production-scale Agent applications with 1M context
178 StepFun5Preview,
179
180 // Evolink gateway models (namespaced as `evolink/<model>`)
181 /// GPT-5.2 served through the Evolink gateway
182 /// GPT-5.5 served through the Evolink gateway
183 /// Gemini 3.1 Pro served through the Evolink gateway (OpenAI SDK format)
184 EvolinkGemini31Pro,
185 /// Gemini 3.5 Flash served through the Evolink gateway (OpenAI SDK format)
186 /// MiniMax-M3 served through the Evolink gateway (OpenAI Chat Completions format)
187 EvolinkMinimaxM3,
188 /// Claude Haiku 4.5 served through the Evolink gateway (Anthropic Messages API)
189 EvolinkClaudeHaiku45,
190
191 /// GLM-5.3 - Z.ai flagship coding model with frontier long-horizon agentic performance
192 ZaiGlm53,
193 /// GLM-5.3 Flash - Z.ai efficient multimodal model with hybrid sparse+linear attention, 320B total / 18B active, 1M context, native vision
194 ZaiGlm53Flash,
195 /// GLM-5.3 FlashX - Z.ai high-speed Flash variant with faster inference (up to 200 tok/s), 320B total / 18B active, 1M context, native vision
196 ZaiGlm53Flashx,
197
198 // MiMo models
199 /// MiMo V2.6 Pro - Xiaomi's flagship reasoning model with 1M context
200 MiMoV26Pro,
201 /// MiMo V2.6 Flash - Xiaomi's efficient high-volume model with 1M context
202 MiMoV26Flash,
203 /// MiMo V2.6 Pro UltraSpeed - Xiaomi's fastest flagship variant with 1M context
204 MiMoV26ProUltraspeed,
205
206 // Moonshot models
207 /// Kimi K3 - Moonshot.ai's 2.8T parameter flagship with Delta Attention, native vision, 1M context
208 MoonshotKimiK3,
209
210 // OpenCode Zen models
211
212 // OpenCode Go models (20 models - https://opencode.ai/docs/go)
213 /// GLM-5.3 - Z.AI flagship for frontier long-horizon coding on OpenCode Go
214 OpenCodeGoGlm53,
215 /// GLM-5.2 - Z.AI flagship model included with OpenCode Go
216 /// GPT-5.6 Luna - OpenAI cost-efficient frontier model on OpenCode Go
217 OpenCodeGoGpt56Luna,
218 /// Kimi K3 - Moonshot flagship 2.8T agentic model on OpenCode Go
219 OpenCodeGoKimiK3,
220 /// MiniMax M3 - Frontier multimodal coding model on OpenCode Go
221 OpenCodeGoMinimaxM3,
222 /// Muse Spark 1.2 Contributor - Meta long-context reasoning on OpenCode Go (limited regions)
223
224 // Qwen models (non-Qwen3 only)
225
226 // Ollama models
227 /// GPT-OSS 20B - Open-weight GPT-OSS 20B model served via Ollama locally
228 OllamaGptOss20b,
229 /// GPT-OSS 20B Cloud - Cloud-hosted GPT-OSS 20B served via Ollama Cloud
230 OllamaGptOss20bCloud,
231 /// GPT-OSS 120B Cloud - Cloud-hosted GPT-OSS 120B served via Ollama Cloud
232 OllamaGptOss120bCloud,
233 /// MiniMax-M3 Cloud - Cloud-hosted MiniMax-M3 model served via Ollama Cloud
234 OllamaMinimaxM3Cloud,
235 /// GLM-5.2 Cloud - Cloud-hosted GLM-5.2 flagship model served via Ollama Cloud
236 /// GLM-5.3 Cloud - Cloud-hosted GLM-5.3 flagship model served via Ollama Cloud
237 OllamaGlm53Cloud,
238 /// Kimi K3 Cloud - Moonshot Kimi K3 via Ollama Cloud
239 OllamaKimiK3Cloud,
240 /// Gemma 4 - Google Gemma 4 model served via Ollama
241 OllamaGemma4,
242 /// Laguna XS.2 - Poolside's 33B MoE model (3B activated) for agentic coding via Ollama
243
244 // llama.cpp models
245 /// Gemma 4 26B A4B - Desktop Gemma 4 MoE model served through llama.cpp
246 LlamaCppGemma426bA4b,
247 /// Gemma 4 E4B - Tiny-footprint Gemma 4 model served through llama.cpp
248 LlamaCppGemma4E4b,
249 /// GPT-OSS 20B - OpenAI open-weight model served through llama.cpp
250 LlamaCppGptOss20b,
251
252 // MiniMax models
253 /// MiniMax-M3 - Frontier multimodal coding model with 1M context
254 MinimaxM3,
255
256 // OpenRouter models
257 /// DeepSeek V4.1 Flash - Latest flash model via OpenRouter
258 OpenRouterDeepSeekFlash,
259 /// OpenAI gpt-oss-120b - Open-weight 120B reasoning model via OpenRouter
260 OpenRouterOpenAIGptOss120b,
261 /// OpenAI gpt-oss-120b:free - Open-weight 120B reasoning model free tier via OpenRouter
262 OpenRouterOpenAIGptOss120bFree,
263 /// OpenAI gpt-oss-20b - Open-weight 20B deployment via OpenRouter
264 OpenRouterOpenAIGptOss20b,
265 /// OpenAI GPT-6 Astra - OpenAI's flagship model for demanding end-to-end work via OpenRouter
266 OpenRouterOpenAIGpt6Astra,
267 /// GPT-6 Sol - Cost-efficient high-end model in the GPT-6 series via OpenRouter
268 OpenRouterOpenAIGpt6Sol,
269 /// GPT-6 Luna - Fast cost-efficient model in the GPT-6 series via OpenRouter
270 OpenRouterOpenAIGpt6Luna,
271
272 /// Meta Muse Glimmer 30B via OpenRouter
273 OpenRouterMetaMuseGlimmer30b,
274 /// Meta Muse Spark 1.2 via OpenRouter
275 /// Meta Muse Spark 1.3 via OpenRouter
276 OpenRouterMetaMuseSpark13,
277 /// Gemini 3.7 Flash - Flash model with 1M context and tunable thinking levels via OpenRouter
278 /// Gemini 3.8 Flash - Most intelligent Flash for long-horizon SWE/agents with 1M context via OpenRouter
279 OpenRouterGoogleGemini38Flash,
280
281 /// Claude Sonnet 5 - Anthropic Claude Sonnet 5 listing
282 OpenRouterAnthropicClaudeSonnet5,
283 /// Mistral Large 3 2512 - Mistral Large 3 2512 model via OpenRouter
284 OpenRouterMistralaiMistralLarge2512,
285 /// DeepSeek V3.1 Nex N1 - Nex AGI DeepSeek V3.1 Nex N1 model via OpenRouter
286 OpenRouterNexAgiDeepseekV31NexN1,
287 /// GLM-5.2 - Z.AI GLM-5.2 flagship model for long-horizon tasks via OpenRouter
288 /// GLM-5.3 Flash - Z.AI efficient multimodal model with hybrid sparse+linear attention via OpenRouter
289 OpenRouterZaiGlm53Flash,
290 /// GLM-5.3 FlashX - Z.AI high-speed Flash variant with faster inference via OpenRouter
291 OpenRouterZaiGlm53Flashx,
292 /// Kimi K3 - Moonshot AI's 2.8T parameter flagship via OpenRouter
293 OpenRouterMoonshotaiKimiK3,
294 /// Grok 4.6 - xAI's flagship reasoning model with reasoning_effort support via OpenRouter
295 OpenRouterXAiGrok46,
296 /// Grok 4.7 - xAI's flagship reasoning model with reasoning_effort support via OpenRouter
297 OpenRouterXAiGrok47,
298 /// MiMo-V2.6-Pro - Xiaomi's flagship agentic model for complex software engineering via OpenRouter
299 OpenRouterXiaomiMimoV26Pro,
300 /// MiMo-V2.6-Flash - Xiaomi's efficient high-volume agentic model via OpenRouter
301 OpenRouterXiaomiMimoV26Flash,
302 /// MiMo-V2.6-Pro-UltraSpeed - Xiaomi's fastest flagship variant via OpenRouter
303 OpenRouterXiaomiMimoV26ProUltraspeed,
304 /// Step 5 Preview - StepFun's flagship agentic model via OpenRouter
305 OpenRouterStepfunStep5Preview,
306
307 // Vercel AI Gateway models (namespaced as `vendor/model` on the gateway)
308 /// Claude Sonnet 5 served through the Vercel AI Gateway
309 VercelAnthropicClaudeSonnet5,
310 /// Claude Opus 5 served through the Vercel AI Gateway
311 VercelAnthropicClaudeOpus5,
312 /// Claude Opus 5.5 served through the Vercel AI Gateway
313 VercelAnthropicClaudeOpus55,
314 /// Claude Haiku 4.5 served through the Vercel AI Gateway
315 VercelAnthropicClaudeHaiku45,
316 /// GPT-5.6 Sol served through the Vercel AI Gateway
317 VercelOpenAiGpt56Sol,
318 /// GPT-6 Astra served through the Vercel AI Gateway
319 VercelOpenAiGpt6Astra,
320 /// GPT-5.6 Luna served through the Vercel AI Gateway
321 VercelOpenAiGpt56Luna,
322 /// GPT-5.3 Codex served through the Vercel AI Gateway
323 /// Gemini 3.1 Pro Preview served through the Vercel AI Gateway
324 /// Gemini 3.8 Flash served through the Vercel AI Gateway
325 VercelGoogleGemini38Flash,
326 /// DeepSeek V4.1 Flash served through the Vercel AI Gateway
327 VercelDeepseekFlash,
328 /// Kimi K3 served through the Vercel AI Gateway
329 VercelMoonshotaiKimiK3,
330 /// MiniMax M3 served through the Vercel AI Gateway
331 VercelMinimaxM3,
332 /// Grok 4.7 served through the Vercel AI Gateway
333 VercelSpacexaiGrok47,
334 // xAI models
335 /// Grok 4.6 - xAI's flagship reasoning model with reasoning_effort support (500k context)
336 XaiGrok46,
337 /// Grok 4.7 - xAI's flagship reasoning model with reasoning_effort support (500k context)
338 XaiGrok47,
339
340 /// User-defined model not in the hardcoded catalog.
341 /// Carries the provider key string and model identifier string.
342 Custom(String, String),
343}