vtcode_llm/provider/request.rs
1use serde::{Deserialize, Serialize};
2use serde_json::{Value, json};
3use std::sync::Arc;
4use vtcode_config::types::{ReasoningEffortLevel, VerbosityLevel};
5
6use super::{Message, ToolDefinition};
7
8/// Fallback model configuration for Anthropic server-side fallback.
9/// The explicit-list form of `fallbacks`, sent with the
10/// `server-side-fallback-2026-06-01` beta header (the `"default"` keyword form
11/// uses `server-side-fallback-2026-07-01`; pairing a header with the other form
12/// is rejected).
13#[derive(Debug, Clone, Serialize, Deserialize, Default)]
14pub struct FallbackModel {
15 /// The model identifier to fall back to (e.g., "claude-opus-5")
16 pub(crate) model: String,
17 /// Optional max_tokens override for this fallback attempt
18 #[serde(skip_serializing_if = "Option::is_none")]
19 pub(crate) max_tokens: Option<u32>,
20 /// Optional thinking configuration override for this fallback attempt
21 #[serde(skip_serializing_if = "Option::is_none")]
22 pub(crate) thinking: Option<AnthropicThinkingConfig>,
23}
24
25/// Anthropic thinking configuration for fallback models.
26#[derive(Debug, Clone, Serialize, Deserialize, Default)]
27#[serde(tag = "type", rename_all = "lowercase")]
28pub enum AnthropicThinkingConfig {
29 #[default]
30 Disabled,
31 Enabled {
32 budget_tokens: u32,
33 #[serde(skip_serializing_if = "Option::is_none")]
34 display: Option<String>,
35 },
36 Adaptive {
37 #[serde(skip_serializing_if = "Option::is_none")]
38 display: Option<String>,
39 },
40}
41
42#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
43#[serde(rename_all = "snake_case")]
44pub enum PromptCacheProfile {
45 BudgetContinuation,
46}
47
48#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
49#[serde(rename_all = "snake_case")]
50pub enum AnthropicThinkingModeOverride {
51 #[default]
52 Inherit,
53 Disabled,
54 Adaptive,
55 ManualBudget(u32),
56}
57
58#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq, Default)]
59#[serde(rename_all = "snake_case")]
60pub enum AnthropicThinkingDisplayOverride {
61 #[default]
62 Inherit,
63 Summarized,
64 Omitted,
65 Updates,
66}
67
68#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
69#[serde(rename_all = "snake_case")]
70pub enum AnthropicOptionalStringOverride {
71 #[default]
72 Inherit,
73 Omit,
74 Explicit(String),
75}
76
77#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
78#[serde(rename_all = "snake_case")]
79pub enum AnthropicOptionalU32Override {
80 #[default]
81 Inherit,
82 Omit,
83 Explicit(u32),
84}
85
86#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
87pub struct AnthropicRequestOverrides {
88 #[serde(default)]
89 pub(crate) thinking_mode: AnthropicThinkingModeOverride,
90 #[serde(default)]
91 pub(crate) thinking_display: AnthropicThinkingDisplayOverride,
92 #[serde(default)]
93 pub(crate) effort: AnthropicOptionalStringOverride,
94 #[serde(default)]
95 pub(crate) task_budget_tokens: AnthropicOptionalU32Override,
96}
97
98/// Universal LLM request structure
99#[derive(Debug, Clone, Serialize, Deserialize, Default)]
100
101pub struct LLMRequest {
102 /// Conversation history shared via `Arc` so per-turn request construction
103 /// and continuation bookkeeping are O(1) instead of deep-copying the
104 /// entire history (which grows unbounded over a session).
105 pub messages: Arc<Vec<Message>>,
106 /// Shared via `Arc<str>` (single allocation) so per-turn request
107 /// construction can reuse the prompt envelope's `Arc<str>` via a cheap
108 /// refcount bump instead of cloning the full system prompt string.
109 pub system_prompt: Option<Arc<str>>,
110 pub tools: Option<Arc<Vec<ToolDefinition>>>,
111 pub model: String,
112 pub max_tokens: Option<u32>,
113 pub temperature: Option<f32>,
114 pub stream: bool,
115
116 /// Optional structured output JSON schema to request from providers that support it
117 /// For Anthropic this will be sent as `output_config.format = { type: "json_schema", schema: ... }`
118 pub output_format: Option<Value>,
119
120 /// Tool choice configuration based on official API docs
121 /// Supports: "auto" (default), "none", "any", or specific tool selection
122 pub tool_choice: Option<ToolChoice>,
123
124 /// Whether to enable parallel tool calls (OpenAI specific)
125 pub parallel_tool_calls: Option<bool>,
126
127 /// Parallel tool use configuration following Anthropic best practices
128 pub parallel_tool_config: Option<Box<ParallelToolConfig>>,
129
130 /// Reasoning effort level for models that support it (none, minimal, low, medium, high, xhigh)
131 /// Applies to: Claude, GPT-5 family, Gemini, Qwen3, DeepSeek with reasoning capability
132 pub reasoning_effort: Option<ReasoningEffortLevel>,
133
134 /// Effort level for overall token usage (low, medium, high, xhigh, max)
135 /// Applies to: Anthropic adaptive-thinking models such as Claude Opus 4.7
136 /// Controls how many tokens Claude uses when responding, trading off between
137 /// response thoroughness and token efficiency.
138 pub effort: Option<String>,
139
140 /// Verbosity level for output text (low, medium, high)
141 /// Applies to: GPT-5.4-family Responses workflows and other models that support verbosity control
142 pub verbosity: Option<VerbosityLevel>,
143
144 /// Advanced generation parameters
145 pub do_sample: Option<bool>,
146 pub top_p: Option<f32>,
147 pub top_k: Option<i32>,
148 pub presence_penalty: Option<f32>,
149 pub frequency_penalty: Option<f32>,
150 pub stop_sequences: Option<Vec<String>>,
151 /// Optional budget for extended thinking (Anthropic specific)
152 /// Minimum value: 1024
153 pub thinking_budget: Option<u32>,
154
155 /// Optional beta headers for Anthropic (and potentially others)
156 pub betas: Option<Vec<String>>,
157
158 /// Optional provider-specific context management configuration (Anthropic compaction/editing).
159 pub context_management: Option<Value>,
160
161 /// Optional coding agent specific settings
162 pub coding_agent_settings: Option<Box<CodingAgentSettings>>,
163
164 /// Optional turn metadata for git context (remote URLs, commit hash, etc.)
165 /// This is sent as X-Turn-Metadata header to providers that support it
166 pub metadata: Option<Value>,
167
168 /// Optional Responses API continuity pointer for server-side context chaining.
169 /// Used by providers that support stateful response continuation.
170 pub previous_response_id: Option<String>,
171
172 /// Optional Responses API storage flag.
173 /// When set, providers that support `store` pass this through directly.
174 pub response_store: Option<bool>,
175
176 /// Optional Responses API include fields (e.g. reasoning encrypted content).
177 /// Passed through only for providers/APIs that support include selectors.
178 pub responses_include: Option<Vec<String>>,
179
180 /// Optional native OpenAI `service_tier` request parameter.
181 /// Passed through only for native OpenAI endpoints that support service tiers.
182 pub service_tier: Option<String>,
183
184 /// Optional provider routing hint for prompt cache stickiness.
185 /// OpenAI uses this value to improve routing locality for repeated prefixes.
186 pub prompt_cache_key: Option<String>,
187
188 /// Optional request-scoped prompt cache profile for provider-specific TTL overrides.
189 pub prompt_cache_profile: Option<PromptCacheProfile>,
190
191 /// Optional explicit fallback models for Anthropic server-side fallback.
192 /// Overrides `provider.anthropic.fallbacks`; sent with the
193 /// `server-side-fallback-2026-06-01` beta header. Subject to the same
194 /// gates as the config list (first-party endpoint, a model that supports
195 /// server-side fallbacks, no credit-token retry) and the same entry
196 /// validation; an empty list sends nothing.
197 pub fallbacks: Option<Vec<FallbackModel>>,
198
199 /// Optional opaque credit token from a refused request's `stop_details.fallback_credit_token`.
200 /// Echoed on the retry to avoid paying the prompt-cache cost twice.
201 /// Sent with the `fallback-credit-2026-07-01` beta header. The refused request must have
202 /// carried that header or `server-side-fallback-2026-07-01`, which grants the same
203 /// `stop_details` fields.
204 pub fallback_credit_token: Option<String>,
205
206 /// Optional Anthropic-specific request overrides used when request semantics must
207 /// not inherit VT Code's provider defaults, such as the Anthropic compatibility server.
208 pub anthropic_request_overrides: Option<AnthropicRequestOverrides>,
209}
210
211/// Per-model sampling parameter overrides resolved from a custom provider
212/// profile. Every field defaults to `None`, meaning the agent loop's global
213/// configuration value applies unchanged.
214#[derive(Debug, Clone, Copy, Default, PartialEq)]
215pub struct SamplingOverrides {
216 /// Sampling temperature (0.0-2.0).
217 pub temperature: Option<f32>,
218 /// Nucleus sampling mass (0.0-1.0).
219 pub top_p: Option<f32>,
220 /// Top-k token cutoff (>= 0).
221 pub top_k: Option<i32>,
222 /// Presence penalty (-2.0-2.0).
223 pub presence_penalty: Option<f32>,
224 /// Frequency penalty (-2.0-2.0).
225 pub frequency_penalty: Option<f32>,
226 /// Maximum output tokens (> 0).
227 pub max_tokens: Option<u32>,
228 /// Reasoning effort for models that accept it.
229 pub reasoning_effort: Option<ReasoningEffortLevel>,
230 /// Whether this model's wire format rejects sampling parameters while
231 /// reasoning is active (e.g. custom endpoints speaking the Anthropic
232 /// Messages shape).
233 pub suppresses_sampling_with_reasoning: bool,
234 /// Set by providers that resolve overrides from an explicit user profile;
235 /// callers use it to trust these values over name-based heuristics.
236 pub profile_aware: bool,
237}
238
239impl SamplingOverrides {
240 /// Whether sampling parameters must be dropped for this request: either
241 /// the model's wire format rejects sampling while reasoning is active, or
242 /// a built-in Anthropic/MiniMax-shaped backend (`native_match`) enforces
243 /// the same rule outside of profile-aware routing.
244 pub fn suppresses_sampling(&self, native_match: bool, reasoning_active: bool) -> bool {
245 reasoning_active && (self.suppresses_sampling_with_reasoning || (!self.profile_aware && native_match))
246 }
247}
248
249/// Optional overrides for standalone Responses compaction requests.
250#[derive(Debug, Clone, Serialize, Deserialize, Default, PartialEq, Eq)]
251pub struct ResponsesCompactionOptions {
252 /// Optional custom instructions appended to the derived replay instructions.
253 pub instructions: Option<String>,
254 /// Optional output token limit for the compaction response.
255 pub max_output_tokens: Option<u32>,
256 /// Optional reasoning effort override for the compaction pass.
257 pub reasoning_effort: Option<ReasoningEffortLevel>,
258 /// Optional verbosity override for the compaction output text settings.
259 pub verbosity: Option<VerbosityLevel>,
260 /// Optional include selectors override.
261 pub responses_include: Option<Vec<String>>,
262 /// Optional storage override.
263 pub response_store: Option<bool>,
264 /// Optional native OpenAI service tier override.
265 pub service_tier: Option<String>,
266 /// Optional prompt cache routing override.
267 pub prompt_cache_key: Option<String>,
268}
269
270/// Settings to refine model behavior for coding agent tasks
271#[derive(Debug, Clone, Serialize, Deserialize, Default)]
272pub struct CodingAgentSettings {
273 /// Optimize for long context by hoisting the largest user message.
274 /// Hoisting reorders earlier turns, so the Anthropic builder skips it on
275 /// preserved-thinking models (Claude Sonnet 5.5, Claude Opus 5.5, Claude
276 /// Fable 5.1).
277 pub(crate) long_context_optimization: bool,
278}
279
280/// Tool choice configuration that works across different providers
281/// Based on OpenAI, Anthropic, and Gemini API specifications
282/// Follows Anthropic's tool use best practices for optimal performance
283#[derive(Debug, Clone, Serialize, Deserialize)]
284#[serde(untagged)]
285#[derive(Default)]
286pub enum ToolChoice {
287 /// Let the model decide whether to call tools ("auto")
288 /// Default behavior - allows model to use tools when appropriate
289 #[default]
290 Auto,
291
292 /// Force the model to not call any tools ("none")
293 /// Useful for pure conversational responses without tool usage
294 None,
295
296 /// Force the model to call at least one tool ("any")
297 /// Ensures tool usage even when model might prefer direct response
298 Any,
299
300 /// Force the model to call a specific tool
301 /// Useful for directing model to use particular functionality
302 Specific(SpecificToolChoice),
303
304 /// OpenAI Responses native allowed-tools filtering.
305 ///
306 /// This is advisory provider-side filtering only; runtime dispatch remains
307 /// authoritative. Providers without native support should degrade this to
308 /// the equivalent auto/required tool choice while keeping the full
309 /// catalogue visible.
310 AllowedTools(AllowedToolsChoice),
311}
312
313#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
314#[serde(rename_all = "lowercase")]
315pub enum AllowedToolsMode {
316 Auto,
317 Required,
318}
319
320impl AllowedToolsMode {
321 pub fn as_str(self) -> &'static str {
322 match self {
323 Self::Auto => "auto",
324 Self::Required => "required",
325 }
326 }
327
328 fn anthropic_mode(self) -> &'static str {
329 match self {
330 Self::Auto => "auto",
331 Self::Required => "any",
332 }
333 }
334
335 fn gemini_mode(self) -> &'static str {
336 match self {
337 Self::Auto => "auto",
338 Self::Required => "any",
339 }
340 }
341}
342
343#[derive(Debug, Clone, Serialize, Deserialize)]
344pub struct AllowedToolsChoice {
345 pub mode: AllowedToolsMode,
346 pub tools: Vec<String>,
347}
348
349/// Specific tool choice for forcing a particular function call
350#[derive(Debug, Clone, Serialize, Deserialize)]
351pub struct SpecificToolChoice {
352 #[serde(rename = "type")]
353 pub(crate) tool_type: String, // "function"
354
355 pub(crate) function: SpecificFunctionChoice,
356}
357
358/// Specific function choice details
359#[derive(Debug, Clone, Serialize, Deserialize)]
360pub struct SpecificFunctionChoice {
361 pub(crate) name: String,
362}
363
364impl ToolChoice {
365 /// Create auto tool choice (default behavior)
366 pub fn auto() -> Self {
367 Self::Auto
368 }
369
370 /// Create none tool choice (disable tool calling)
371 pub fn none() -> Self {
372 Self::None
373 }
374
375 /// Create any tool choice (force at least one tool call)
376 pub(crate) fn any() -> Self {
377 Self::Any
378 }
379
380 /// Create specific function tool choice
381 pub(crate) fn function(name: String) -> Self {
382 Self::Specific(SpecificToolChoice {
383 tool_type: "function".to_owned(),
384 function: SpecificFunctionChoice { name },
385 })
386 }
387
388 pub fn allowed_tools_auto(tools: Vec<String>) -> Self {
389 Self::AllowedTools(AllowedToolsChoice { mode: AllowedToolsMode::Auto, tools })
390 }
391
392 /// Check if this tool choice allows parallel tool use
393 /// Based on Anthropic's parallel tool use guidelines
394 pub fn allows_parallel_tools(&self) -> bool {
395 match self {
396 // Auto allows parallel tools by default
397 Self::Auto => true,
398 // Any forces at least one tool, may allow parallel
399 Self::Any => true,
400 Self::AllowedTools(choice) => matches!(choice.mode, AllowedToolsMode::Auto),
401 // Specific forces one particular tool, typically no parallel
402 Self::Specific(_) => false,
403 // None disables tools entirely
404 Self::None => false,
405 }
406 }
407
408 /// Get human-readable description of tool choice behavior
409 pub fn description(&self) -> &'static str {
410 match self {
411 Self::Auto => "Model decides when to use tools (allows parallel)",
412 Self::None => "No tools will be used",
413 Self::Any => "At least one tool must be used (allows parallel)",
414 Self::AllowedTools(_) => "Model decides when to use the currently allowed tool subset",
415 Self::Specific(_) => "Specific tool must be used (no parallel)",
416 }
417 }
418
419 /// OpenAI-compatible providers that share the same tool_choice format
420 const OPENAI_STYLE_PROVIDERS: &'static [&'static str] = &[
421 "openai",
422 "deepseek",
423 "huggingface",
424 "mistral",
425 "openrouter",
426 "zai",
427 "moonshot",
428 "stepfun",
429 "evolink",
430 "lmstudio",
431 "llamacpp",
432 "meta",
433 ];
434
435 /// Convert to provider-specific format
436 #[inline]
437 pub(crate) fn to_provider_format(&self, provider: &str) -> Value {
438 if Self::OPENAI_STYLE_PROVIDERS.contains(&provider) {
439 return self.to_openai_format();
440 }
441
442 match provider {
443 "anthropic" => self.to_anthropic_format(),
444 "gemini" => self.to_gemini_format(),
445 _ => self.to_openai_format(), // Default to OpenAI format
446 }
447 }
448
449 #[inline]
450 fn to_openai_format(&self) -> Value {
451 match self {
452 Self::Auto => json!("auto"),
453 Self::None => json!("none"),
454 Self::Any => json!("required"),
455 Self::Specific(choice) => json!(choice),
456 Self::AllowedTools(choice) => json!(choice.mode.as_str()),
457 }
458 }
459
460 #[inline]
461 fn to_anthropic_format(&self) -> Value {
462 match self {
463 Self::Auto => json!({"type": "auto"}),
464 Self::None => json!({"type": "none"}),
465 Self::Any => json!({"type": "any"}),
466 Self::Specific(choice) => json!({"type": "tool", "name": &choice.function.name}),
467 Self::AllowedTools(choice) => json!({"type": choice.mode.anthropic_mode()}),
468 }
469 }
470
471 #[inline]
472 fn to_gemini_format(&self) -> Value {
473 match self {
474 Self::Auto => json!({"mode": "auto"}),
475 Self::None => json!({"mode": "none"}),
476 Self::Any => json!({"mode": "any"}),
477 Self::AllowedTools(choice) => json!({"mode": choice.mode.gemini_mode()}),
478 Self::Specific(choice) => {
479 json!({"mode": "any", "allowed_function_names": [&choice.function.name]})
480 }
481 }
482 }
483}
484
485/// Configuration for parallel tool use behavior
486/// Based on Anthropic's parallel tool use guidelines
487#[derive(Debug, Clone, Serialize, Deserialize)]
488pub struct ParallelToolConfig {
489 /// Whether to disable parallel tool use
490 /// When true, forces sequential tool execution
491 pub(crate) disable_parallel_tool_use: bool,
492
493 /// Maximum number of tools to execute in parallel
494 /// None means no limit (provider default)
495 pub(crate) max_parallel_tools: Option<usize>,
496
497 /// Whether to encourage parallel tool use in prompts
498 pub(crate) encourage_parallel: bool,
499}
500
501impl Default for ParallelToolConfig {
502 fn default() -> Self {
503 Self {
504 disable_parallel_tool_use: false,
505 max_parallel_tools: Some(5), // Reasonable default
506 encourage_parallel: true,
507 }
508 }
509}
510
511impl ParallelToolConfig {
512 /// Create configuration optimized for Anthropic models
513 pub fn anthropic_optimized() -> Self {
514 Self {
515 disable_parallel_tool_use: false,
516 max_parallel_tools: None, // Let Anthropic decide
517 encourage_parallel: true,
518 }
519 }
520
521 /// Create configuration for sequential tool use
522 pub(crate) fn sequential_only() -> Self {
523 Self {
524 disable_parallel_tool_use: true,
525 max_parallel_tools: Some(1),
526 encourage_parallel: false,
527 }
528 }
529}