Skip to main content

vtcode_llm/provider/
request.rs

1use serde::{Deserialize, Serialize};
2use serde_json::{Value, json};
3use std::sync::Arc;
4use vtcode_config::types::{ReasoningEffortLevel, VerbosityLevel};
5
6use super::{Message, ToolDefinition};
7
8/// Fallback model configuration for Anthropic server-side fallback.
9/// The explicit-list form of `fallbacks`, sent with the
10/// `server-side-fallback-2026-06-01` beta header (the `"default"` keyword form
11/// uses `server-side-fallback-2026-07-01`; pairing a header with the other form
12/// is rejected).
13#[derive(Debug, Clone, Serialize, Deserialize, Default)]
14pub struct FallbackModel {
15    /// The model identifier to fall back to (e.g., "claude-opus-5")
16    pub(crate) model: String,
17    /// Optional max_tokens override for this fallback attempt
18    #[serde(skip_serializing_if = "Option::is_none")]
19    pub(crate) max_tokens: Option<u32>,
20    /// Optional thinking configuration override for this fallback attempt
21    #[serde(skip_serializing_if = "Option::is_none")]
22    pub(crate) thinking: Option<AnthropicThinkingConfig>,
23}
24
25/// Anthropic thinking configuration for fallback models.
26#[derive(Debug, Clone, Serialize, Deserialize, Default)]
27#[serde(tag = "type", rename_all = "lowercase")]
28pub enum AnthropicThinkingConfig {
29    #[default]
30    Disabled,
31    Enabled {
32        budget_tokens: u32,
33        #[serde(skip_serializing_if = "Option::is_none")]
34        display: Option<String>,
35    },
36    Adaptive {
37        #[serde(skip_serializing_if = "Option::is_none")]
38        display: Option<String>,
39    },
40}
41
42#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
43#[serde(rename_all = "snake_case")]
44pub enum PromptCacheProfile {
45    BudgetContinuation,
46}
47
48#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
49#[serde(rename_all = "snake_case")]
50pub enum AnthropicThinkingModeOverride {
51    #[default]
52    Inherit,
53    Disabled,
54    Adaptive,
55    ManualBudget(u32),
56}
57
58#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq, Default)]
59#[serde(rename_all = "snake_case")]
60pub enum AnthropicThinkingDisplayOverride {
61    #[default]
62    Inherit,
63    Summarized,
64    Omitted,
65    Updates,
66}
67
68#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
69#[serde(rename_all = "snake_case")]
70pub enum AnthropicOptionalStringOverride {
71    #[default]
72    Inherit,
73    Omit,
74    Explicit(String),
75}
76
77#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
78#[serde(rename_all = "snake_case")]
79pub enum AnthropicOptionalU32Override {
80    #[default]
81    Inherit,
82    Omit,
83    Explicit(u32),
84}
85
86#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Default)]
87pub struct AnthropicRequestOverrides {
88    #[serde(default)]
89    pub(crate) thinking_mode: AnthropicThinkingModeOverride,
90    #[serde(default)]
91    pub(crate) thinking_display: AnthropicThinkingDisplayOverride,
92    #[serde(default)]
93    pub(crate) effort: AnthropicOptionalStringOverride,
94    #[serde(default)]
95    pub(crate) task_budget_tokens: AnthropicOptionalU32Override,
96}
97
98/// Universal LLM request structure
99#[derive(Debug, Clone, Serialize, Deserialize, Default)]
100
101pub struct LLMRequest {
102    /// Conversation history shared via `Arc` so per-turn request construction
103    /// and continuation bookkeeping are O(1) instead of deep-copying the
104    /// entire history (which grows unbounded over a session).
105    pub messages: Arc<Vec<Message>>,
106    /// Shared via `Arc<str>` (single allocation) so per-turn request
107    /// construction can reuse the prompt envelope's `Arc<str>` via a cheap
108    /// refcount bump instead of cloning the full system prompt string.
109    pub system_prompt: Option<Arc<str>>,
110    pub tools: Option<Arc<Vec<ToolDefinition>>>,
111    pub model: String,
112    pub max_tokens: Option<u32>,
113    pub temperature: Option<f32>,
114    pub stream: bool,
115
116    /// Optional structured output JSON schema to request from providers that support it
117    /// For Anthropic this will be sent as `output_config.format = { type: "json_schema", schema: ... }`
118    pub output_format: Option<Value>,
119
120    /// Tool choice configuration based on official API docs
121    /// Supports: "auto" (default), "none", "any", or specific tool selection
122    pub tool_choice: Option<ToolChoice>,
123
124    /// Whether to enable parallel tool calls (OpenAI specific)
125    pub parallel_tool_calls: Option<bool>,
126
127    /// Parallel tool use configuration following Anthropic best practices
128    pub parallel_tool_config: Option<Box<ParallelToolConfig>>,
129
130    /// Reasoning effort level for models that support it (none, minimal, low, medium, high, xhigh)
131    /// Applies to: Claude, GPT-5 family, Gemini, Qwen3, DeepSeek with reasoning capability
132    pub reasoning_effort: Option<ReasoningEffortLevel>,
133
134    /// Effort level for overall token usage (low, medium, high, xhigh, max)
135    /// Applies to: Anthropic adaptive-thinking models such as Claude Opus 4.7
136    /// Controls how many tokens Claude uses when responding, trading off between
137    /// response thoroughness and token efficiency.
138    pub effort: Option<String>,
139
140    /// Verbosity level for output text (low, medium, high)
141    /// Applies to: GPT-5.4-family Responses workflows and other models that support verbosity control
142    pub verbosity: Option<VerbosityLevel>,
143
144    /// Advanced generation parameters
145    pub do_sample: Option<bool>,
146    pub top_p: Option<f32>,
147    pub top_k: Option<i32>,
148    pub presence_penalty: Option<f32>,
149    pub frequency_penalty: Option<f32>,
150    pub stop_sequences: Option<Vec<String>>,
151    /// Optional budget for extended thinking (Anthropic specific)
152    /// Minimum value: 1024
153    pub thinking_budget: Option<u32>,
154
155    /// Optional beta headers for Anthropic (and potentially others)
156    pub betas: Option<Vec<String>>,
157
158    /// Optional provider-specific context management configuration (Anthropic compaction/editing).
159    pub context_management: Option<Value>,
160
161    /// Optional coding agent specific settings
162    pub coding_agent_settings: Option<Box<CodingAgentSettings>>,
163
164    /// Optional turn metadata for git context (remote URLs, commit hash, etc.)
165    /// This is sent as X-Turn-Metadata header to providers that support it
166    pub metadata: Option<Value>,
167
168    /// Optional Responses API continuity pointer for server-side context chaining.
169    /// Used by providers that support stateful response continuation.
170    pub previous_response_id: Option<String>,
171
172    /// Optional Responses API storage flag.
173    /// When set, providers that support `store` pass this through directly.
174    pub response_store: Option<bool>,
175
176    /// Optional Responses API include fields (e.g. reasoning encrypted content).
177    /// Passed through only for providers/APIs that support include selectors.
178    pub responses_include: Option<Vec<String>>,
179
180    /// Optional native OpenAI `service_tier` request parameter.
181    /// Passed through only for native OpenAI endpoints that support service tiers.
182    pub service_tier: Option<String>,
183
184    /// Optional provider routing hint for prompt cache stickiness.
185    /// OpenAI uses this value to improve routing locality for repeated prefixes.
186    pub prompt_cache_key: Option<String>,
187
188    /// Optional request-scoped prompt cache profile for provider-specific TTL overrides.
189    pub prompt_cache_profile: Option<PromptCacheProfile>,
190
191    /// Optional explicit fallback models for Anthropic server-side fallback.
192    /// Overrides `provider.anthropic.fallbacks`; sent with the
193    /// `server-side-fallback-2026-06-01` beta header. Subject to the same
194    /// gates as the config list (first-party endpoint, a model that supports
195    /// server-side fallbacks, no credit-token retry) and the same entry
196    /// validation; an empty list sends nothing.
197    pub fallbacks: Option<Vec<FallbackModel>>,
198
199    /// Optional opaque credit token from a refused request's `stop_details.fallback_credit_token`.
200    /// Echoed on the retry to avoid paying the prompt-cache cost twice.
201    /// Sent with the `fallback-credit-2026-07-01` beta header. The refused request must have
202    /// carried that header or `server-side-fallback-2026-07-01`, which grants the same
203    /// `stop_details` fields.
204    pub fallback_credit_token: Option<String>,
205
206    /// Optional Anthropic-specific request overrides used when request semantics must
207    /// not inherit VT Code's provider defaults, such as the Anthropic compatibility server.
208    pub anthropic_request_overrides: Option<AnthropicRequestOverrides>,
209}
210
211/// Per-model sampling parameter overrides resolved from a custom provider
212/// profile. Every field defaults to `None`, meaning the agent loop's global
213/// configuration value applies unchanged.
214#[derive(Debug, Clone, Copy, Default, PartialEq)]
215pub struct SamplingOverrides {
216    /// Sampling temperature (0.0-2.0).
217    pub temperature: Option<f32>,
218    /// Nucleus sampling mass (0.0-1.0).
219    pub top_p: Option<f32>,
220    /// Top-k token cutoff (>= 0).
221    pub top_k: Option<i32>,
222    /// Presence penalty (-2.0-2.0).
223    pub presence_penalty: Option<f32>,
224    /// Frequency penalty (-2.0-2.0).
225    pub frequency_penalty: Option<f32>,
226    /// Maximum output tokens (> 0).
227    pub max_tokens: Option<u32>,
228    /// Reasoning effort for models that accept it.
229    pub reasoning_effort: Option<ReasoningEffortLevel>,
230    /// Whether this model's wire format rejects sampling parameters while
231    /// reasoning is active (e.g. custom endpoints speaking the Anthropic
232    /// Messages shape).
233    pub suppresses_sampling_with_reasoning: bool,
234    /// Set by providers that resolve overrides from an explicit user profile;
235    /// callers use it to trust these values over name-based heuristics.
236    pub profile_aware: bool,
237}
238
239impl SamplingOverrides {
240    /// Whether sampling parameters must be dropped for this request: either
241    /// the model's wire format rejects sampling while reasoning is active, or
242    /// a built-in Anthropic/MiniMax-shaped backend (`native_match`) enforces
243    /// the same rule outside of profile-aware routing.
244    pub fn suppresses_sampling(&self, native_match: bool, reasoning_active: bool) -> bool {
245        reasoning_active && (self.suppresses_sampling_with_reasoning || (!self.profile_aware && native_match))
246    }
247}
248
249/// Optional overrides for standalone Responses compaction requests.
250#[derive(Debug, Clone, Serialize, Deserialize, Default, PartialEq, Eq)]
251pub struct ResponsesCompactionOptions {
252    /// Optional custom instructions appended to the derived replay instructions.
253    pub instructions: Option<String>,
254    /// Optional output token limit for the compaction response.
255    pub max_output_tokens: Option<u32>,
256    /// Optional reasoning effort override for the compaction pass.
257    pub reasoning_effort: Option<ReasoningEffortLevel>,
258    /// Optional verbosity override for the compaction output text settings.
259    pub verbosity: Option<VerbosityLevel>,
260    /// Optional include selectors override.
261    pub responses_include: Option<Vec<String>>,
262    /// Optional storage override.
263    pub response_store: Option<bool>,
264    /// Optional native OpenAI service tier override.
265    pub service_tier: Option<String>,
266    /// Optional prompt cache routing override.
267    pub prompt_cache_key: Option<String>,
268}
269
270/// Settings to refine model behavior for coding agent tasks
271#[derive(Debug, Clone, Serialize, Deserialize, Default)]
272pub struct CodingAgentSettings {
273    /// Optimize for long context by hoisting the largest user message.
274    /// Hoisting reorders earlier turns, so the Anthropic builder skips it on
275    /// preserved-thinking models (Claude Sonnet 5.5, Claude Opus 5.5, Claude
276    /// Fable 5.1).
277    pub(crate) long_context_optimization: bool,
278}
279
280/// Tool choice configuration that works across different providers
281/// Based on OpenAI, Anthropic, and Gemini API specifications
282/// Follows Anthropic's tool use best practices for optimal performance
283#[derive(Debug, Clone, Serialize, Deserialize)]
284#[serde(untagged)]
285#[derive(Default)]
286pub enum ToolChoice {
287    /// Let the model decide whether to call tools ("auto")
288    /// Default behavior - allows model to use tools when appropriate
289    #[default]
290    Auto,
291
292    /// Force the model to not call any tools ("none")
293    /// Useful for pure conversational responses without tool usage
294    None,
295
296    /// Force the model to call at least one tool ("any")
297    /// Ensures tool usage even when model might prefer direct response
298    Any,
299
300    /// Force the model to call a specific tool
301    /// Useful for directing model to use particular functionality
302    Specific(SpecificToolChoice),
303
304    /// OpenAI Responses native allowed-tools filtering.
305    ///
306    /// This is advisory provider-side filtering only; runtime dispatch remains
307    /// authoritative. Providers without native support should degrade this to
308    /// the equivalent auto/required tool choice while keeping the full
309    /// catalogue visible.
310    AllowedTools(AllowedToolsChoice),
311}
312
313#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
314#[serde(rename_all = "lowercase")]
315pub enum AllowedToolsMode {
316    Auto,
317    Required,
318}
319
320impl AllowedToolsMode {
321    pub fn as_str(self) -> &'static str {
322        match self {
323            Self::Auto => "auto",
324            Self::Required => "required",
325        }
326    }
327
328    fn anthropic_mode(self) -> &'static str {
329        match self {
330            Self::Auto => "auto",
331            Self::Required => "any",
332        }
333    }
334
335    fn gemini_mode(self) -> &'static str {
336        match self {
337            Self::Auto => "auto",
338            Self::Required => "any",
339        }
340    }
341}
342
343#[derive(Debug, Clone, Serialize, Deserialize)]
344pub struct AllowedToolsChoice {
345    pub mode: AllowedToolsMode,
346    pub tools: Vec<String>,
347}
348
349/// Specific tool choice for forcing a particular function call
350#[derive(Debug, Clone, Serialize, Deserialize)]
351pub struct SpecificToolChoice {
352    #[serde(rename = "type")]
353    pub(crate) tool_type: String, // "function"
354
355    pub(crate) function: SpecificFunctionChoice,
356}
357
358/// Specific function choice details
359#[derive(Debug, Clone, Serialize, Deserialize)]
360pub struct SpecificFunctionChoice {
361    pub(crate) name: String,
362}
363
364impl ToolChoice {
365    /// Create auto tool choice (default behavior)
366    pub fn auto() -> Self {
367        Self::Auto
368    }
369
370    /// Create none tool choice (disable tool calling)
371    pub fn none() -> Self {
372        Self::None
373    }
374
375    /// Create any tool choice (force at least one tool call)
376    pub(crate) fn any() -> Self {
377        Self::Any
378    }
379
380    /// Create specific function tool choice
381    pub(crate) fn function(name: String) -> Self {
382        Self::Specific(SpecificToolChoice {
383            tool_type: "function".to_owned(),
384            function: SpecificFunctionChoice { name },
385        })
386    }
387
388    pub fn allowed_tools_auto(tools: Vec<String>) -> Self {
389        Self::AllowedTools(AllowedToolsChoice { mode: AllowedToolsMode::Auto, tools })
390    }
391
392    /// Check if this tool choice allows parallel tool use
393    /// Based on Anthropic's parallel tool use guidelines
394    pub fn allows_parallel_tools(&self) -> bool {
395        match self {
396            // Auto allows parallel tools by default
397            Self::Auto => true,
398            // Any forces at least one tool, may allow parallel
399            Self::Any => true,
400            Self::AllowedTools(choice) => matches!(choice.mode, AllowedToolsMode::Auto),
401            // Specific forces one particular tool, typically no parallel
402            Self::Specific(_) => false,
403            // None disables tools entirely
404            Self::None => false,
405        }
406    }
407
408    /// Get human-readable description of tool choice behavior
409    pub fn description(&self) -> &'static str {
410        match self {
411            Self::Auto => "Model decides when to use tools (allows parallel)",
412            Self::None => "No tools will be used",
413            Self::Any => "At least one tool must be used (allows parallel)",
414            Self::AllowedTools(_) => "Model decides when to use the currently allowed tool subset",
415            Self::Specific(_) => "Specific tool must be used (no parallel)",
416        }
417    }
418
419    /// OpenAI-compatible providers that share the same tool_choice format
420    const OPENAI_STYLE_PROVIDERS: &'static [&'static str] = &[
421        "openai",
422        "deepseek",
423        "huggingface",
424        "mistral",
425        "openrouter",
426        "zai",
427        "moonshot",
428        "stepfun",
429        "evolink",
430        "lmstudio",
431        "llamacpp",
432        "meta",
433    ];
434
435    /// Convert to provider-specific format
436    #[inline]
437    pub(crate) fn to_provider_format(&self, provider: &str) -> Value {
438        if Self::OPENAI_STYLE_PROVIDERS.contains(&provider) {
439            return self.to_openai_format();
440        }
441
442        match provider {
443            "anthropic" => self.to_anthropic_format(),
444            "gemini" => self.to_gemini_format(),
445            _ => self.to_openai_format(), // Default to OpenAI format
446        }
447    }
448
449    #[inline]
450    fn to_openai_format(&self) -> Value {
451        match self {
452            Self::Auto => json!("auto"),
453            Self::None => json!("none"),
454            Self::Any => json!("required"),
455            Self::Specific(choice) => json!(choice),
456            Self::AllowedTools(choice) => json!(choice.mode.as_str()),
457        }
458    }
459
460    #[inline]
461    fn to_anthropic_format(&self) -> Value {
462        match self {
463            Self::Auto => json!({"type": "auto"}),
464            Self::None => json!({"type": "none"}),
465            Self::Any => json!({"type": "any"}),
466            Self::Specific(choice) => json!({"type": "tool", "name": &choice.function.name}),
467            Self::AllowedTools(choice) => json!({"type": choice.mode.anthropic_mode()}),
468        }
469    }
470
471    #[inline]
472    fn to_gemini_format(&self) -> Value {
473        match self {
474            Self::Auto => json!({"mode": "auto"}),
475            Self::None => json!({"mode": "none"}),
476            Self::Any => json!({"mode": "any"}),
477            Self::AllowedTools(choice) => json!({"mode": choice.mode.gemini_mode()}),
478            Self::Specific(choice) => {
479                json!({"mode": "any", "allowed_function_names": [&choice.function.name]})
480            }
481        }
482    }
483}
484
485/// Configuration for parallel tool use behavior
486/// Based on Anthropic's parallel tool use guidelines
487#[derive(Debug, Clone, Serialize, Deserialize)]
488pub struct ParallelToolConfig {
489    /// Whether to disable parallel tool use
490    /// When true, forces sequential tool execution
491    pub(crate) disable_parallel_tool_use: bool,
492
493    /// Maximum number of tools to execute in parallel
494    /// None means no limit (provider default)
495    pub(crate) max_parallel_tools: Option<usize>,
496
497    /// Whether to encourage parallel tool use in prompts
498    pub(crate) encourage_parallel: bool,
499}
500
501impl Default for ParallelToolConfig {
502    fn default() -> Self {
503        Self {
504            disable_parallel_tool_use: false,
505            max_parallel_tools: Some(5), // Reasonable default
506            encourage_parallel: true,
507        }
508    }
509}
510
511impl ParallelToolConfig {
512    /// Create configuration optimized for Anthropic models
513    pub fn anthropic_optimized() -> Self {
514        Self {
515            disable_parallel_tool_use: false,
516            max_parallel_tools: None, // Let Anthropic decide
517            encourage_parallel: true,
518        }
519    }
520
521    /// Create configuration for sequential tool use
522    pub(crate) fn sequential_only() -> Self {
523        Self {
524            disable_parallel_tool_use: true,
525            max_parallel_tools: Some(1),
526            encourage_parallel: false,
527        }
528    }
529}