Skip to main content

vtcode_llm/provider/
provider_trait.rs

1use async_stream::try_stream;
2use async_trait::async_trait;
3use compact_str::format_compact;
4use once_cell::sync::Lazy;
5use rustc_hash::FxHashMap;
6use std::sync::RwLock;
7use vtcode_commons::llm::BackendKind;
8
9use vtcode_commons::tool_types::CompactStr;
10
11use super::{
12    LLMNormalizedStream, LLMRequest, LLMResponse, LLMStream, LLMStreamEvent, Message, ResponsesCompactionOptions,
13    SamplingOverrides,
14};
15pub use vtcode_commons::llm::{LLMError, LLMErrorMetadata};
16
17/// Generic effort levels supported by providers that advertise configurable
18/// reasoning without exposing model-specific catalog metadata.
19pub(crate) const GENERIC_REASONING_EFFORTS: &[&str] = &["low", "medium", "high"];
20
21/// Return the catalog's effort metadata while preserving the distinction
22/// between an unknown route and a known route that only exposes structured
23/// reasoning. An empty catalog list is authoritative: it means the route does
24/// not accept a configurable effort payload.
25pub(crate) fn catalog_reasoning_efforts(provider: &str, model: &str) -> Option<&'static [&'static str]> {
26    vtcode_config::models::model_catalog_entry(provider, model).map(|entry| entry.reasoning_efforts)
27}
28
29/// Resolve the catalog's exact effort list while retaining the explicit
30/// generic contract for provider routes that advertise effort support without
31/// a built-in model entry (for example, an OpenAI-compatible custom endpoint).
32pub(crate) fn catalog_or_generic_reasoning_efforts(provider: &str, model: &str) -> &'static [&'static str] {
33    catalog_reasoning_efforts(provider, model).unwrap_or(GENERIC_REASONING_EFFORTS)
34}
35
36/// Resolve catalog levels while allowing an explicitly configured custom route
37/// to advertise the shared generic effort contract. Unsupported unknown routes
38/// remain empty so request builders fail closed.
39pub(crate) fn catalog_or_explicit_reasoning_efforts(
40    provider: &str,
41    model: &str,
42    explicitly_supported: bool,
43) -> &'static [&'static str] {
44    catalog_reasoning_efforts(provider, model)
45        .or_else(|| explicitly_supported.then_some(GENERIC_REASONING_EFFORTS))
46        .unwrap_or(&[])
47}
48
49/// Resolve a model's catalog context window with a provider-specific fallback.
50///
51/// Provider adapters use this helper instead of maintaining independent model
52/// name tables. Explicit provider overrides remain the caller's responsibility
53/// and are applied before this lookup.
54pub(crate) fn catalog_context_window(provider: &str, model: &str, fallback: usize) -> usize {
55    vtcode_config::models::model_catalog_entry(provider, model)
56        .map(|entry| entry.context_window)
57        .filter(|context_window| *context_window > 0)
58        .unwrap_or(fallback)
59}
60
61/// Cached provider capabilities to reduce repeated trait method calls
62#[derive(Debug, Clone)]
63pub struct ProviderCapabilities {
64    pub(crate) provider_name: String,
65    pub(crate) model: String,
66    pub streaming: bool,
67    pub reasoning: bool,
68    pub reasoning_effort: bool,
69    pub tools: bool,
70    pub parallel_tool_config: bool,
71    pub(crate) structured_output: bool,
72    pub(crate) context_caching: bool,
73    pub responses_compaction: bool,
74    pub context_edits: bool,
75    /// Whether the selected provider/model can carry Anthropic's native
76    /// turn-scoped system-message lifecycle field on the wire.
77    pub turn_scoped_system_messages: bool,
78    pub(crate) vision: bool,
79    pub(crate) context_size: usize,
80}
81
82impl ProviderCapabilities {
83    fn detect(provider: &dyn LLMProvider, model: &str) -> Self {
84        Self {
85            provider_name: provider.name().to_string(),
86            model: model.to_string(),
87            streaming: provider.supports_streaming(),
88            reasoning: provider.supports_reasoning(model),
89            reasoning_effort: !provider.supported_reasoning_efforts(model).is_empty(),
90            tools: provider.supports_tools(model),
91            parallel_tool_config: provider.supports_parallel_tool_config(model),
92            structured_output: provider.supports_structured_output(model),
93            context_caching: provider.supports_context_caching(model),
94            responses_compaction: provider.supports_responses_compaction(model),
95            context_edits: provider.supports_context_edits(model),
96            turn_scoped_system_messages: provider.supports_turn_scoped_system_messages(model),
97            vision: provider.supports_vision(model),
98            context_size: provider.effective_context_size(model),
99        }
100    }
101
102    pub(crate) fn has_advanced_features(&self) -> bool {
103        self.reasoning || self.structured_output || self.context_caching || self.reasoning_effort
104    }
105
106    pub(crate) fn summary(&self) -> String {
107        let mut features = Vec::new();
108
109        if self.streaming {
110            features.push("streaming");
111        }
112        if self.reasoning {
113            features.push("advanced-reasoning");
114        }
115        if self.reasoning_effort {
116            features.push("reasoning-effort");
117        }
118        if self.structured_output {
119            features.push("structured-output");
120        }
121        if self.context_caching {
122            features.push("context-caching");
123        }
124        if self.parallel_tool_config {
125            features.push("parallel-tools");
126        }
127        if self.responses_compaction {
128            features.push("responses-compaction");
129        }
130        if self.context_edits {
131            features.push("context-edits");
132        }
133
134        let features_str = if features.is_empty() {
135            "basic".to_string()
136        } else {
137            features.join(", ")
138        };
139
140        format!("{} ({} tokens): {}", self.model, self.context_size, features_str)
141    }
142}
143
144/// Global cache for provider capabilities (provider_name::model -> capabilities)
145static CAPABILITY_CACHE: Lazy<RwLock<FxHashMap<CompactStr, ProviderCapabilities>>> =
146    Lazy::new(|| RwLock::new(FxHashMap::default()));
147
148/// Extract and cache provider capabilities for a given provider and model
149pub fn get_cached_capabilities(provider: &dyn LLMProvider, model: &str) -> ProviderCapabilities {
150    let cache_key = format_compact!("{}::{}::{}", provider.name(), model, provider.effective_context_size(model));
151
152    // Check if already cached
153    if let Ok(cache) = CAPABILITY_CACHE.read()
154        && let Some(caps) = cache.get(&cache_key)
155    {
156        return caps.clone();
157    }
158
159    // Compute capabilities
160    let caps = ProviderCapabilities::detect(provider, model);
161
162    // Cache for future use
163    if let Ok(mut cache) = CAPABILITY_CACHE.write() {
164        cache.insert(cache_key, caps.clone());
165    }
166
167    caps
168}
169
170/// Universal LLM provider trait
171#[async_trait]
172pub trait LLMProvider: Send + Sync {
173    /// Provider name (e.g., "gemini", "openai", "anthropic")
174    fn name(&self) -> &str;
175
176    /// The canonical backend kind for this provider.
177    ///
178    /// Defaults to matching on [`name()`](LLMProvider::name) against the
179    /// well-known provider names. Providers should override this when their
180    /// name does not match the canonical mapping (e.g., dynamic names).
181    fn backend_kind(&self) -> BackendKind {
182        match self.name() {
183            "gemini" => BackendKind::Gemini,
184            "openai" => BackendKind::OpenAI,
185            "anthropic" => BackendKind::Anthropic,
186            "deepseek" => BackendKind::DeepSeek,
187            "meta" => BackendKind::Meta,
188            "mistral" => BackendKind::Mistral,
189            "openrouter" => BackendKind::OpenRouter,
190            "ollama" => BackendKind::Ollama,
191            "llamacpp" => BackendKind::LlamaCpp,
192            "zai" => BackendKind::ZAI,
193            "moonshot" => BackendKind::Moonshot,
194            "huggingface" => BackendKind::HuggingFace,
195            "minimax" => BackendKind::Minimax,
196            "mimo" => BackendKind::MiMo,
197            "opencode-zen" => BackendKind::OpenCodeZen,
198            "opencode-go" => BackendKind::OpenCodeGo,
199            "qwen" => BackendKind::Qwen,
200            "stepfun" => BackendKind::StepFun,
201            "evolink" => BackendKind::Evolink,
202            "poolside" => BackendKind::Poolside,
203            "nvidia" => BackendKind::Nvidia,
204            "merge-gateway" => BackendKind::MergeGateway,
205            "vercel" => BackendKind::Vercel,
206            _ => BackendKind::OpenAI,
207        }
208    }
209
210    /// Whether the provider has native streaming support
211    fn supports_streaming(&self) -> bool {
212        false
213    }
214
215    /// Whether the provider can service non-streaming generation requests for the model.
216    fn supports_non_streaming(&self, _model: &str) -> bool {
217        true
218    }
219
220    /// Whether the provider surfaces structured reasoning traces for the given model
221    fn supports_reasoning(&self, _model: &str) -> bool {
222        false
223    }
224
225    /// Whether the provider accepts configurable reasoning effort for the model
226    fn supports_reasoning_effort(&self, _model: &str) -> bool {
227        false
228    }
229
230    /// Exact levels accepted by this route; providers may override catalog metadata.
231    fn supported_reasoning_efforts(&self, model: &str) -> &'static [&'static str] {
232        if !self.supports_reasoning_effort(model) {
233            return &[];
234        }
235        catalog_or_generic_reasoning_efforts(self.name(), model)
236    }
237
238    /// Provider/model-specific sampling parameter overrides.
239    ///
240    /// Custom-provider profiles may pin exact sampling values per model; every
241    /// field defaults to `None`, meaning the agent loop's global config value
242    /// applies unchanged.
243    fn sampling_overrides(&self, _model: &str) -> SamplingOverrides {
244        SamplingOverrides::default()
245    }
246
247    /// Whether the provider supports structured tool calling for the given model
248    fn supports_tools(&self, _model: &str) -> bool {
249        true
250    }
251
252    /// Whether the provider understands parallel tool configuration payloads
253    fn supports_parallel_tool_config(&self, _model: &str) -> bool {
254        false
255    }
256
257    /// Whether the provider supports structured output (JSON schema guarantees)
258    fn supports_structured_output(&self, _model: &str) -> bool {
259        false
260    }
261
262    /// Whether the provider supports prompt/context caching
263    fn supports_context_caching(&self, _model: &str) -> bool {
264        false
265    }
266
267    /// Whether the provider supports vision (image analysis) for given model
268    fn supports_vision(&self, _model: &str) -> bool {
269        false
270    }
271
272    /// Whether the provider supports Responses API server-side compaction.
273    fn supports_responses_compaction(&self, _model: &str) -> bool {
274        false
275    }
276
277    /// Whether the request path can enforce a provider-native `allowed_tools` subset.
278    fn supports_native_allowed_tools(&self, _model: &str) -> bool {
279        false
280    }
281
282    /// Whether the provider supports provider-native context editing such as
283    /// tool-result clearing.
284    fn supports_context_edits(&self, _model: &str) -> bool {
285        false
286    }
287
288    /// Whether the selected provider/model can carry Anthropic's native
289    /// turn-scoped system-message lifecycle field on the wire.
290    ///
291    /// This is intentionally narrower than [`Self::supports_context_edits`]: a
292    /// provider can expose Anthropic-shaped requests without supporting the
293    /// `clear_at` field, and a provider name does not necessarily identify the
294    /// wire protocol for every model. The runtime keeps the typed marker in
295    /// canonical history either way, translating it to an ordinary system or
296    /// history directive when this capability is false.
297    fn supports_turn_scoped_system_messages(&self, _model: &str) -> bool {
298        false
299    }
300
301    /// Whether the provider supports the interactive manual `/compact` command path.
302    ///
303    /// This is narrower than general Responses compaction support and may exclude
304    /// compatible endpoints that do not match VT Code's native OpenAI UX contract.
305    fn supports_manual_openai_compaction(&self, _model: &str) -> bool {
306        false
307    }
308
309    /// Whether the provider supports threshold-triggered inline compaction via
310    /// request fields (Anthropic `compact_20260112`).
311    ///
312    /// This is distinct from [`supports_responses_compaction`](LLMProvider::supports_responses_compaction),
313    /// which is overloaded: OpenAI-compatible endpoints report it for their
314    /// standalone `/responses/compact` endpoint while Anthropic reports it for
315    /// inline compaction. Only the latter can be driven through `generate` with a
316    /// `compact_20260112` context-management edit, so the unified compaction
317    /// dispatch uses this method (not the overloaded flag) to pick the
318    /// `NativeInline` strategy and avoid sending an Anthropic-specific payload to
319    /// an OpenAI-compatible endpoint (which would only be rejected and fall back
320    /// to local summarization anyway).
321    fn supports_native_inline_compaction(&self, _model: &str) -> bool {
322        false
323    }
324
325    /// Explain why the `--native-only` manual `/compact` path is unavailable.
326    ///
327    /// This message only surfaces when the user explicitly passes `--native-only`
328    /// and the provider does not expose a native server-side compaction endpoint.
329    /// The plain `/compact` command is provider-agnostic and always falls back to
330    /// local summarization, so it is never refused on capability grounds.
331    fn manual_openai_compaction_unavailable_message(&self, model: &str) -> String {
332        format!(
333            "`--native-only` `/compact` requires a provider that exposes a native server-side compaction endpoint, which this provider does not. Active provider/model: {} / {}. Run `/compact` without `--native-only` to compact via the universal local summarization fallback.",
334            self.name(),
335            model,
336        )
337    }
338
339    /// Get the effective context window size for a model.
340    ///
341    /// Curated catalog capacity is the default for providers that do not have
342    /// a narrower route-specific limit. Adapters with an explicit endpoint
343    /// ceiling can still override this method and keep that ceiling intact.
344    fn effective_context_size(&self, model: &str) -> usize {
345        catalog_context_window(self.name(), model, 128_000)
346    }
347
348    /// Compact conversation history using provider-native Responses `/compact`
349    /// support when available.
350    async fn compact_history(&self, _model: &str, _history: &[Message]) -> Result<Vec<Message>, LLMError> {
351        Err(LLMError::Provider {
352            message: "Conversation compaction is not supported by this provider".to_string(),
353            metadata: None,
354        })
355    }
356
357    /// Compact conversation history with standalone Responses compaction options.
358    async fn compact_history_with_options(
359        &self,
360        _model: &str,
361        _history: &[Message],
362        _options: &ResponsesCompactionOptions,
363    ) -> Result<Vec<Message>, LLMError> {
364        Err(LLMError::Provider {
365            message: "manual OpenAI compaction is not supported by this provider".to_string(),
366            metadata: None,
367        })
368    }
369
370    /// Generate completion
371    async fn generate(&self, request: LLMRequest) -> Result<LLMResponse, LLMError>;
372
373    /// Stream completion (optional)
374    async fn stream(&self, request: LLMRequest) -> Result<LLMStream, LLMError> {
375        // Default implementation falls back to non-streaming
376        let response = self.generate(request).await?;
377        let stream = try_stream! {
378            yield LLMStreamEvent::Completed { response: Box::new(response) };
379        };
380        Ok(Box::pin(stream))
381    }
382
383    /// Normalized streaming contract layered on top of the legacy provider stream.
384    async fn stream_normalized(&self, request: LLMRequest) -> Result<LLMNormalizedStream, LLMError> {
385        let mut legacy_stream = self.stream(request).await?;
386        let stream = try_stream! {
387            while let Some(event) = futures::StreamExt::next(&mut legacy_stream).await {
388                for normalized in event?.into_normalized() {
389                    yield normalized;
390                }
391            }
392        };
393        Ok(Box::pin(stream))
394    }
395
396    /// Provider-specific streaming path that can service interactive runtime
397    /// requests while the stream is active. Copilot uses this to bridge ACP
398    /// tool calls and permission prompts back into VT Code's turn runtime.
399    #[cfg(feature = "copilot")]
400    fn start_copilot_prompt_session<'a>(
401        &'a self,
402        _request: LLMRequest,
403        _tools: &'a [super::ToolDefinition],
404    ) -> Option<crate::copilot::CopilotPromptSessionFuture<'a>> {
405        None
406    }
407
408    /// Get supported models
409    fn supported_models(&self) -> Vec<String>;
410
411    /// Fetch account balance for this provider, if supported.
412    async fn get_balance(&self) -> Result<Option<vtcode_commons::llm::BalanceInfo>, LLMError> {
413        Ok(None)
414    }
415
416    /// Validate request for this provider
417    fn validate_request(&self, request: &LLMRequest) -> Result<(), LLMError>;
418}
419
420/// Provider-local context capacity discovered for the selected model.
421/// All transport and capability behavior remains delegated to the underlying provider.
422///
423/// The `Box<dyn LLMProvider>` layer here is intentional: unlike
424/// `LlamaCppProvider`/`LmStudioProvider` (which always wrap `OpenAIProvider`
425/// and therefore store it concretely), the wrapped provider behind this type
426/// is selected at runtime and genuinely heterogeneous, so dynamic dispatch is
427/// required. `wrap` passes the box through untouched when no override applies,
428/// avoiding a second vtable layer.
429pub struct ContextWindowProvider {
430    inner: Box<dyn LLMProvider>,
431    model: CompactStr,
432    context_window: usize,
433}
434
435impl ContextWindowProvider {
436    /// Unknown/zero metadata preserves the backend's usable capacity.
437    pub fn wrap(inner: Box<dyn LLMProvider>, model: &str, context_window: Option<usize>) -> Box<dyn LLMProvider> {
438        match context_window.filter(|value| *value > 0) {
439            Some(context_window) => Box::new(Self { inner, model: model.into(), context_window }),
440            None => inner,
441        }
442    }
443}
444
445#[async_trait]
446impl LLMProvider for ContextWindowProvider {
447    fn name(&self) -> &str {
448        self.inner.name()
449    }
450
451    fn backend_kind(&self) -> BackendKind {
452        self.inner.backend_kind()
453    }
454
455    fn supports_streaming(&self) -> bool {
456        self.inner.supports_streaming()
457    }
458
459    fn supports_non_streaming(&self, model: &str) -> bool {
460        self.inner.supports_non_streaming(model)
461    }
462
463    fn supports_reasoning(&self, model: &str) -> bool {
464        self.inner.supports_reasoning(model)
465    }
466
467    fn supports_reasoning_effort(&self, model: &str) -> bool {
468        self.inner.supports_reasoning_effort(model)
469    }
470
471    fn supported_reasoning_efforts(&self, model: &str) -> &'static [&'static str] {
472        self.inner.supported_reasoning_efforts(model)
473    }
474
475    fn sampling_overrides(&self, model: &str) -> SamplingOverrides {
476        self.inner.sampling_overrides(model)
477    }
478
479    fn supports_tools(&self, model: &str) -> bool {
480        self.inner.supports_tools(model)
481    }
482
483    fn supports_parallel_tool_config(&self, model: &str) -> bool {
484        self.inner.supports_parallel_tool_config(model)
485    }
486
487    fn supports_structured_output(&self, model: &str) -> bool {
488        self.inner.supports_structured_output(model)
489    }
490
491    fn supports_context_caching(&self, model: &str) -> bool {
492        self.inner.supports_context_caching(model)
493    }
494
495    fn supports_vision(&self, model: &str) -> bool {
496        self.inner.supports_vision(model)
497    }
498
499    fn supports_responses_compaction(&self, model: &str) -> bool {
500        self.inner.supports_responses_compaction(model)
501    }
502
503    fn supports_native_allowed_tools(&self, model: &str) -> bool {
504        self.inner.supports_native_allowed_tools(model)
505    }
506
507    fn supports_context_edits(&self, model: &str) -> bool {
508        self.inner.supports_context_edits(model)
509    }
510
511    fn supports_turn_scoped_system_messages(&self, model: &str) -> bool {
512        self.inner.supports_turn_scoped_system_messages(model)
513    }
514
515    fn supports_manual_openai_compaction(&self, model: &str) -> bool {
516        self.inner.supports_manual_openai_compaction(model)
517    }
518
519    fn supports_native_inline_compaction(&self, model: &str) -> bool {
520        self.inner.supports_native_inline_compaction(model)
521    }
522
523    fn manual_openai_compaction_unavailable_message(&self, model: &str) -> String {
524        self.inner.manual_openai_compaction_unavailable_message(model)
525    }
526
527    fn effective_context_size(&self, model: &str) -> usize {
528        let requested_model = if model.trim().is_empty() {
529            self.model.as_str()
530        } else {
531            model
532        };
533        if requested_model == self.model.as_str() {
534            self.context_window
535        } else {
536            self.inner.effective_context_size(requested_model)
537        }
538    }
539
540    async fn compact_history(&self, model: &str, history: &[Message]) -> Result<Vec<Message>, LLMError> {
541        self.inner.compact_history(model, history).await
542    }
543
544    async fn compact_history_with_options(
545        &self,
546        model: &str,
547        history: &[Message],
548        options: &ResponsesCompactionOptions,
549    ) -> Result<Vec<Message>, LLMError> {
550        self.inner.compact_history_with_options(model, history, options).await
551    }
552
553    async fn generate(&self, request: LLMRequest) -> Result<LLMResponse, LLMError> {
554        self.inner.generate(request).await
555    }
556
557    async fn stream(&self, request: LLMRequest) -> Result<LLMStream, LLMError> {
558        self.inner.stream(request).await
559    }
560
561    async fn stream_normalized(&self, request: LLMRequest) -> Result<LLMNormalizedStream, LLMError> {
562        self.inner.stream_normalized(request).await
563    }
564
565    #[cfg(feature = "copilot")]
566    fn start_copilot_prompt_session<'a>(
567        &'a self,
568        request: LLMRequest,
569        tools: &'a [super::ToolDefinition],
570    ) -> Option<crate::copilot::CopilotPromptSessionFuture<'a>> {
571        self.inner.start_copilot_prompt_session(request, tools)
572    }
573
574    fn supported_models(&self) -> Vec<String> {
575        self.inner.supported_models()
576    }
577
578    async fn get_balance(&self) -> Result<Option<vtcode_commons::llm::BalanceInfo>, LLMError> {
579        self.inner.get_balance().await
580    }
581
582    fn validate_request(&self, request: &LLMRequest) -> Result<(), LLMError> {
583        self.inner.validate_request(request)
584    }
585}