Skip to main content

vtcode_llm/provider/
provider_trait.rs

1use async_stream::try_stream;
2use async_trait::async_trait;
3use compact_str::format_compact;
4use once_cell::sync::Lazy;
5use rustc_hash::FxHashMap;
6use std::sync::RwLock;
7use vtcode_commons::llm::BackendKind;
8
9use vtcode_commons::tool_types::CompactStr;
10
11use super::{
12    LLMNormalizedStream, LLMRequest, LLMResponse, LLMStream, LLMStreamEvent, Message, ResponsesCompactionOptions,
13    SamplingOverrides,
14};
15pub use vtcode_commons::llm::{LLMError, LLMErrorMetadata};
16
17/// Generic effort levels supported by providers that advertise configurable
18/// reasoning without exposing model-specific catalog metadata.
19pub(crate) const GENERIC_REASONING_EFFORTS: &[&str] = &["low", "medium", "high"];
20
21/// Return the catalog's effort metadata while preserving the distinction
22/// between an unknown route and a known route that only exposes structured
23/// reasoning. An empty catalog list is authoritative: it means the route does
24/// not accept a configurable effort payload.
25pub(crate) fn catalog_reasoning_efforts(provider: &str, model: &str) -> Option<&'static [&'static str]> {
26    vtcode_config::models::model_catalog_entry(provider, model).map(|entry| entry.reasoning_efforts)
27}
28
29/// Resolve the catalog's exact effort list while retaining the explicit
30/// generic contract for provider routes that advertise effort support without
31/// a built-in model entry (for example, an OpenAI-compatible custom endpoint).
32pub(crate) fn catalog_or_generic_reasoning_efforts(provider: &str, model: &str) -> &'static [&'static str] {
33    catalog_reasoning_efforts(provider, model).unwrap_or(GENERIC_REASONING_EFFORTS)
34}
35
36/// Resolve catalog levels while allowing an explicitly configured custom route
37/// to advertise the shared generic effort contract. Unsupported unknown routes
38/// remain empty so request builders fail closed.
39pub(crate) fn catalog_or_explicit_reasoning_efforts(
40    provider: &str,
41    model: &str,
42    explicitly_supported: bool,
43) -> &'static [&'static str] {
44    catalog_reasoning_efforts(provider, model)
45        .or_else(|| explicitly_supported.then_some(GENERIC_REASONING_EFFORTS))
46        .unwrap_or(&[])
47}
48
49/// Resolve a model's catalog context window with a provider-specific fallback.
50///
51/// Provider adapters use this helper instead of maintaining independent model
52/// name tables. Explicit provider overrides remain the caller's responsibility
53/// and are applied before this lookup.
54pub(crate) fn catalog_context_window(provider: &str, model: &str, fallback: usize) -> usize {
55    vtcode_config::models::model_catalog_entry(provider, model)
56        .map(|entry| entry.context_window)
57        .filter(|context_window| *context_window > 0)
58        .unwrap_or(fallback)
59}
60
61/// Cached provider capabilities to reduce repeated trait method calls
62#[derive(Debug, Clone)]
63pub struct ProviderCapabilities {
64    pub(crate) provider_name: String,
65    pub(crate) model: String,
66    pub streaming: bool,
67    pub reasoning: bool,
68    pub reasoning_effort: bool,
69    pub tools: bool,
70    pub parallel_tool_config: bool,
71    pub(crate) structured_output: bool,
72    pub(crate) context_caching: bool,
73    pub responses_compaction: bool,
74    pub context_edits: bool,
75    /// Whether the selected provider/model can carry Anthropic's native
76    /// turn-scoped system-message lifecycle field on the wire.
77    pub turn_scoped_system_messages: bool,
78    pub(crate) vision: bool,
79    pub(crate) context_size: usize,
80}
81
82impl ProviderCapabilities {
83    fn detect(provider: &dyn LLMProvider, model: &str) -> Self {
84        Self {
85            provider_name: provider.name().to_string(),
86            model: model.to_string(),
87            streaming: provider.supports_streaming(),
88            reasoning: provider.supports_reasoning(model),
89            reasoning_effort: !provider.supported_reasoning_efforts(model).is_empty(),
90            tools: provider.supports_tools(model),
91            parallel_tool_config: provider.supports_parallel_tool_config(model),
92            structured_output: provider.supports_structured_output(model),
93            context_caching: provider.supports_context_caching(model),
94            responses_compaction: provider.supports_responses_compaction(model),
95            context_edits: provider.supports_context_edits(model),
96            turn_scoped_system_messages: provider.supports_turn_scoped_system_messages(model),
97            vision: provider.supports_vision(model),
98            context_size: provider.effective_context_size(model),
99        }
100    }
101
102    pub(crate) fn has_advanced_features(&self) -> bool {
103        self.reasoning || self.structured_output || self.context_caching || self.reasoning_effort
104    }
105
106    pub(crate) fn summary(&self) -> String {
107        let mut features = Vec::new();
108
109        if self.streaming {
110            features.push("streaming");
111        }
112        if self.reasoning {
113            features.push("advanced-reasoning");
114        }
115        if self.reasoning_effort {
116            features.push("reasoning-effort");
117        }
118        if self.structured_output {
119            features.push("structured-output");
120        }
121        if self.context_caching {
122            features.push("context-caching");
123        }
124        if self.parallel_tool_config {
125            features.push("parallel-tools");
126        }
127        if self.responses_compaction {
128            features.push("responses-compaction");
129        }
130        if self.context_edits {
131            features.push("context-edits");
132        }
133
134        let features_str = if features.is_empty() {
135            "basic".to_string()
136        } else {
137            features.join(", ")
138        };
139
140        format!("{} ({} tokens): {}", self.model, self.context_size, features_str)
141    }
142}
143
144/// Global cache for provider capabilities (provider_name::model -> capabilities)
145static CAPABILITY_CACHE: Lazy<RwLock<FxHashMap<CompactStr, ProviderCapabilities>>> =
146    Lazy::new(|| RwLock::new(FxHashMap::default()));
147
148/// Extract and cache provider capabilities for a given provider and model
149pub fn get_cached_capabilities(provider: &dyn LLMProvider, model: &str) -> ProviderCapabilities {
150    let cache_key = format_compact!("{}::{}::{}", provider.name(), model, provider.effective_context_size(model));
151
152    // Check if already cached
153    if let Ok(cache) = CAPABILITY_CACHE.read()
154        && let Some(caps) = cache.get(&cache_key)
155    {
156        return caps.clone();
157    }
158
159    // Compute capabilities
160    let caps = ProviderCapabilities::detect(provider, model);
161
162    // Cache for future use
163    if let Ok(mut cache) = CAPABILITY_CACHE.write() {
164        cache.insert(cache_key, caps.clone());
165    }
166
167    caps
168}
169
170/// Universal LLM provider trait
171#[async_trait]
172pub trait LLMProvider: Send + Sync {
173    /// Provider name (e.g., "gemini", "openai", "anthropic")
174    fn name(&self) -> &str;
175
176    /// Whether this instance can use the direct API-key Decisions endpoint.
177    /// Capability checks must not discover credentials or dispatch requests.
178    fn supports_decisions(&self) -> bool {
179        false
180    }
181
182    /// Evaluate one text-only choice question. Unsupported providers do no I/O.
183    async fn decide_choice(
184        &self,
185        _request: super::ChoiceDecisionRequest,
186    ) -> Result<super::ChoiceDecisionResponse, LLMError> {
187        Err(LLMError::Provider {
188            message: "Decisions is unsupported for this provider".to_owned(),
189            metadata: None,
190        })
191    }
192
193    /// The canonical backend kind for this provider.
194    ///
195    /// Defaults to matching on [`name()`](LLMProvider::name) against the
196    /// well-known provider names. Providers should override this when their
197    /// name does not match the canonical mapping (e.g., dynamic names).
198    fn backend_kind(&self) -> BackendKind {
199        match self.name() {
200            "gemini" => BackendKind::Gemini,
201            "openai" => BackendKind::OpenAI,
202            "anthropic" => BackendKind::Anthropic,
203            "deepseek" => BackendKind::DeepSeek,
204            "meta" => BackendKind::Meta,
205            "mistral" => BackendKind::Mistral,
206            "openrouter" => BackendKind::OpenRouter,
207            "ollama" => BackendKind::Ollama,
208            "llamacpp" => BackendKind::LlamaCpp,
209            "zai" => BackendKind::ZAI,
210            "moonshot" => BackendKind::Moonshot,
211            "huggingface" => BackendKind::HuggingFace,
212            "minimax" => BackendKind::Minimax,
213            "mimo" => BackendKind::MiMo,
214            "opencode-zen" => BackendKind::OpenCodeZen,
215            "opencode-go" => BackendKind::OpenCodeGo,
216            "qwen" => BackendKind::Qwen,
217            "stepfun" => BackendKind::StepFun,
218            "evolink" => BackendKind::Evolink,
219            "poolside" => BackendKind::Poolside,
220            "nvidia" => BackendKind::Nvidia,
221            "merge-gateway" => BackendKind::MergeGateway,
222            "vercel" => BackendKind::Vercel,
223            _ => BackendKind::OpenAI,
224        }
225    }
226
227    /// Whether the provider has native streaming support
228    fn supports_streaming(&self) -> bool {
229        false
230    }
231
232    /// Whether the provider can service non-streaming generation requests for the model.
233    fn supports_non_streaming(&self, _model: &str) -> bool {
234        true
235    }
236
237    /// Whether the provider surfaces structured reasoning traces for the given model
238    fn supports_reasoning(&self, _model: &str) -> bool {
239        false
240    }
241
242    /// Whether the provider accepts configurable reasoning effort for the model
243    fn supports_reasoning_effort(&self, _model: &str) -> bool {
244        false
245    }
246
247    /// Exact levels accepted by this route; providers may override catalog metadata.
248    fn supported_reasoning_efforts(&self, model: &str) -> &'static [&'static str] {
249        if !self.supports_reasoning_effort(model) {
250            return &[];
251        }
252        catalog_or_generic_reasoning_efforts(self.name(), model)
253    }
254
255    /// Provider/model-specific sampling parameter overrides.
256    ///
257    /// Custom-provider profiles may pin exact sampling values per model; every
258    /// field defaults to `None`, meaning the agent loop's global config value
259    /// applies unchanged.
260    fn sampling_overrides(&self, _model: &str) -> SamplingOverrides {
261        SamplingOverrides::default()
262    }
263
264    /// Whether the provider supports structured tool calling for the given model
265    fn supports_tools(&self, _model: &str) -> bool {
266        true
267    }
268
269    /// Whether the provider understands parallel tool configuration payloads
270    fn supports_parallel_tool_config(&self, _model: &str) -> bool {
271        false
272    }
273
274    /// Whether the provider supports structured output (JSON schema guarantees)
275    fn supports_structured_output(&self, _model: &str) -> bool {
276        false
277    }
278
279    /// Whether the provider supports prompt/context caching
280    fn supports_context_caching(&self, _model: &str) -> bool {
281        false
282    }
283
284    /// Whether the provider supports vision (image analysis) for given model
285    fn supports_vision(&self, _model: &str) -> bool {
286        false
287    }
288
289    /// Whether the provider supports Responses API server-side compaction.
290    fn supports_responses_compaction(&self, _model: &str) -> bool {
291        false
292    }
293
294    /// Whether the request path can enforce a provider-native `allowed_tools` subset.
295    fn supports_native_allowed_tools(&self, _model: &str) -> bool {
296        false
297    }
298
299    /// Whether the provider supports provider-native context editing such as
300    /// tool-result clearing.
301    fn supports_context_edits(&self, _model: &str) -> bool {
302        false
303    }
304
305    /// Whether the selected provider/model can carry Anthropic's native
306    /// turn-scoped system-message lifecycle field on the wire.
307    ///
308    /// This is intentionally narrower than [`Self::supports_context_edits`]: a
309    /// provider can expose Anthropic-shaped requests without supporting the
310    /// `clear_at` field, and a provider name does not necessarily identify the
311    /// wire protocol for every model. The runtime keeps the typed marker in
312    /// canonical history either way, translating it to an ordinary system or
313    /// history directive when this capability is false.
314    fn supports_turn_scoped_system_messages(&self, _model: &str) -> bool {
315        false
316    }
317
318    /// Whether the provider supports the interactive manual `/compact` command path.
319    ///
320    /// This is narrower than general Responses compaction support and may exclude
321    /// compatible endpoints that do not match VT Code's native OpenAI UX contract.
322    fn supports_manual_openai_compaction(&self, _model: &str) -> bool {
323        false
324    }
325
326    /// Whether the provider supports threshold-triggered inline compaction via
327    /// request fields (Anthropic `compact_20260112`).
328    ///
329    /// This is distinct from [`supports_responses_compaction`](LLMProvider::supports_responses_compaction),
330    /// which is overloaded: OpenAI-compatible endpoints report it for their
331    /// standalone `/responses/compact` endpoint while Anthropic reports it for
332    /// inline compaction. Only the latter can be driven through `generate` with a
333    /// `compact_20260112` context-management edit, so the unified compaction
334    /// dispatch uses this method (not the overloaded flag) to pick the
335    /// `NativeInline` strategy and avoid sending an Anthropic-specific payload to
336    /// an OpenAI-compatible endpoint (which would only be rejected and fall back
337    /// to local summarization anyway).
338    fn supports_native_inline_compaction(&self, _model: &str) -> bool {
339        false
340    }
341
342    /// Explain why the `--native-only` manual `/compact` path is unavailable.
343    ///
344    /// This message only surfaces when the user explicitly passes `--native-only`
345    /// and the provider does not expose a native server-side compaction endpoint.
346    /// The plain `/compact` command is provider-agnostic and always falls back to
347    /// local summarization, so it is never refused on capability grounds.
348    fn manual_openai_compaction_unavailable_message(&self, model: &str) -> String {
349        format!(
350            "`--native-only` `/compact` requires a provider that exposes a native server-side compaction endpoint, which this provider does not. Active provider/model: {} / {}. Run `/compact` without `--native-only` to compact via the universal local summarization fallback.",
351            self.name(),
352            model,
353        )
354    }
355
356    /// Get the effective context window size for a model.
357    ///
358    /// Curated catalog capacity is the default for providers that do not have
359    /// a narrower route-specific limit. Adapters with an explicit endpoint
360    /// ceiling can still override this method and keep that ceiling intact.
361    fn effective_context_size(&self, model: &str) -> usize {
362        catalog_context_window(self.name(), model, 128_000)
363    }
364
365    /// Compact conversation history using provider-native Responses `/compact`
366    /// support when available.
367    async fn compact_history(&self, _model: &str, _history: &[Message]) -> Result<Vec<Message>, LLMError> {
368        Err(LLMError::Provider {
369            message: "Conversation compaction is not supported by this provider".to_string(),
370            metadata: None,
371        })
372    }
373
374    /// Compact conversation history with standalone Responses compaction options.
375    async fn compact_history_with_options(
376        &self,
377        _model: &str,
378        _history: &[Message],
379        _options: &ResponsesCompactionOptions,
380    ) -> Result<Vec<Message>, LLMError> {
381        Err(LLMError::Provider {
382            message: "manual OpenAI compaction is not supported by this provider".to_string(),
383            metadata: None,
384        })
385    }
386
387    /// Generate completion
388    async fn generate(&self, request: LLMRequest) -> Result<LLMResponse, LLMError>;
389
390    /// Stream completion (optional)
391    async fn stream(&self, request: LLMRequest) -> Result<LLMStream, LLMError> {
392        // Default implementation falls back to non-streaming
393        let response = self.generate(request).await?;
394        let stream = try_stream! {
395            yield LLMStreamEvent::Completed { response: Box::new(response) };
396        };
397        Ok(Box::pin(stream))
398    }
399
400    /// Normalized streaming contract layered on top of the legacy provider stream.
401    async fn stream_normalized(&self, request: LLMRequest) -> Result<LLMNormalizedStream, LLMError> {
402        let mut legacy_stream = self.stream(request).await?;
403        let stream = try_stream! {
404            while let Some(event) = futures::StreamExt::next(&mut legacy_stream).await {
405                for normalized in event?.into_normalized() {
406                    yield normalized;
407                }
408            }
409        };
410        Ok(Box::pin(stream))
411    }
412
413    /// Provider-specific streaming path that can service interactive runtime
414    /// requests while the stream is active. Copilot uses this to bridge ACP
415    /// tool calls and permission prompts back into VT Code's turn runtime.
416    #[cfg(feature = "copilot")]
417    fn start_copilot_prompt_session<'a>(
418        &'a self,
419        _request: LLMRequest,
420        _tools: &'a [super::ToolDefinition],
421    ) -> Option<crate::copilot::CopilotPromptSessionFuture<'a>> {
422        None
423    }
424
425    /// Get supported models
426    fn supported_models(&self) -> Vec<String>;
427
428    /// Fetch account balance for this provider, if supported.
429    async fn get_balance(&self) -> Result<Option<vtcode_commons::llm::BalanceInfo>, LLMError> {
430        Ok(None)
431    }
432
433    /// Validate request for this provider
434    fn validate_request(&self, request: &LLMRequest) -> Result<(), LLMError>;
435}
436
437/// Provider-local context capacity discovered for the selected model.
438/// All transport and capability behavior remains delegated to the underlying provider.
439///
440/// The `Box<dyn LLMProvider>` layer here is intentional: unlike
441/// `LlamaCppProvider`/`LmStudioProvider` (which always wrap `OpenAIProvider`
442/// and therefore store it concretely), the wrapped provider behind this type
443/// is selected at runtime and genuinely heterogeneous, so dynamic dispatch is
444/// required. `wrap` passes the box through untouched when no override applies,
445/// avoiding a second vtable layer.
446pub struct ContextWindowProvider {
447    inner: Box<dyn LLMProvider>,
448    model: CompactStr,
449    context_window: usize,
450}
451
452impl ContextWindowProvider {
453    /// Unknown/zero metadata preserves the backend's usable capacity.
454    pub fn wrap(inner: Box<dyn LLMProvider>, model: &str, context_window: Option<usize>) -> Box<dyn LLMProvider> {
455        match context_window.filter(|value| *value > 0) {
456            Some(context_window) => Box::new(Self { inner, model: model.into(), context_window }),
457            None => inner,
458        }
459    }
460}
461
462#[async_trait]
463impl LLMProvider for ContextWindowProvider {
464    fn name(&self) -> &str {
465        self.inner.name()
466    }
467
468    fn supports_decisions(&self) -> bool {
469        self.inner.supports_decisions()
470    }
471
472    async fn decide_choice(
473        &self,
474        request: super::ChoiceDecisionRequest,
475    ) -> Result<super::ChoiceDecisionResponse, LLMError> {
476        self.inner.decide_choice(request).await
477    }
478
479    fn backend_kind(&self) -> BackendKind {
480        self.inner.backend_kind()
481    }
482
483    fn supports_streaming(&self) -> bool {
484        self.inner.supports_streaming()
485    }
486
487    fn supports_non_streaming(&self, model: &str) -> bool {
488        self.inner.supports_non_streaming(model)
489    }
490
491    fn supports_reasoning(&self, model: &str) -> bool {
492        self.inner.supports_reasoning(model)
493    }
494
495    fn supports_reasoning_effort(&self, model: &str) -> bool {
496        self.inner.supports_reasoning_effort(model)
497    }
498
499    fn supported_reasoning_efforts(&self, model: &str) -> &'static [&'static str] {
500        self.inner.supported_reasoning_efforts(model)
501    }
502
503    fn sampling_overrides(&self, model: &str) -> SamplingOverrides {
504        self.inner.sampling_overrides(model)
505    }
506
507    fn supports_tools(&self, model: &str) -> bool {
508        self.inner.supports_tools(model)
509    }
510
511    fn supports_parallel_tool_config(&self, model: &str) -> bool {
512        self.inner.supports_parallel_tool_config(model)
513    }
514
515    fn supports_structured_output(&self, model: &str) -> bool {
516        self.inner.supports_structured_output(model)
517    }
518
519    fn supports_context_caching(&self, model: &str) -> bool {
520        self.inner.supports_context_caching(model)
521    }
522
523    fn supports_vision(&self, model: &str) -> bool {
524        self.inner.supports_vision(model)
525    }
526
527    fn supports_responses_compaction(&self, model: &str) -> bool {
528        self.inner.supports_responses_compaction(model)
529    }
530
531    fn supports_native_allowed_tools(&self, model: &str) -> bool {
532        self.inner.supports_native_allowed_tools(model)
533    }
534
535    fn supports_context_edits(&self, model: &str) -> bool {
536        self.inner.supports_context_edits(model)
537    }
538
539    fn supports_turn_scoped_system_messages(&self, model: &str) -> bool {
540        self.inner.supports_turn_scoped_system_messages(model)
541    }
542
543    fn supports_manual_openai_compaction(&self, model: &str) -> bool {
544        self.inner.supports_manual_openai_compaction(model)
545    }
546
547    fn supports_native_inline_compaction(&self, model: &str) -> bool {
548        self.inner.supports_native_inline_compaction(model)
549    }
550
551    fn manual_openai_compaction_unavailable_message(&self, model: &str) -> String {
552        self.inner.manual_openai_compaction_unavailable_message(model)
553    }
554
555    fn effective_context_size(&self, model: &str) -> usize {
556        let requested_model = if model.trim().is_empty() {
557            self.model.as_str()
558        } else {
559            model
560        };
561        if requested_model == self.model.as_str() {
562            self.context_window
563        } else {
564            self.inner.effective_context_size(requested_model)
565        }
566    }
567
568    async fn compact_history(&self, model: &str, history: &[Message]) -> Result<Vec<Message>, LLMError> {
569        self.inner.compact_history(model, history).await
570    }
571
572    async fn compact_history_with_options(
573        &self,
574        model: &str,
575        history: &[Message],
576        options: &ResponsesCompactionOptions,
577    ) -> Result<Vec<Message>, LLMError> {
578        self.inner.compact_history_with_options(model, history, options).await
579    }
580
581    async fn generate(&self, request: LLMRequest) -> Result<LLMResponse, LLMError> {
582        self.inner.generate(request).await
583    }
584
585    async fn stream(&self, request: LLMRequest) -> Result<LLMStream, LLMError> {
586        self.inner.stream(request).await
587    }
588
589    async fn stream_normalized(&self, request: LLMRequest) -> Result<LLMNormalizedStream, LLMError> {
590        self.inner.stream_normalized(request).await
591    }
592
593    #[cfg(feature = "copilot")]
594    fn start_copilot_prompt_session<'a>(
595        &'a self,
596        request: LLMRequest,
597        tools: &'a [super::ToolDefinition],
598    ) -> Option<crate::copilot::CopilotPromptSessionFuture<'a>> {
599        self.inner.start_copilot_prompt_session(request, tools)
600    }
601
602    fn supported_models(&self) -> Vec<String> {
603        self.inner.supported_models()
604    }
605
606    async fn get_balance(&self) -> Result<Option<vtcode_commons::llm::BalanceInfo>, LLMError> {
607        self.inner.get_balance().await
608    }
609
610    fn validate_request(&self, request: &LLMRequest) -> Result<(), LLMError> {
611        self.inner.validate_request(request)
612    }
613}