Skip to main content

vtcode_llm/providers/
local_readiness.rs

1//! Unified readiness pre-flight for local inference providers.
2//!
3//! Local servers (Ollama, LM Studio, llama.cpp) are frequently stopped or have
4//! no model loaded. Previously generation simply failed with a raw connection
5//! or 404 error. This module centralizes the "is the server up?" and "is the
6//! requested model available?" checks so every local provider can return a
7//! single, actionable error (with the exact fix command) instead of a cryptic
8//! one. It also resolves a placeholder/default request model to the single
9//! loaded model when appropriate.
10//!
11//! Design notes:
12//! - Cloud Ollama models (`:cloud` / `-cloud`) are remote and bypass readiness.
13//! - Results are cached per-process for a short TTL so we do not probe the
14//!   local server on every single generation, while still allowing a
15//!   `ServerDown` error to clear quickly after the user starts the server.
16//! - This module is intentionally provider-agnostic: it only consumes the
17//!   `LocalProvider` enum and the per-provider `fetch_*_models` helpers.
18
19use std::collections::HashMap;
20use std::sync::{LazyLock, Mutex};
21use std::time::{Duration, Instant};
22
23use vtcode_commons::llm::LLMError;
24
25use super::llamacpp::fetch_llamacpp_models;
26use super::lmstudio::fetch_lmstudio_models;
27use super::local_server::{LocalProvider, probe};
28use super::ollama::fetch_ollama_models;
29
30const READINESS_CACHE_TTL: Duration = Duration::from_secs(15);
31
32struct CacheEntry {
33    verified_at: Instant,
34    models: Vec<String>,
35}
36
37static READINESS_CACHE: LazyLock<Mutex<HashMap<LocalProvider, CacheEntry>>> =
38    LazyLock::new(|| Mutex::new(HashMap::new()));
39
40/// Failure modes surfaced before a generation attempt.
41#[derive(Debug, Clone, PartialEq, Eq)]
42pub enum LocalReadinessError {
43    /// The local server process is not reachable.
44    ServerDown { provider: LocalProvider },
45    /// The server is up but the requested model is not available/loaded.
46    ModelMissing { provider: LocalProvider, model: String },
47}
48
49impl LocalReadinessError {
50    /// Stable tag stored in `LLMErrorMetadata.code` so the runloop can render
51    /// an interactive "start server" / "pull model" offer.
52    fn code(&self) -> &'static str {
53        match self {
54            Self::ServerDown { .. } => "local_server_down",
55            Self::ModelMissing { .. } => "local_model_missing",
56        }
57    }
58
59    /// The exact command/instruction the user needs to run to recover.
60    fn fix_command(&self) -> String {
61        match self {
62            Self::ServerDown { provider } => format!("/local start {}", provider.key()),
63            Self::ModelMissing { provider, model } => match provider {
64                LocalProvider::Ollama => format!("ollama pull {model}"),
65                LocalProvider::LmStudio => format!("lms load {model}"),
66                LocalProvider::LlamaCpp => {
67                    format!("load '{model}' in llama.cpp (set LLAMACPP_MODEL_PATH and run /local start llamacpp)")
68                }
69            },
70        }
71    }
72
73    /// Human-readable recovery instruction (used in logs/troubleshooting).
74    fn recovery_hint(&self) -> String {
75        match self {
76            Self::ServerDown { provider } => format!(
77                "{} server is not running. Start it with `/local start {}` (or the app/CLI), \
78                 then retry.",
79                provider.display_name(),
80                provider.key()
81            ),
82            Self::ModelMissing { provider, model } => {
83                format!("Model '{model}' is not available on {}. Fix: {}", provider.display_name(), self.fix_command())
84            }
85        }
86    }
87
88    /// Convert into a structured `LLMError` carrying the recovery code so the
89    /// runloop can offer the user a one-tap fix.
90    pub(crate) fn to_llm_error(&self, display: &str) -> LLMError {
91        LLMError::Provider {
92            message: self.recovery_hint(),
93            metadata: Some(vtcode_commons::llm::LLMErrorMetadata::new(
94                display,
95                None,
96                Some(self.code().to_string()),
97                None,
98                None,
99                None,
100                None,
101            )),
102        }
103    }
104}
105
106fn is_cloud_ollama_model(model: &str) -> bool {
107    model.contains(":cloud") || model.contains("-cloud")
108}
109
110/// Resolve the model that should actually be used for the request.
111///
112/// Returns the (possibly substituted) model id, or a [`LocalReadinessError`]
113/// describing what is wrong. An explicit, non-empty `requested` model that is
114/// not available is always reported as missing (with the recovery command) —
115/// we never silently swap an explicit selection for a different loaded model.
116/// Only when *no* model is specified (empty request) do we fall back to the
117/// single loaded model.
118pub(crate) async fn resolve_local_model(
119    provider: LocalProvider,
120    requested: &str,
121    base_url: Option<&str>,
122) -> Result<String, LocalReadinessError> {
123    if provider == LocalProvider::Ollama && is_cloud_ollama_model(requested) {
124        return Ok(requested.to_string());
125    }
126
127    let status = probe(provider).await;
128    if !status.running {
129        return Err(LocalReadinessError::ServerDown { provider });
130    }
131
132    let models = cached_models(provider, base_url).await;
133
134    match models {
135        Some(list) if !list.is_empty() => {
136            if list.iter().any(|m| m == requested) {
137                return Ok(requested.to_string());
138            }
139            if requested.trim().is_empty() && list.len() == 1 {
140                return Ok(list[0].clone());
141            }
142            Err(LocalReadinessError::ModelMissing { provider, model: requested.to_string() })
143        }
144        // Server is up but we could not enumerate models: trust the request id
145        // rather than blocking the user with a false negative.
146        _ => Ok(requested.to_string()),
147    }
148}
149
150async fn cached_models(provider: LocalProvider, base_url: Option<&str>) -> Option<Vec<String>> {
151    {
152        let guard = READINESS_CACHE.lock().unwrap_or_else(|e| e.into_inner());
153        if let Some(entry) = guard.get(&provider)
154            && entry.verified_at.elapsed() < READINESS_CACHE_TTL
155        {
156            return Some(entry.models.clone());
157        }
158    }
159
160    let fetched = fetch_models(provider, base_url).await;
161    if let Ok(models) = fetched {
162        if let Ok(mut guard) = READINESS_CACHE.lock() {
163            guard.insert(
164                provider,
165                CacheEntry {
166                    verified_at: Instant::now(),
167                    models: models.clone(),
168                },
169            );
170        }
171        Some(models)
172    } else {
173        None
174    }
175}
176
177async fn fetch_models(provider: LocalProvider, base_url: Option<&str>) -> anyhow::Result<Vec<String>> {
178    let base = base_url.map(str::to_string);
179    match provider {
180        LocalProvider::Ollama => fetch_ollama_models(base).await,
181        LocalProvider::LmStudio => fetch_lmstudio_models(base).await,
182        LocalProvider::LlamaCpp => fetch_llamacpp_models(base).await,
183    }
184}
185
186/// Clear the cached readiness state (used by tests and after a server lifecycle
187/// change so a subsequent generation re-probes immediately).
188pub(crate) fn invalidate_readiness_cache() {
189    if let Ok(mut guard) = READINESS_CACHE.lock() {
190        guard.clear();
191    }
192}
193
194#[cfg(test)]
195mod tests {
196    use super::*;
197
198    #[test]
199    fn fix_command_for_server_down() {
200        let err = LocalReadinessError::ServerDown { provider: LocalProvider::Ollama };
201        assert_eq!(err.code(), "local_server_down");
202        assert_eq!(err.fix_command(), "/local start ollama");
203    }
204
205    #[test]
206    fn fix_command_for_missing_model() {
207        let ollama = LocalReadinessError::ModelMissing {
208            provider: LocalProvider::Ollama,
209            model: "gpt-oss:20b".to_string(),
210        };
211        assert_eq!(ollama.fix_command(), "ollama pull gpt-oss:20b");
212        assert_eq!(ollama.code(), "local_model_missing");
213
214        let lm = LocalReadinessError::ModelMissing {
215            provider: LocalProvider::LmStudio,
216            model: "my-model".to_string(),
217        };
218        assert_eq!(lm.fix_command(), "lms load my-model");
219    }
220
221    #[test]
222    fn cloud_ollama_models_bypass_check() {
223        assert!(is_cloud_ollama_model("deepseek-flash:cloud"));
224        assert!(is_cloud_ollama_model("glm-5.2-cloud"));
225        assert!(!is_cloud_ollama_model("gpt-oss:20b"));
226    }
227}