vtcode_llm/providers/
local_readiness.rs1use std::collections::HashMap;
20use std::sync::{LazyLock, Mutex};
21use std::time::{Duration, Instant};
22
23use vtcode_commons::llm::LLMError;
24
25use super::llamacpp::fetch_llamacpp_models;
26use super::lmstudio::fetch_lmstudio_models;
27use super::local_server::{LocalProvider, probe};
28use super::ollama::fetch_ollama_models;
29
30const READINESS_CACHE_TTL: Duration = Duration::from_secs(15);
31
32struct CacheEntry {
33 verified_at: Instant,
34 models: Vec<String>,
35}
36
37static READINESS_CACHE: LazyLock<Mutex<HashMap<LocalProvider, CacheEntry>>> =
38 LazyLock::new(|| Mutex::new(HashMap::new()));
39
40#[derive(Debug, Clone, PartialEq, Eq)]
42pub enum LocalReadinessError {
43 ServerDown { provider: LocalProvider },
45 ModelMissing { provider: LocalProvider, model: String },
47}
48
49impl LocalReadinessError {
50 fn code(&self) -> &'static str {
53 match self {
54 Self::ServerDown { .. } => "local_server_down",
55 Self::ModelMissing { .. } => "local_model_missing",
56 }
57 }
58
59 fn fix_command(&self) -> String {
61 match self {
62 Self::ServerDown { provider } => format!("/local start {}", provider.key()),
63 Self::ModelMissing { provider, model } => match provider {
64 LocalProvider::Ollama => format!("ollama pull {model}"),
65 LocalProvider::LmStudio => format!("lms load {model}"),
66 LocalProvider::LlamaCpp => {
67 format!("load '{model}' in llama.cpp (set LLAMACPP_MODEL_PATH and run /local start llamacpp)")
68 }
69 },
70 }
71 }
72
73 fn recovery_hint(&self) -> String {
75 match self {
76 Self::ServerDown { provider } => format!(
77 "{} server is not running. Start it with `/local start {}` (or the app/CLI), \
78 then retry.",
79 provider.display_name(),
80 provider.key()
81 ),
82 Self::ModelMissing { provider, model } => {
83 format!("Model '{model}' is not available on {}. Fix: {}", provider.display_name(), self.fix_command())
84 }
85 }
86 }
87
88 pub(crate) fn to_llm_error(&self, display: &str) -> LLMError {
91 LLMError::Provider {
92 message: self.recovery_hint(),
93 metadata: Some(vtcode_commons::llm::LLMErrorMetadata::new(
94 display,
95 None,
96 Some(self.code().to_string()),
97 None,
98 None,
99 None,
100 None,
101 )),
102 }
103 }
104}
105
106fn is_cloud_ollama_model(model: &str) -> bool {
107 model.contains(":cloud") || model.contains("-cloud")
108}
109
110pub(crate) async fn resolve_local_model(
119 provider: LocalProvider,
120 requested: &str,
121 base_url: Option<&str>,
122) -> Result<String, LocalReadinessError> {
123 if provider == LocalProvider::Ollama && is_cloud_ollama_model(requested) {
124 return Ok(requested.to_string());
125 }
126
127 let status = probe(provider).await;
128 if !status.running {
129 return Err(LocalReadinessError::ServerDown { provider });
130 }
131
132 let models = cached_models(provider, base_url).await;
133
134 match models {
135 Some(list) if !list.is_empty() => {
136 if list.iter().any(|m| m == requested) {
137 return Ok(requested.to_string());
138 }
139 if requested.trim().is_empty() && list.len() == 1 {
140 return Ok(list[0].clone());
141 }
142 Err(LocalReadinessError::ModelMissing { provider, model: requested.to_string() })
143 }
144 _ => Ok(requested.to_string()),
147 }
148}
149
150async fn cached_models(provider: LocalProvider, base_url: Option<&str>) -> Option<Vec<String>> {
151 {
152 let guard = READINESS_CACHE.lock().unwrap_or_else(|e| e.into_inner());
153 if let Some(entry) = guard.get(&provider)
154 && entry.verified_at.elapsed() < READINESS_CACHE_TTL
155 {
156 return Some(entry.models.clone());
157 }
158 }
159
160 let fetched = fetch_models(provider, base_url).await;
161 if let Ok(models) = fetched {
162 if let Ok(mut guard) = READINESS_CACHE.lock() {
163 guard.insert(
164 provider,
165 CacheEntry {
166 verified_at: Instant::now(),
167 models: models.clone(),
168 },
169 );
170 }
171 Some(models)
172 } else {
173 None
174 }
175}
176
177async fn fetch_models(provider: LocalProvider, base_url: Option<&str>) -> anyhow::Result<Vec<String>> {
178 let base = base_url.map(str::to_string);
179 match provider {
180 LocalProvider::Ollama => fetch_ollama_models(base).await,
181 LocalProvider::LmStudio => fetch_lmstudio_models(base).await,
182 LocalProvider::LlamaCpp => fetch_llamacpp_models(base).await,
183 }
184}
185
186pub(crate) fn invalidate_readiness_cache() {
189 if let Ok(mut guard) = READINESS_CACHE.lock() {
190 guard.clear();
191 }
192}
193
194#[cfg(test)]
195mod tests {
196 use super::*;
197
198 #[test]
199 fn fix_command_for_server_down() {
200 let err = LocalReadinessError::ServerDown { provider: LocalProvider::Ollama };
201 assert_eq!(err.code(), "local_server_down");
202 assert_eq!(err.fix_command(), "/local start ollama");
203 }
204
205 #[test]
206 fn fix_command_for_missing_model() {
207 let ollama = LocalReadinessError::ModelMissing {
208 provider: LocalProvider::Ollama,
209 model: "gpt-oss:20b".to_string(),
210 };
211 assert_eq!(ollama.fix_command(), "ollama pull gpt-oss:20b");
212 assert_eq!(ollama.code(), "local_model_missing");
213
214 let lm = LocalReadinessError::ModelMissing {
215 provider: LocalProvider::LmStudio,
216 model: "my-model".to_string(),
217 };
218 assert_eq!(lm.fix_command(), "lms load my-model");
219 }
220
221 #[test]
222 fn cloud_ollama_models_bypass_check() {
223 assert!(is_cloud_ollama_model("deepseek-flash:cloud"));
224 assert!(is_cloud_ollama_model("glm-5.2-cloud"));
225 assert!(!is_cloud_ollama_model("gpt-oss:20b"));
226 }
227}