Skip to main content

vtcode_core/pods/
catalog.rs

1use super::state::PodState;
2use serde::{Deserialize, Serialize};
3use std::collections::BTreeMap;
4
5/// Root catalog describing known deployment profiles.
6#[derive(Debug, Clone, Serialize, Deserialize)]
7pub struct PodCatalog {
8    pub version: String,
9    #[serde(default)]
10    pub profiles: Vec<PodProfile>,
11}
12
13impl Default for PodCatalog {
14    fn default() -> Self {
15        Self::embedded_default()
16    }
17}
18
19/// A single deployment profile for a model.
20#[derive(Debug, Clone, Serialize, Deserialize)]
21pub struct PodProfile {
22    /// Short identifier for this profile.
23    pub name: String,
24    /// Hugging Face model identifier.
25    pub model: String,
26    /// Number of GPUs required by this profile.
27    pub gpu_count: usize,
28    /// Optional GPU type constraints (substring-matched against GPU names).
29    #[serde(default)]
30    pub gpu_types: Vec<String>,
31    /// Command template with `{{MODEL_ID}}`, `{{NAME}}`, `{{PORT}}`, and `{{VLLM_ARGS}}` placeholders.
32    #[serde(default = "default_command_template")]
33    pub command_template: String,
34    /// Extra command-line arguments passed to vLLM.
35    #[serde(default)]
36    pub vllm_args: Vec<String>,
37    /// Environment variables set before launching the server.
38    #[serde(default)]
39    pub env: BTreeMap<String, String>,
40}
41
42impl PodCatalog {
43    /// Return the built-in catalog compiled into the binary.
44    pub fn embedded_default() -> Self {
45        match serde_json::from_str(include_str!("default_catalog.json")) {
46            Ok(catalog) => catalog,
47            Err(_) => Self {
48                version: "2".to_string(),
49                profiles: vec![PodProfile {
50                    name: "qwen3-30b-a3b".to_string(),
51                    model: "Qwen/Qwen3-30B-A3B".to_string(),
52                    gpu_count: 1,
53                    gpu_types: vec![],
54                    command_template: default_command_template(),
55                    vllm_args: vec![
56                        "--trust-remote-code".to_string(),
57                        "--dtype".to_string(),
58                        "bfloat16".to_string(),
59                        "--gpu-memory-utilization".to_string(),
60                        "0.90".to_string(),
61                        "--max-model-len".to_string(),
62                        "32768".to_string(),
63                    ],
64                    env: BTreeMap::new(),
65                }],
66            },
67        }
68    }
69
70    /// Return all profiles whose name or model field matches `model`.
71    pub fn profiles_for_model(&self, model: &str) -> Vec<&PodProfile> {
72        self.profiles
73            .iter()
74            .filter(|profile| profile.name == model || profile.model == model)
75            .collect()
76    }
77
78    /// Split all profiles into those compatible with `pod` and those that are not.
79    pub fn compatible_profiles<'a>(&'a self, pod: &PodState) -> (Vec<&'a PodProfile>, Vec<&'a PodProfile>) {
80        let mut compatible = Vec::new();
81        let mut incompatible = Vec::new();
82
83        for profile in &self.profiles {
84            if profile.matches_pod(pod) {
85                compatible.push(profile);
86            } else {
87                incompatible.push(profile);
88            }
89        }
90
91        (compatible, incompatible)
92    }
93}
94
95impl PodProfile {
96    /// Return `true` if `pod` has enough GPUs of the required type for this profile.
97    pub fn matches_pod(&self, pod: &PodState) -> bool {
98        if self.gpu_count > pod.gpu_count() {
99            return false;
100        }
101
102        if self.gpu_types.is_empty() {
103            return true;
104        }
105
106        let gpu_types = self
107            .gpu_types
108            .iter()
109            .map(|gpu_type| gpu_type.to_lowercase())
110            .collect::<Vec<_>>();
111
112        pod.gpus
113            .iter()
114            .filter(|gpu| {
115                let gpu_name = gpu.name.to_lowercase();
116                gpu_types.iter().any(|gpu_type| gpu_name.contains(gpu_type))
117            })
118            .take(self.gpu_count)
119            .count()
120            >= self.gpu_count
121    }
122
123    /// Return `true` if the profile requires exactly `count` GPUs.
124    pub fn matches_gpu_count(&self, count: usize) -> bool {
125        self.gpu_count == count
126    }
127}
128
129fn default_command_template() -> String {
130    "vllm serve {{MODEL_ID}} --served-model-name {{NAME}} --port {{PORT}} {{VLLM_ARGS}}".to_string()
131}
132
133#[cfg(test)]
134mod tests {
135    use super::*;
136    use crate::pods::state::{PodGpu, PodState};
137
138    #[test]
139    fn profile_matches_gpu_types_by_substring() {
140        let profile = PodProfile {
141            name: "test".to_string(),
142            model: "model".to_string(),
143            gpu_count: 1,
144            gpu_types: vec!["A100".to_string()],
145            command_template: default_command_template(),
146            vllm_args: vec![],
147            env: BTreeMap::new(),
148        };
149        let pod = PodState {
150            name: "pod".to_string(),
151            ssh: "ssh root@example.com".to_string(),
152            models_path: None,
153            gpus: vec![PodGpu { id: 0, name: "NVIDIA A100-SXM4-80GB".to_string() }],
154            models: BTreeMap::new(),
155        };
156
157        assert!(profile.matches_pod(&pod));
158    }
159
160    #[test]
161    fn profile_requires_enough_matching_gpu_types() {
162        let profile = PodProfile {
163            name: "dual-a100".to_string(),
164            model: "model".to_string(),
165            gpu_count: 2,
166            gpu_types: vec!["A100".to_string()],
167            command_template: default_command_template(),
168            vllm_args: vec![],
169            env: BTreeMap::new(),
170        };
171        let pod = PodState {
172            name: "pod".to_string(),
173            ssh: "ssh root@example.com".to_string(),
174            models_path: None,
175            gpus: vec![
176                PodGpu { id: 0, name: "NVIDIA A100-SXM4-80GB".to_string() },
177                PodGpu { id: 1, name: "NVIDIA RTX 4090".to_string() },
178            ],
179            models: BTreeMap::new(),
180        };
181
182        assert!(!profile.matches_pod(&pod));
183    }
184}