Skip to main content

alien_core/
instance_catalog.rs

1//! Instance type catalog and selection algorithm for cloud compute infrastructure.
2//!
3//! This module provides:
4//! - A static catalog of known instance types across AWS, GCP, and Azure
5//! - Resource quantity parsing (CPU strings, Kubernetes-style memory/storage quantities)
6//! - An algorithm to select the optimal instance type for a given workload
7//!
8//! The catalog is the single source of truth for instance type specifications.
9//! It is used by the preflights system to automatically populate `CapacityGroup.instance_type`
10//! and `CapacityGroup.profile` based on the containers in a stack.
11
12use crate::{GpuSpec, MachineProfile, Platform};
13use serde::{Deserialize, Serialize};
14
15// ---------------------------------------------------------------------------
16// Resource quantity parsing
17// ---------------------------------------------------------------------------
18
19/// Parse a CPU quantity string to f64.
20///
21/// Accepts plain numbers ("1", "0.5", "2.0") and millicore suffixes ("500m" = 0.5).
22pub fn parse_cpu(s: &str) -> Result<f64, String> {
23    let s = s.trim();
24    if s.is_empty() {
25        return Err("empty CPU string".to_string());
26    }
27
28    if let Some(millis) = s.strip_suffix('m') {
29        let v: f64 = millis
30            .parse()
31            .map_err(|_| format!("invalid CPU millicore value: '{s}'"))?;
32        Ok(v / 1000.0)
33    } else {
34        s.parse().map_err(|_| format!("invalid CPU value: '{s}'"))
35    }
36}
37
38/// Parse a memory or storage quantity string to bytes.
39///
40/// Supports Kubernetes-style binary suffixes (Ki, Mi, Gi, Ti) and
41/// decimal suffixes (k, M, G, T). Plain numbers are interpreted as bytes.
42pub fn parse_memory_bytes(s: &str) -> Result<u64, String> {
43    let s = s.trim();
44    if s.is_empty() {
45        return Err("empty memory/storage string".to_string());
46    }
47
48    // Binary suffixes (powers of 1024)
49    if let Some(num) = s.strip_suffix("Ti") {
50        let v: f64 = num
51            .parse()
52            .map_err(|_| format!("invalid memory value: '{s}'"))?;
53        return Ok((v * 1024.0 * 1024.0 * 1024.0 * 1024.0) as u64);
54    }
55    if let Some(num) = s.strip_suffix("Gi") {
56        let v: f64 = num
57            .parse()
58            .map_err(|_| format!("invalid memory value: '{s}'"))?;
59        return Ok((v * 1024.0 * 1024.0 * 1024.0) as u64);
60    }
61    if let Some(num) = s.strip_suffix("Mi") {
62        let v: f64 = num
63            .parse()
64            .map_err(|_| format!("invalid memory value: '{s}'"))?;
65        return Ok((v * 1024.0 * 1024.0) as u64);
66    }
67    if let Some(num) = s.strip_suffix("Ki") {
68        let v: f64 = num
69            .parse()
70            .map_err(|_| format!("invalid memory value: '{s}'"))?;
71        return Ok((v * 1024.0) as u64);
72    }
73
74    // Decimal suffixes (powers of 1000)
75    if let Some(num) = s.strip_suffix('T') {
76        let v: f64 = num
77            .parse()
78            .map_err(|_| format!("invalid memory value: '{s}'"))?;
79        return Ok((v * 1_000_000_000_000.0) as u64);
80    }
81    if let Some(num) = s.strip_suffix('G') {
82        let v: f64 = num
83            .parse()
84            .map_err(|_| format!("invalid memory value: '{s}'"))?;
85        return Ok((v * 1_000_000_000.0) as u64);
86    }
87    if let Some(num) = s.strip_suffix('M') {
88        let v: f64 = num
89            .parse()
90            .map_err(|_| format!("invalid memory value: '{s}'"))?;
91        return Ok((v * 1_000_000.0) as u64);
92    }
93    if let Some(num) = s.strip_suffix('k') {
94        let v: f64 = num
95            .parse()
96            .map_err(|_| format!("invalid memory value: '{s}'"))?;
97        return Ok((v * 1000.0) as u64);
98    }
99
100    // Plain bytes
101    s.parse()
102        .map_err(|_| format!("invalid memory value: '{s}'"))
103}
104
105// ---------------------------------------------------------------------------
106// Instance type catalog
107// ---------------------------------------------------------------------------
108
109/// Instance family classification.
110#[derive(Debug, Clone, Copy, PartialEq, Eq)]
111pub enum InstanceFamily {
112    Burstable,
113    GeneralPurpose,
114    ComputeOptimized,
115    MemoryOptimized,
116    StorageOptimized,
117    GpuCompute,
118}
119
120/// CPU architecture.
121#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
122#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
123#[serde(rename_all = "snake_case")]
124pub enum Architecture {
125    Arm64,
126    X86_64,
127}
128
129/// Default machine architecture for images built for a managed cloud.
130pub fn default_architecture(platform: Platform) -> Option<Architecture> {
131    match platform {
132        Platform::Aws => Some(Architecture::Arm64),
133        Platform::Gcp | Platform::Azure => Some(Architecture::X86_64),
134        Platform::Kubernetes | Platform::Machines | Platform::Local | Platform::Test => None,
135    }
136}
137
138/// Static GPU specification for catalog entries (no heap allocation).
139#[derive(Debug, Clone, Copy, PartialEq, Eq)]
140pub struct CatalogGpu {
141    pub gpu_type: &'static str,
142    pub count: u32,
143}
144
145/// A known instance type with its hardware specifications.
146///
147/// All fields are compile-time constants. The catalog is a flat array of these.
148#[derive(Debug, Clone)]
149pub struct InstanceTypeSpec {
150    pub name: &'static str,
151    pub platform: Platform,
152    pub family: InstanceFamily,
153    pub architecture: Architecture,
154    /// vCPU count (hardware total)
155    pub vcpu: u32,
156    /// Memory in bytes (hardware total)
157    pub memory_bytes: u64,
158    /// Ephemeral storage in bytes (hardware total, NVMe for storage-optimized)
159    pub ephemeral_storage_bytes: u64,
160    /// GPU specification (for GPU instances)
161    pub gpu: Option<CatalogGpu>,
162}
163
164impl InstanceTypeSpec {
165    /// Whether ephemeral storage is a provider disk that can be sized at
166    /// deployment time instead of fixed local instance storage.
167    pub fn has_configurable_ephemeral_storage(&self) -> bool {
168        matches!(
169            self.family,
170            InstanceFamily::Burstable
171                | InstanceFamily::GeneralPurpose
172                | InstanceFamily::ComputeOptimized
173                | InstanceFamily::MemoryOptimized
174        )
175    }
176
177    /// Whether this instance type supports nested virtualization.
178    ///
179    /// Classify by documented provider families rather than adding a flag to
180    /// every catalog row. GCP still requires the instance template to opt in;
181    /// Azure exposes the capability automatically on supported VM sizes.
182    pub fn is_nested_virt_capable(&self) -> bool {
183        match self.platform {
184            Platform::Aws => {
185                let name = self.name;
186                name.starts_with("m8i.")
187                    || name.starts_with("c8i.")
188                    || name.starts_with("r8i.")
189                    || name.starts_with("m8i-flex.")
190                    || name.starts_with("c8i-flex.")
191                    || name.starts_with("r8i-flex.")
192            }
193            Platform::Gcp => self.name.starts_with("n2-standard-"),
194            Platform::Azure => {
195                let name = self.name;
196                (name.starts_with("Standard_D") && name.ends_with("s_v5"))
197                    || (name.starts_with("Standard_E") && name.ends_with("s_v5"))
198                    || (name.starts_with("Standard_F") && name.ends_with("s_v2"))
199            }
200            _ => false,
201        }
202    }
203
204    /// Convert this catalog entry into a `MachineProfile` for use in `CapacityGroup`.
205    pub fn to_machine_profile(&self) -> MachineProfile {
206        MachineProfile {
207            cpu: format!("{}.0", self.vcpu),
208            memory_bytes: self.memory_bytes,
209            ephemeral_storage_bytes: self.ephemeral_storage_bytes,
210            architecture: Some(self.architecture),
211            gpu: self.gpu.map(|g| GpuSpec {
212                gpu_type: g.gpu_type.to_string(),
213                count: g.count,
214            }),
215        }
216    }
217
218    /// Convert this entry to the profile controllers should provision for a
219    /// workload. Configurable cloud disks grow to the requested capacity;
220    /// fixed local disks retain their catalog capacity.
221    pub fn to_machine_profile_for_storage(&self, requested_bytes: u64) -> MachineProfile {
222        let mut profile = self.to_machine_profile();
223        if self.has_configurable_ephemeral_storage() {
224            profile.ephemeral_storage_bytes = profile.ephemeral_storage_bytes.max(requested_bytes);
225        }
226        profile
227    }
228}
229
230// Helpers for readable byte constants
231const KI: u64 = 1024;
232const MI: u64 = KI * 1024;
233const GI: u64 = MI * 1024;
234const TI: u64 = GI * 1024;
235
236/// Maximum workload storage that the current cloud controllers can turn into
237/// their provider-backed root disk after adding 25% filesystem headroom and
238/// 12 GiB of runtime overhead. Keep these bounds in sync with the provider
239/// controller contract; rejecting here is preferable to failing after cloud
240/// resources have already been created.
241pub fn max_configurable_ephemeral_storage_bytes(platform: Platform) -> Option<u64> {
242    let disk_limit_gib = match platform {
243        // AWS gp3 and GCP persistent disks currently top out at 64 TiB.
244        Platform::Aws | Platform::Gcp => 64 * 1024,
245        // Azure managed OS disks currently top out at 4,095 GiB.
246        Platform::Azure => 4_095,
247        Platform::Kubernetes | Platform::Machines | Platform::Local | Platform::Test => {
248            return None;
249        }
250    };
251
252    // Controllers first round the requested bytes up to a whole GiB, then
253    // round the 25% headroom up again. Returning a fractional-GiB bound would
254    // therefore admit `bound + 1 byte` and materialize a disk one GiB over the
255    // provider limit.
256    let max_requested_gib = (disk_limit_gib - 12) * 4 / 5;
257    Some(max_requested_gib * GI)
258}
259
260/// The complete instance type catalog.
261///
262/// This is the single source of truth for instance type specifications.
263/// Update this array when adding support for new instance types.
264///
265/// NOTE: Ephemeral storage values for non-NVMe instances are conservative defaults
266/// (EBS-backed root volumes). Storage-optimized instances list their NVMe capacity.
267static CATALOG: &[InstanceTypeSpec] = &[
268    // =========================================================================
269    // AWS — ARM (Graviton) preferred for cost efficiency
270    // =========================================================================
271
272    // Burstable (t4g — ARM Graviton2)
273    InstanceTypeSpec {
274        name: "t4g.micro",
275        platform: Platform::Aws,
276        family: InstanceFamily::Burstable,
277        architecture: Architecture::Arm64,
278        vcpu: 2,
279        memory_bytes: 1 * GI,
280        ephemeral_storage_bytes: 20 * GI,
281        gpu: None,
282    },
283    InstanceTypeSpec {
284        name: "t4g.small",
285        platform: Platform::Aws,
286        family: InstanceFamily::Burstable,
287        architecture: Architecture::Arm64,
288        vcpu: 2,
289        memory_bytes: 2 * GI,
290        ephemeral_storage_bytes: 20 * GI,
291        gpu: None,
292    },
293    InstanceTypeSpec {
294        name: "t4g.medium",
295        platform: Platform::Aws,
296        family: InstanceFamily::Burstable,
297        architecture: Architecture::Arm64,
298        vcpu: 2,
299        memory_bytes: 4 * GI,
300        ephemeral_storage_bytes: 20 * GI,
301        gpu: None,
302    },
303    InstanceTypeSpec {
304        name: "t4g.large",
305        platform: Platform::Aws,
306        family: InstanceFamily::Burstable,
307        architecture: Architecture::Arm64,
308        vcpu: 2,
309        memory_bytes: 8 * GI,
310        ephemeral_storage_bytes: 20 * GI,
311        gpu: None,
312    },
313    InstanceTypeSpec {
314        name: "t3.xlarge",
315        platform: Platform::Aws,
316        family: InstanceFamily::Burstable,
317        architecture: Architecture::X86_64,
318        vcpu: 4,
319        memory_bytes: 16 * GI,
320        ephemeral_storage_bytes: 20 * GI,
321        gpu: None,
322    },
323    InstanceTypeSpec {
324        name: "t4g.xlarge",
325        platform: Platform::Aws,
326        family: InstanceFamily::Burstable,
327        architecture: Architecture::Arm64,
328        vcpu: 4,
329        memory_bytes: 16 * GI,
330        ephemeral_storage_bytes: 20 * GI,
331        gpu: None,
332    },
333    // General Purpose (m7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
334    InstanceTypeSpec {
335        name: "m7g.medium",
336        platform: Platform::Aws,
337        family: InstanceFamily::GeneralPurpose,
338        architecture: Architecture::Arm64,
339        vcpu: 1,
340        memory_bytes: 4 * GI,
341        ephemeral_storage_bytes: 20 * GI,
342        gpu: None,
343    },
344    InstanceTypeSpec {
345        name: "m7i.large",
346        platform: Platform::Aws,
347        family: InstanceFamily::GeneralPurpose,
348        architecture: Architecture::X86_64,
349        vcpu: 2,
350        memory_bytes: 8 * GI,
351        ephemeral_storage_bytes: 20 * GI,
352        gpu: None,
353    },
354    InstanceTypeSpec {
355        name: "m7g.large",
356        platform: Platform::Aws,
357        family: InstanceFamily::GeneralPurpose,
358        architecture: Architecture::Arm64,
359        vcpu: 2,
360        memory_bytes: 8 * GI,
361        ephemeral_storage_bytes: 20 * GI,
362        gpu: None,
363    },
364    // 8th-gen Intel AWS families accept
365    // `CpuOptions.NestedVirtualization=enabled`. The catalog filter in
366    // `select_instance_type` includes these entries only when the
367    // workload requests nested virt, so ordinary workloads continue to
368    // pick the cost-efficient Graviton (m7g) above. The pairwise
369    // interleave keeps the per-family vCPU-non-decreasing invariant
370    // (see `test_catalog_instance_types_sorted_by_vcpu_within_family`).
371    InstanceTypeSpec {
372        name: "m8i.large",
373        platform: Platform::Aws,
374        family: InstanceFamily::GeneralPurpose,
375        architecture: Architecture::X86_64,
376        vcpu: 2,
377        memory_bytes: 8 * GI,
378        ephemeral_storage_bytes: 20 * GI,
379        gpu: None,
380    },
381    InstanceTypeSpec {
382        name: "m7i.xlarge",
383        platform: Platform::Aws,
384        family: InstanceFamily::GeneralPurpose,
385        architecture: Architecture::X86_64,
386        vcpu: 4,
387        memory_bytes: 16 * GI,
388        ephemeral_storage_bytes: 20 * GI,
389        gpu: None,
390    },
391    InstanceTypeSpec {
392        name: "m7g.xlarge",
393        platform: Platform::Aws,
394        family: InstanceFamily::GeneralPurpose,
395        architecture: Architecture::Arm64,
396        vcpu: 4,
397        memory_bytes: 16 * GI,
398        ephemeral_storage_bytes: 20 * GI,
399        gpu: None,
400    },
401    InstanceTypeSpec {
402        name: "m8i.xlarge",
403        platform: Platform::Aws,
404        family: InstanceFamily::GeneralPurpose,
405        architecture: Architecture::X86_64,
406        vcpu: 4,
407        memory_bytes: 16 * GI,
408        ephemeral_storage_bytes: 20 * GI,
409        gpu: None,
410    },
411    InstanceTypeSpec {
412        name: "m7i.2xlarge",
413        platform: Platform::Aws,
414        family: InstanceFamily::GeneralPurpose,
415        architecture: Architecture::X86_64,
416        vcpu: 8,
417        memory_bytes: 32 * GI,
418        ephemeral_storage_bytes: 20 * GI,
419        gpu: None,
420    },
421    InstanceTypeSpec {
422        name: "m7g.2xlarge",
423        platform: Platform::Aws,
424        family: InstanceFamily::GeneralPurpose,
425        architecture: Architecture::Arm64,
426        vcpu: 8,
427        memory_bytes: 32 * GI,
428        ephemeral_storage_bytes: 20 * GI,
429        gpu: None,
430    },
431    InstanceTypeSpec {
432        name: "m8i.2xlarge",
433        platform: Platform::Aws,
434        family: InstanceFamily::GeneralPurpose,
435        architecture: Architecture::X86_64,
436        vcpu: 8,
437        memory_bytes: 32 * GI,
438        ephemeral_storage_bytes: 20 * GI,
439        gpu: None,
440    },
441    InstanceTypeSpec {
442        name: "m7i.4xlarge",
443        platform: Platform::Aws,
444        family: InstanceFamily::GeneralPurpose,
445        architecture: Architecture::X86_64,
446        vcpu: 16,
447        memory_bytes: 64 * GI,
448        ephemeral_storage_bytes: 20 * GI,
449        gpu: None,
450    },
451    InstanceTypeSpec {
452        name: "m7g.4xlarge",
453        platform: Platform::Aws,
454        family: InstanceFamily::GeneralPurpose,
455        architecture: Architecture::Arm64,
456        vcpu: 16,
457        memory_bytes: 64 * GI,
458        ephemeral_storage_bytes: 20 * GI,
459        gpu: None,
460    },
461    InstanceTypeSpec {
462        name: "m8i.4xlarge",
463        platform: Platform::Aws,
464        family: InstanceFamily::GeneralPurpose,
465        architecture: Architecture::X86_64,
466        vcpu: 16,
467        memory_bytes: 64 * GI,
468        ephemeral_storage_bytes: 20 * GI,
469        gpu: None,
470    },
471    // Compute Optimized (c7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
472    InstanceTypeSpec {
473        name: "c7g.medium",
474        platform: Platform::Aws,
475        family: InstanceFamily::ComputeOptimized,
476        architecture: Architecture::Arm64,
477        vcpu: 1,
478        memory_bytes: 2 * GI,
479        ephemeral_storage_bytes: 20 * GI,
480        gpu: None,
481    },
482    InstanceTypeSpec {
483        name: "c7g.large",
484        platform: Platform::Aws,
485        family: InstanceFamily::ComputeOptimized,
486        architecture: Architecture::Arm64,
487        vcpu: 2,
488        memory_bytes: 4 * GI,
489        ephemeral_storage_bytes: 20 * GI,
490        gpu: None,
491    },
492    InstanceTypeSpec {
493        name: "c8i.large",
494        platform: Platform::Aws,
495        family: InstanceFamily::ComputeOptimized,
496        architecture: Architecture::X86_64,
497        vcpu: 2,
498        memory_bytes: 4 * GI,
499        ephemeral_storage_bytes: 20 * GI,
500        gpu: None,
501    },
502    InstanceTypeSpec {
503        name: "c7g.xlarge",
504        platform: Platform::Aws,
505        family: InstanceFamily::ComputeOptimized,
506        architecture: Architecture::Arm64,
507        vcpu: 4,
508        memory_bytes: 8 * GI,
509        ephemeral_storage_bytes: 20 * GI,
510        gpu: None,
511    },
512    InstanceTypeSpec {
513        name: "c8i.xlarge",
514        platform: Platform::Aws,
515        family: InstanceFamily::ComputeOptimized,
516        architecture: Architecture::X86_64,
517        vcpu: 4,
518        memory_bytes: 8 * GI,
519        ephemeral_storage_bytes: 20 * GI,
520        gpu: None,
521    },
522    InstanceTypeSpec {
523        name: "c7g.2xlarge",
524        platform: Platform::Aws,
525        family: InstanceFamily::ComputeOptimized,
526        architecture: Architecture::Arm64,
527        vcpu: 8,
528        memory_bytes: 16 * GI,
529        ephemeral_storage_bytes: 20 * GI,
530        gpu: None,
531    },
532    InstanceTypeSpec {
533        name: "c8i.2xlarge",
534        platform: Platform::Aws,
535        family: InstanceFamily::ComputeOptimized,
536        architecture: Architecture::X86_64,
537        vcpu: 8,
538        memory_bytes: 16 * GI,
539        ephemeral_storage_bytes: 20 * GI,
540        gpu: None,
541    },
542    InstanceTypeSpec {
543        name: "c7g.4xlarge",
544        platform: Platform::Aws,
545        family: InstanceFamily::ComputeOptimized,
546        architecture: Architecture::Arm64,
547        vcpu: 16,
548        memory_bytes: 32 * GI,
549        ephemeral_storage_bytes: 20 * GI,
550        gpu: None,
551    },
552    InstanceTypeSpec {
553        name: "c8i.4xlarge",
554        platform: Platform::Aws,
555        family: InstanceFamily::ComputeOptimized,
556        architecture: Architecture::X86_64,
557        vcpu: 16,
558        memory_bytes: 32 * GI,
559        ephemeral_storage_bytes: 20 * GI,
560        gpu: None,
561    },
562    // Memory Optimized (r7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
563    InstanceTypeSpec {
564        name: "r7g.medium",
565        platform: Platform::Aws,
566        family: InstanceFamily::MemoryOptimized,
567        architecture: Architecture::Arm64,
568        vcpu: 1,
569        memory_bytes: 8 * GI,
570        ephemeral_storage_bytes: 20 * GI,
571        gpu: None,
572    },
573    InstanceTypeSpec {
574        name: "r7g.large",
575        platform: Platform::Aws,
576        family: InstanceFamily::MemoryOptimized,
577        architecture: Architecture::Arm64,
578        vcpu: 2,
579        memory_bytes: 16 * GI,
580        ephemeral_storage_bytes: 20 * GI,
581        gpu: None,
582    },
583    InstanceTypeSpec {
584        name: "r7g.xlarge",
585        platform: Platform::Aws,
586        family: InstanceFamily::MemoryOptimized,
587        architecture: Architecture::Arm64,
588        vcpu: 4,
589        memory_bytes: 32 * GI,
590        ephemeral_storage_bytes: 20 * GI,
591        gpu: None,
592    },
593    InstanceTypeSpec {
594        name: "r7g.2xlarge",
595        platform: Platform::Aws,
596        family: InstanceFamily::MemoryOptimized,
597        architecture: Architecture::Arm64,
598        vcpu: 8,
599        memory_bytes: 64 * GI,
600        ephemeral_storage_bytes: 20 * GI,
601        gpu: None,
602    },
603    InstanceTypeSpec {
604        name: "r7g.4xlarge",
605        platform: Platform::Aws,
606        family: InstanceFamily::MemoryOptimized,
607        architecture: Architecture::Arm64,
608        vcpu: 16,
609        memory_bytes: 128 * GI,
610        ephemeral_storage_bytes: 20 * GI,
611        gpu: None,
612    },
613    // Storage Optimized (i4i — x86_64, NVMe)
614    InstanceTypeSpec {
615        name: "i4i.xlarge",
616        platform: Platform::Aws,
617        family: InstanceFamily::StorageOptimized,
618        architecture: Architecture::X86_64,
619        vcpu: 4,
620        memory_bytes: 32 * GI,
621        ephemeral_storage_bytes: 937 * GI,
622        gpu: None,
623    },
624    InstanceTypeSpec {
625        name: "i4i.2xlarge",
626        platform: Platform::Aws,
627        family: InstanceFamily::StorageOptimized,
628        architecture: Architecture::X86_64,
629        vcpu: 8,
630        memory_bytes: 64 * GI,
631        ephemeral_storage_bytes: 1875 * GI,
632        gpu: None,
633    },
634    InstanceTypeSpec {
635        name: "i4i.4xlarge",
636        platform: Platform::Aws,
637        family: InstanceFamily::StorageOptimized,
638        architecture: Architecture::X86_64,
639        vcpu: 16,
640        memory_bytes: 128 * GI,
641        ephemeral_storage_bytes: 3750 * GI,
642        gpu: None,
643    },
644    InstanceTypeSpec {
645        name: "i4i.8xlarge",
646        platform: Platform::Aws,
647        family: InstanceFamily::StorageOptimized,
648        architecture: Architecture::X86_64,
649        vcpu: 32,
650        memory_bytes: 256 * GI,
651        ephemeral_storage_bytes: 7500 * GI,
652        gpu: None,
653    },
654    // GPU — NVIDIA T4 (g5 — x86_64)
655    InstanceTypeSpec {
656        name: "g5.xlarge",
657        platform: Platform::Aws,
658        family: InstanceFamily::GpuCompute,
659        architecture: Architecture::X86_64,
660        vcpu: 4,
661        memory_bytes: 16 * GI,
662        ephemeral_storage_bytes: 250 * GI,
663        gpu: Some(CatalogGpu {
664            gpu_type: "nvidia-t4",
665            count: 1,
666        }),
667    },
668    InstanceTypeSpec {
669        name: "g5.2xlarge",
670        platform: Platform::Aws,
671        family: InstanceFamily::GpuCompute,
672        architecture: Architecture::X86_64,
673        vcpu: 8,
674        memory_bytes: 32 * GI,
675        ephemeral_storage_bytes: 450 * GI,
676        gpu: Some(CatalogGpu {
677            gpu_type: "nvidia-t4",
678            count: 1,
679        }),
680    },
681    // GPU — NVIDIA A100 (p4d — x86_64)
682    InstanceTypeSpec {
683        name: "p4d.24xlarge",
684        platform: Platform::Aws,
685        family: InstanceFamily::GpuCompute,
686        architecture: Architecture::X86_64,
687        vcpu: 96,
688        memory_bytes: 1152 * GI,
689        ephemeral_storage_bytes: 8000 * GI,
690        gpu: Some(CatalogGpu {
691            gpu_type: "nvidia-a100",
692            count: 8,
693        }),
694    },
695    // GPU — NVIDIA H100 (p5 — x86_64)
696    InstanceTypeSpec {
697        name: "p5.48xlarge",
698        platform: Platform::Aws,
699        family: InstanceFamily::GpuCompute,
700        architecture: Architecture::X86_64,
701        vcpu: 192,
702        memory_bytes: 2048 * GI,
703        ephemeral_storage_bytes: 8000 * GI,
704        gpu: Some(CatalogGpu {
705            gpu_type: "nvidia-h100",
706            count: 8,
707        }),
708    },
709    // =========================================================================
710    // GCP
711    // =========================================================================
712
713    // Burstable (e2)
714    InstanceTypeSpec {
715        name: "e2-micro",
716        platform: Platform::Gcp,
717        family: InstanceFamily::Burstable,
718        architecture: Architecture::X86_64,
719        vcpu: 2,
720        memory_bytes: 1 * GI,
721        ephemeral_storage_bytes: 20 * GI,
722        gpu: None,
723    },
724    InstanceTypeSpec {
725        name: "e2-small",
726        platform: Platform::Gcp,
727        family: InstanceFamily::Burstable,
728        architecture: Architecture::X86_64,
729        vcpu: 2,
730        memory_bytes: 2 * GI,
731        ephemeral_storage_bytes: 20 * GI,
732        gpu: None,
733    },
734    InstanceTypeSpec {
735        name: "e2-medium",
736        platform: Platform::Gcp,
737        family: InstanceFamily::Burstable,
738        architecture: Architecture::X86_64,
739        vcpu: 2,
740        memory_bytes: 4 * GI,
741        ephemeral_storage_bytes: 20 * GI,
742        gpu: None,
743    },
744    // General Purpose (n2-standard, up to 16 vCPU)
745    InstanceTypeSpec {
746        name: "n2-standard-2",
747        platform: Platform::Gcp,
748        family: InstanceFamily::GeneralPurpose,
749        architecture: Architecture::X86_64,
750        vcpu: 2,
751        memory_bytes: 8 * GI,
752        ephemeral_storage_bytes: 20 * GI,
753        gpu: None,
754    },
755    InstanceTypeSpec {
756        name: "n2-standard-4",
757        platform: Platform::Gcp,
758        family: InstanceFamily::GeneralPurpose,
759        architecture: Architecture::X86_64,
760        vcpu: 4,
761        memory_bytes: 16 * GI,
762        ephemeral_storage_bytes: 20 * GI,
763        gpu: None,
764    },
765    InstanceTypeSpec {
766        name: "n2-standard-8",
767        platform: Platform::Gcp,
768        family: InstanceFamily::GeneralPurpose,
769        architecture: Architecture::X86_64,
770        vcpu: 8,
771        memory_bytes: 32 * GI,
772        ephemeral_storage_bytes: 20 * GI,
773        gpu: None,
774    },
775    InstanceTypeSpec {
776        name: "n2-standard-16",
777        platform: Platform::Gcp,
778        family: InstanceFamily::GeneralPurpose,
779        architecture: Architecture::X86_64,
780        vcpu: 16,
781        memory_bytes: 64 * GI,
782        ephemeral_storage_bytes: 20 * GI,
783        gpu: None,
784    },
785    // Compute Optimized (c3-standard, up to 8 vCPU)
786    InstanceTypeSpec {
787        name: "c3-standard-4",
788        platform: Platform::Gcp,
789        family: InstanceFamily::ComputeOptimized,
790        architecture: Architecture::X86_64,
791        vcpu: 4,
792        memory_bytes: 8 * GI,
793        ephemeral_storage_bytes: 20 * GI,
794        gpu: None,
795    },
796    InstanceTypeSpec {
797        name: "c3-standard-8",
798        platform: Platform::Gcp,
799        family: InstanceFamily::ComputeOptimized,
800        architecture: Architecture::X86_64,
801        vcpu: 8,
802        memory_bytes: 16 * GI,
803        ephemeral_storage_bytes: 20 * GI,
804        gpu: None,
805    },
806    // Memory Optimized (n2-highmem, up to 8 vCPU)
807    InstanceTypeSpec {
808        name: "n2-highmem-2",
809        platform: Platform::Gcp,
810        family: InstanceFamily::MemoryOptimized,
811        architecture: Architecture::X86_64,
812        vcpu: 2,
813        memory_bytes: 16 * GI,
814        ephemeral_storage_bytes: 20 * GI,
815        gpu: None,
816    },
817    InstanceTypeSpec {
818        name: "n2-highmem-4",
819        platform: Platform::Gcp,
820        family: InstanceFamily::MemoryOptimized,
821        architecture: Architecture::X86_64,
822        vcpu: 4,
823        memory_bytes: 32 * GI,
824        ephemeral_storage_bytes: 20 * GI,
825        gpu: None,
826    },
827    InstanceTypeSpec {
828        name: "n2-highmem-8",
829        platform: Platform::Gcp,
830        family: InstanceFamily::MemoryOptimized,
831        architecture: Architecture::X86_64,
832        vcpu: 8,
833        memory_bytes: 64 * GI,
834        ephemeral_storage_bytes: 20 * GI,
835        gpu: None,
836    },
837    InstanceTypeSpec {
838        name: "n2-highmem-16",
839        platform: Platform::Gcp,
840        family: InstanceFamily::MemoryOptimized,
841        architecture: Architecture::X86_64,
842        vcpu: 16,
843        memory_bytes: 128 * GI,
844        ephemeral_storage_bytes: 20 * GI,
845        gpu: None,
846    },
847    InstanceTypeSpec {
848        name: "n2-highmem-32",
849        platform: Platform::Gcp,
850        family: InstanceFamily::MemoryOptimized,
851        architecture: Architecture::X86_64,
852        vcpu: 32,
853        memory_bytes: 256 * GI,
854        ephemeral_storage_bytes: 20 * GI,
855        gpu: None,
856    },
857    // Storage Optimized (c3d-standard with local SSD)
858    InstanceTypeSpec {
859        name: "c3d-standard-8",
860        platform: Platform::Gcp,
861        family: InstanceFamily::StorageOptimized,
862        architecture: Architecture::X86_64,
863        vcpu: 8,
864        memory_bytes: 32 * GI,
865        ephemeral_storage_bytes: 480 * GI,
866        gpu: None,
867    },
868    InstanceTypeSpec {
869        name: "c3d-standard-16",
870        platform: Platform::Gcp,
871        family: InstanceFamily::StorageOptimized,
872        architecture: Architecture::X86_64,
873        vcpu: 16,
874        memory_bytes: 64 * GI,
875        ephemeral_storage_bytes: 960 * GI,
876        gpu: None,
877    },
878    InstanceTypeSpec {
879        name: "c3d-standard-30",
880        platform: Platform::Gcp,
881        family: InstanceFamily::StorageOptimized,
882        architecture: Architecture::X86_64,
883        vcpu: 30,
884        memory_bytes: 120 * GI,
885        ephemeral_storage_bytes: 1920 * GI,
886        gpu: None,
887    },
888    // GPU — NVIDIA T4 (n1-standard + T4)
889    InstanceTypeSpec {
890        name: "n1-standard-4-t4",
891        platform: Platform::Gcp,
892        family: InstanceFamily::GpuCompute,
893        architecture: Architecture::X86_64,
894        vcpu: 4,
895        memory_bytes: 15 * GI,
896        ephemeral_storage_bytes: 100 * GI,
897        gpu: Some(CatalogGpu {
898            gpu_type: "nvidia-t4",
899            count: 1,
900        }),
901    },
902    // GPU — NVIDIA A100 (a2-highgpu)
903    InstanceTypeSpec {
904        name: "a2-highgpu-1g",
905        platform: Platform::Gcp,
906        family: InstanceFamily::GpuCompute,
907        architecture: Architecture::X86_64,
908        vcpu: 12,
909        memory_bytes: 85 * GI,
910        ephemeral_storage_bytes: 100 * GI,
911        gpu: Some(CatalogGpu {
912            gpu_type: "nvidia-a100",
913            count: 1,
914        }),
915    },
916    InstanceTypeSpec {
917        name: "a2-highgpu-8g",
918        platform: Platform::Gcp,
919        family: InstanceFamily::GpuCompute,
920        architecture: Architecture::X86_64,
921        vcpu: 96,
922        memory_bytes: 1360 * GI,
923        ephemeral_storage_bytes: 100 * GI,
924        gpu: Some(CatalogGpu {
925            gpu_type: "nvidia-a100",
926            count: 8,
927        }),
928    },
929    // GPU — NVIDIA H100 (a3-highgpu)
930    InstanceTypeSpec {
931        name: "a3-highgpu-8g",
932        platform: Platform::Gcp,
933        family: InstanceFamily::GpuCompute,
934        architecture: Architecture::X86_64,
935        vcpu: 208,
936        memory_bytes: 1872 * GI,
937        ephemeral_storage_bytes: 100 * GI,
938        gpu: Some(CatalogGpu {
939            gpu_type: "nvidia-h100",
940            count: 8,
941        }),
942    },
943    // =========================================================================
944    // Azure
945    // =========================================================================
946
947    // Burstable (B-series v2)
948    InstanceTypeSpec {
949        name: "Standard_B1s",
950        platform: Platform::Azure,
951        family: InstanceFamily::Burstable,
952        architecture: Architecture::X86_64,
953        vcpu: 1,
954        memory_bytes: 1 * GI,
955        ephemeral_storage_bytes: 20 * GI,
956        gpu: None,
957    },
958    InstanceTypeSpec {
959        name: "Standard_B2s",
960        platform: Platform::Azure,
961        family: InstanceFamily::Burstable,
962        architecture: Architecture::X86_64,
963        vcpu: 2,
964        memory_bytes: 4 * GI,
965        ephemeral_storage_bytes: 20 * GI,
966        gpu: None,
967    },
968    InstanceTypeSpec {
969        name: "Standard_B2ms",
970        platform: Platform::Azure,
971        family: InstanceFamily::Burstable,
972        architecture: Architecture::X86_64,
973        vcpu: 2,
974        memory_bytes: 8 * GI,
975        ephemeral_storage_bytes: 20 * GI,
976        gpu: None,
977    },
978    InstanceTypeSpec {
979        name: "Standard_B4ms",
980        platform: Platform::Azure,
981        family: InstanceFamily::Burstable,
982        architecture: Architecture::X86_64,
983        vcpu: 4,
984        memory_bytes: 16 * GI,
985        ephemeral_storage_bytes: 20 * GI,
986        gpu: None,
987    },
988    // General Purpose (Dv5-series, up to 16 vCPU)
989    InstanceTypeSpec {
990        name: "Standard_D2s_v5",
991        platform: Platform::Azure,
992        family: InstanceFamily::GeneralPurpose,
993        architecture: Architecture::X86_64,
994        vcpu: 2,
995        memory_bytes: 8 * GI,
996        ephemeral_storage_bytes: 20 * GI,
997        gpu: None,
998    },
999    InstanceTypeSpec {
1000        name: "Standard_D4s_v5",
1001        platform: Platform::Azure,
1002        family: InstanceFamily::GeneralPurpose,
1003        architecture: Architecture::X86_64,
1004        vcpu: 4,
1005        memory_bytes: 16 * GI,
1006        ephemeral_storage_bytes: 20 * GI,
1007        gpu: None,
1008    },
1009    InstanceTypeSpec {
1010        name: "Standard_D8s_v5",
1011        platform: Platform::Azure,
1012        family: InstanceFamily::GeneralPurpose,
1013        architecture: Architecture::X86_64,
1014        vcpu: 8,
1015        memory_bytes: 32 * GI,
1016        ephemeral_storage_bytes: 20 * GI,
1017        gpu: None,
1018    },
1019    InstanceTypeSpec {
1020        name: "Standard_D16s_v5",
1021        platform: Platform::Azure,
1022        family: InstanceFamily::GeneralPurpose,
1023        architecture: Architecture::X86_64,
1024        vcpu: 16,
1025        memory_bytes: 64 * GI,
1026        ephemeral_storage_bytes: 20 * GI,
1027        gpu: None,
1028    },
1029    // Compute Optimized (Fv2-series, up to 16 vCPU)
1030    InstanceTypeSpec {
1031        name: "Standard_F2s_v2",
1032        platform: Platform::Azure,
1033        family: InstanceFamily::ComputeOptimized,
1034        architecture: Architecture::X86_64,
1035        vcpu: 2,
1036        memory_bytes: 4 * GI,
1037        ephemeral_storage_bytes: 20 * GI,
1038        gpu: None,
1039    },
1040    InstanceTypeSpec {
1041        name: "Standard_F4s_v2",
1042        platform: Platform::Azure,
1043        family: InstanceFamily::ComputeOptimized,
1044        architecture: Architecture::X86_64,
1045        vcpu: 4,
1046        memory_bytes: 8 * GI,
1047        ephemeral_storage_bytes: 20 * GI,
1048        gpu: None,
1049    },
1050    InstanceTypeSpec {
1051        name: "Standard_F8s_v2",
1052        platform: Platform::Azure,
1053        family: InstanceFamily::ComputeOptimized,
1054        architecture: Architecture::X86_64,
1055        vcpu: 8,
1056        memory_bytes: 16 * GI,
1057        ephemeral_storage_bytes: 20 * GI,
1058        gpu: None,
1059    },
1060    InstanceTypeSpec {
1061        name: "Standard_F16s_v2",
1062        platform: Platform::Azure,
1063        family: InstanceFamily::ComputeOptimized,
1064        architecture: Architecture::X86_64,
1065        vcpu: 16,
1066        memory_bytes: 32 * GI,
1067        ephemeral_storage_bytes: 20 * GI,
1068        gpu: None,
1069    },
1070    // Memory Optimized (Ev5-series, up to 16 vCPU)
1071    InstanceTypeSpec {
1072        name: "Standard_E2s_v5",
1073        platform: Platform::Azure,
1074        family: InstanceFamily::MemoryOptimized,
1075        architecture: Architecture::X86_64,
1076        vcpu: 2,
1077        memory_bytes: 16 * GI,
1078        ephemeral_storage_bytes: 20 * GI,
1079        gpu: None,
1080    },
1081    InstanceTypeSpec {
1082        name: "Standard_E4s_v5",
1083        platform: Platform::Azure,
1084        family: InstanceFamily::MemoryOptimized,
1085        architecture: Architecture::X86_64,
1086        vcpu: 4,
1087        memory_bytes: 32 * GI,
1088        ephemeral_storage_bytes: 20 * GI,
1089        gpu: None,
1090    },
1091    InstanceTypeSpec {
1092        name: "Standard_E8s_v5",
1093        platform: Platform::Azure,
1094        family: InstanceFamily::MemoryOptimized,
1095        architecture: Architecture::X86_64,
1096        vcpu: 8,
1097        memory_bytes: 64 * GI,
1098        ephemeral_storage_bytes: 20 * GI,
1099        gpu: None,
1100    },
1101    InstanceTypeSpec {
1102        name: "Standard_E16s_v5",
1103        platform: Platform::Azure,
1104        family: InstanceFamily::MemoryOptimized,
1105        architecture: Architecture::X86_64,
1106        vcpu: 16,
1107        memory_bytes: 128 * GI,
1108        ephemeral_storage_bytes: 20 * GI,
1109        gpu: None,
1110    },
1111    // Storage Optimized (Lsv3-series with NVMe)
1112    InstanceTypeSpec {
1113        name: "Standard_L8s_v3",
1114        platform: Platform::Azure,
1115        family: InstanceFamily::StorageOptimized,
1116        architecture: Architecture::X86_64,
1117        vcpu: 8,
1118        memory_bytes: 64 * GI,
1119        ephemeral_storage_bytes: 1788 * GI,
1120        gpu: None,
1121    },
1122    InstanceTypeSpec {
1123        name: "Standard_L16s_v3",
1124        platform: Platform::Azure,
1125        family: InstanceFamily::StorageOptimized,
1126        architecture: Architecture::X86_64,
1127        vcpu: 16,
1128        memory_bytes: 128 * GI,
1129        ephemeral_storage_bytes: 3576 * GI,
1130        gpu: None,
1131    },
1132    InstanceTypeSpec {
1133        name: "Standard_L32s_v3",
1134        platform: Platform::Azure,
1135        family: InstanceFamily::StorageOptimized,
1136        architecture: Architecture::X86_64,
1137        vcpu: 32,
1138        memory_bytes: 256 * GI,
1139        ephemeral_storage_bytes: 7154 * GI,
1140        gpu: None,
1141    },
1142    // GPU — NVIDIA T4 (NCasT4_v3-series)
1143    InstanceTypeSpec {
1144        name: "Standard_NC4as_T4_v3",
1145        platform: Platform::Azure,
1146        family: InstanceFamily::GpuCompute,
1147        architecture: Architecture::X86_64,
1148        vcpu: 4,
1149        memory_bytes: 28 * GI,
1150        ephemeral_storage_bytes: 176 * GI,
1151        gpu: Some(CatalogGpu {
1152            gpu_type: "nvidia-t4",
1153            count: 1,
1154        }),
1155    },
1156    // GPU — NVIDIA A100 (NC A100 v4-series)
1157    InstanceTypeSpec {
1158        name: "Standard_NC24ads_A100_v4",
1159        platform: Platform::Azure,
1160        family: InstanceFamily::GpuCompute,
1161        architecture: Architecture::X86_64,
1162        vcpu: 24,
1163        memory_bytes: 220 * GI,
1164        ephemeral_storage_bytes: 958 * GI,
1165        gpu: Some(CatalogGpu {
1166            gpu_type: "nvidia-a100",
1167            count: 1,
1168        }),
1169    },
1170    InstanceTypeSpec {
1171        name: "Standard_NC96ads_A100_v4",
1172        platform: Platform::Azure,
1173        family: InstanceFamily::GpuCompute,
1174        architecture: Architecture::X86_64,
1175        vcpu: 96,
1176        memory_bytes: 880 * GI,
1177        ephemeral_storage_bytes: 3916 * GI,
1178        gpu: Some(CatalogGpu {
1179            gpu_type: "nvidia-a100",
1180            count: 4,
1181        }),
1182    },
1183    // GPU — NVIDIA H100 (ND H100 v5-series)
1184    InstanceTypeSpec {
1185        name: "Standard_ND96isr_H100_v5",
1186        platform: Platform::Azure,
1187        family: InstanceFamily::GpuCompute,
1188        architecture: Architecture::X86_64,
1189        vcpu: 96,
1190        memory_bytes: 1900 * GI,
1191        ephemeral_storage_bytes: 1000 * GI,
1192        gpu: Some(CatalogGpu {
1193            gpu_type: "nvidia-h100",
1194            count: 8,
1195        }),
1196    },
1197];
1198
1199// ---------------------------------------------------------------------------
1200// Catalog lookup
1201// ---------------------------------------------------------------------------
1202
1203/// Get all instance types for a given platform.
1204pub fn catalog_for_platform(platform: Platform) -> Vec<&'static InstanceTypeSpec> {
1205    CATALOG
1206        .iter()
1207        .filter(|spec| spec.platform == platform)
1208        .collect()
1209}
1210
1211/// Find a specific instance type by name and platform.
1212pub fn find_instance_type(platform: Platform, name: &str) -> Option<&'static InstanceTypeSpec> {
1213    CATALOG
1214        .iter()
1215        .find(|spec| spec.platform == platform && spec.name == name)
1216}
1217
1218/// Whether a capacity group may move from AWS machine `old` to `new` without setup: both are
1219/// catalog machines of one CPU architecture, so the stack's images still run on the new one.
1220pub fn is_same_architecture_aws_machine(old: &str, new: &str) -> bool {
1221    match (
1222        find_instance_type(Platform::Aws, old),
1223        find_instance_type(Platform::Aws, new),
1224    ) {
1225        (Some(old), Some(new)) => old.architecture == new.architecture,
1226        _ => false,
1227    }
1228}
1229
1230// ---------------------------------------------------------------------------
1231// Instance type selection
1232// ---------------------------------------------------------------------------
1233
1234/// Aggregated resource requirements from all containers in a capacity group.
1235#[derive(Debug, Clone)]
1236pub struct WorkloadRequirements {
1237    /// Total CPU needed at desired scale (sum of desired CPU * desired_replicas per container)
1238    pub total_cpu_at_desired: f64,
1239    /// Total memory needed at desired scale (sum of desired memory * desired_replicas per container)
1240    pub total_memory_bytes_at_desired: u64,
1241    /// Total CPU needed at maximum scale (sum of desired CPU * max_replicas per container)
1242    pub total_cpu_at_max: f64,
1243    /// Total memory needed at maximum scale (sum of desired memory * max_replicas per container)
1244    pub total_memory_bytes_at_max: u64,
1245    /// Largest CPU request among all individual containers (single replica)
1246    pub max_cpu_per_container: f64,
1247    /// Largest memory request among all individual containers (single replica)
1248    pub max_memory_per_container: u64,
1249    /// Maximum ephemeral storage any single container requires
1250    pub max_ephemeral_storage_bytes: u64,
1251    /// GPU requirement (if any container needs GPU)
1252    pub gpu: Option<GpuSpec>,
1253    /// Required CPU architecture, when source explicitly constrains it.
1254    pub architecture: Option<Architecture>,
1255    /// If true, only instance types that expose nested virtualization (VT-x/EPT)
1256    /// to guest VMs are eligible. Required by workloads that run QEMU/KVM
1257    /// inside a container.
1258    pub nested_virt: bool,
1259}
1260
1261/// Result of instance type selection.
1262#[derive(Debug, Clone)]
1263pub struct InstanceSelection {
1264    /// Selected instance type name (e.g., "m7g.2xlarge")
1265    pub instance_type: &'static str,
1266    /// Machine profile derived from the instance type
1267    pub profile: MachineProfile,
1268    /// Recommended minimum number of machines
1269    pub min_machines: u32,
1270    /// Recommended maximum number of machines
1271    pub max_machines: u32,
1272}
1273
1274/// Ephemeral storage threshold above which storage-optimized instances are selected.
1275const STORAGE_OPTIMIZED_THRESHOLD: u64 = 200 * GI;
1276
1277/// Maximum number of machines per cluster.
1278const MAX_MACHINES_PER_CLUSTER: u32 = 10;
1279
1280/// Hard cap on vCPUs for non-GPU/non-storage workloads. Equivalent to AWS 2xlarge.
1281/// Beyond this, horizontal scaling is always preferred over bigger machines.
1282const MAX_STANDARD_VCPU: u32 = 8;
1283
1284/// Runtime CPU reserved for system processes on each managed container machine.
1285const SYSTEM_RESERVE_CPU: f64 = 0.5;
1286
1287/// Runtime planning headroom for total desired/max workload.
1288const WORKLOAD_HEADROOM_FACTOR: f64 = 1.15;
1289
1290/// Select the best instance type for a workload on a given platform.
1291///
1292/// The algorithm:
1293/// 1. GPU workloads: Match by GPU type, find smallest instance with enough GPUs.
1294/// 2. Storage-heavy workloads (>200Gi ephemeral): Use storage-optimized instances.
1295/// 3. All other workloads: Size the machine to fit a small HA-friendly baseline,
1296///    capped at 8 vCPUs. Use GeneralPurpose family for broad availability and
1297///    reasonable cost. Scale horizontally for more capacity.
1298///
1299/// Returns an error if no suitable instance type is found.
1300pub fn select_instance_type(
1301    platform: Platform,
1302    requirements: &WorkloadRequirements,
1303) -> Result<InstanceSelection, String> {
1304    let architecture = requirements
1305        .architecture
1306        .or_else(|| default_architecture(platform))
1307        .ok_or_else(|| format!("platform {platform} has no default compute architecture"))?;
1308
1309    // Determine which family to use. Nested virt isn't available on
1310    // burstable hardware on any cloud, so a workload that classifies as
1311    // Burstable but needs nested virt must be upgraded to GeneralPurpose
1312    // (the family that actually has nested-virt-capable entries).
1313    let raw_family = select_family(requirements);
1314    let family = if requirements.nested_virt && raw_family == InstanceFamily::Burstable {
1315        InstanceFamily::GeneralPurpose
1316    } else {
1317        raw_family
1318    };
1319
1320    let candidates: Vec<&InstanceTypeSpec> = CATALOG
1321        .iter()
1322        .filter(|spec| spec.platform == platform && spec.family == family)
1323        .filter(|spec| {
1324            if requirements.nested_virt {
1325                spec.is_nested_virt_capable()
1326            } else {
1327                platform != Platform::Aws || !spec.is_nested_virt_capable()
1328            }
1329        })
1330        .filter(|spec| spec.architecture == architecture)
1331        .collect();
1332
1333    // A storage-heavy workload may have no fixed-local candidate for the
1334    // requested architecture/nested-virtualization contract. That is exactly
1335    // when we must consider a provider-backed disk on a general-purpose VM;
1336    // rejecting here made the fallback below unreachable.
1337    if candidates.is_empty() && family != InstanceFamily::StorageOptimized {
1338        let family_has_other_architecture = CATALOG.iter().any(|spec| {
1339            spec.platform == platform
1340                && spec.family == family
1341                && if requirements.nested_virt {
1342                    spec.is_nested_virt_capable()
1343                } else {
1344                    platform != Platform::Aws || !spec.is_nested_virt_capable()
1345                }
1346        });
1347        if family_has_other_architecture {
1348            return Err(format!(
1349                "architecture {architecture:?} is unavailable for this workload on platform {platform}"
1350            ));
1351        }
1352        return Err(if requirements.nested_virt {
1353            format!(
1354                "no nested-virt-capable {family:?} instance types in catalog for platform {platform}"
1355            )
1356        } else {
1357            format!("no {family:?} instance types in catalog for platform {platform}")
1358        });
1359    }
1360
1361    // For GPU workloads, filter by GPU type
1362    let candidates = if let Some(ref gpu) = requirements.gpu {
1363        let filtered: Vec<&InstanceTypeSpec> = candidates
1364            .into_iter()
1365            .filter(|spec| {
1366                spec.gpu.as_ref().map_or(false, |g| {
1367                    g.gpu_type == gpu.gpu_type && g.count >= gpu.count
1368                })
1369            })
1370            .collect();
1371        if filtered.is_empty() {
1372            return Err(format!(
1373                "no instance type for GPU type '{}' x{} on platform {platform}",
1374                gpu.gpu_type, gpu.count
1375            ));
1376        }
1377        filtered
1378    } else {
1379        candidates
1380    };
1381
1382    // For storage workloads, filter by ephemeral storage capacity
1383    let candidates = if family == InstanceFamily::StorageOptimized {
1384        let filtered: Vec<&InstanceTypeSpec> = candidates
1385            .into_iter()
1386            .filter(|spec| spec.ephemeral_storage_bytes >= requirements.max_ephemeral_storage_bytes)
1387            .collect();
1388        if filtered.is_empty() {
1389            // Fixed-local NVMe is preferred for large requests, but provider
1390            // disks on ordinary cloud machines remain configurable. Fall back
1391            // when no fixed-local catalog entry can satisfy the request.
1392            CATALOG
1393                .iter()
1394                .filter(|spec| {
1395                    spec.platform == platform
1396                        && spec.family == InstanceFamily::GeneralPurpose
1397                        && spec.has_configurable_ephemeral_storage()
1398                        && max_configurable_ephemeral_storage_bytes(platform)
1399                            .is_some_and(|max| requirements.max_ephemeral_storage_bytes <= max)
1400                })
1401                .filter(|spec| {
1402                    if requirements.nested_virt {
1403                        spec.is_nested_virt_capable()
1404                    } else {
1405                        platform != Platform::Aws || !spec.is_nested_virt_capable()
1406                    }
1407                })
1408                .filter(|spec| spec.architecture == architecture)
1409                .collect()
1410        } else {
1411            filtered
1412        }
1413    } else {
1414        candidates
1415    };
1416
1417    if candidates.is_empty() {
1418        return Err(format!(
1419            "architecture {architecture:?} is unavailable for this workload on platform {platform}"
1420        ));
1421    }
1422
1423    // Apply the policy for the candidates we will actually select from. A
1424    // storage-heavy request can fall back from fixed-local storage machines to
1425    // general-purpose machines with provider-backed disks; those machines must
1426    // retain the normal horizontal-scaling cap.
1427    let effective_family = candidates[0].family;
1428    let vcpu_cap = if effective_family == InstanceFamily::GpuCompute
1429        || effective_family == InstanceFamily::StorageOptimized
1430    {
1431        u32::MAX
1432    } else {
1433        MAX_STANDARD_VCPU
1434    };
1435
1436    let desired_target_machines = desired_target_machines(requirements);
1437    let target_cpu = requirements
1438        .max_cpu_per_container
1439        .max(requirements.total_cpu_at_desired / desired_target_machines as f64)
1440        * WORKLOAD_HEADROOM_FACTOR;
1441    let target_memory = (requirements.max_memory_per_container as f64)
1442        .max(requirements.total_memory_bytes_at_desired as f64 / desired_target_machines as f64)
1443        * WORKLOAD_HEADROOM_FACTOR;
1444
1445    // Find the smallest instance whose allocatable capacity meets the workload
1446    // target after host reserve and workload headroom. Machine count already
1447    // accounts for multiple replicas; requiring space for an arbitrary second
1448    // copy here would size the same demand twice.
1449    let selected = candidates
1450        .iter()
1451        .filter(|spec| {
1452            spec.vcpu <= vcpu_cap
1453                && allocatable_cpu(spec) >= target_cpu
1454                && allocatable_memory_bytes(spec) as f64 >= target_memory
1455        })
1456        .min_by_key(|spec| spec.vcpu)
1457        .or_else(|| {
1458            // If nothing fits within the cap, pick the largest instance under the cap
1459            candidates
1460                .iter()
1461                .filter(|spec| spec.vcpu <= vcpu_cap)
1462                .max_by_key(|spec| spec.vcpu)
1463        })
1464        .or_else(|| {
1465            // Last resort: pick the smallest available instance (for GPU/storage)
1466            candidates.iter().min_by_key(|spec| spec.vcpu)
1467        })
1468        .ok_or_else(|| format!("no instance types available for platform {platform}"))?;
1469
1470    // Calculate machine counts
1471    let max_machines = compute_max_machines(requirements, selected);
1472    let min_machines = compute_min_machines(requirements, selected, max_machines);
1473
1474    Ok(InstanceSelection {
1475        instance_type: selected.name,
1476        profile: selected.to_machine_profile_for_storage(requirements.max_ephemeral_storage_bytes),
1477        min_machines,
1478        max_machines,
1479    })
1480}
1481
1482/// Select instance family based on workload characteristics.
1483///
1484/// Uses GeneralPurpose for all standard workloads — widely available across
1485/// regions and cost-effective. Only specialized workloads (GPU, large ephemeral
1486/// storage) get specialized families. Very small workloads get burstable.
1487pub fn select_family(requirements: &WorkloadRequirements) -> InstanceFamily {
1488    // GPU workloads always get GPU instances
1489    if requirements.gpu.is_some() {
1490        return InstanceFamily::GpuCompute;
1491    }
1492
1493    // Large ephemeral storage needs NVMe (storage-optimized)
1494    if requirements.max_ephemeral_storage_bytes > STORAGE_OPTIMIZED_THRESHOLD {
1495        return InstanceFamily::StorageOptimized;
1496    }
1497
1498    // Very small workloads use burstable instances
1499    if requirements.total_cpu_at_max < 2.0 {
1500        return InstanceFamily::Burstable;
1501    }
1502
1503    // All other workloads use GeneralPurpose — available everywhere, good pricing
1504    InstanceFamily::GeneralPurpose
1505}
1506
1507/// Calculate maximum machines needed to fit the workload with headroom.
1508fn compute_max_machines(requirements: &WorkloadRequirements, instance: &InstanceTypeSpec) -> u32 {
1509    let cpu_with_headroom = requirements.total_cpu_at_max * WORKLOAD_HEADROOM_FACTOR;
1510    let cpu_machines = (cpu_with_headroom / allocatable_cpu(instance)).ceil() as u32;
1511
1512    let mem_with_headroom =
1513        requirements.total_memory_bytes_at_max as f64 * WORKLOAD_HEADROOM_FACTOR;
1514    let mem_machines =
1515        (mem_with_headroom / allocatable_memory_bytes(instance) as f64).ceil() as u32;
1516
1517    // Take the larger of CPU-based and memory-based, clamped to cluster limit
1518    cpu_machines
1519        .max(mem_machines)
1520        .max(1)
1521        .min(MAX_MACHINES_PER_CLUSTER)
1522}
1523
1524/// Calculate minimum machines for HA.
1525fn compute_min_machines(
1526    requirements: &WorkloadRequirements,
1527    instance: &InstanceTypeSpec,
1528    max_machines: u32,
1529) -> u32 {
1530    let cpu_with_headroom = requirements.total_cpu_at_desired * WORKLOAD_HEADROOM_FACTOR;
1531    let cpu_machines = (cpu_with_headroom / allocatable_cpu(instance)).ceil() as u32;
1532
1533    let mem_with_headroom =
1534        requirements.total_memory_bytes_at_desired as f64 * WORKLOAD_HEADROOM_FACTOR;
1535    let mem_machines =
1536        (mem_with_headroom / allocatable_memory_bytes(instance) as f64).ceil() as u32;
1537
1538    cpu_machines
1539        .max(mem_machines)
1540        .max(1)
1541        .min(2)
1542        .min(max_machines)
1543}
1544
1545fn desired_target_machines(requirements: &WorkloadRequirements) -> u32 {
1546    if requirements.total_cpu_at_desired >= 2.0
1547        || requirements.total_memory_bytes_at_desired >= 4 * GI
1548    {
1549        2
1550    } else {
1551        1
1552    }
1553}
1554
1555fn allocatable_cpu(instance: &InstanceTypeSpec) -> f64 {
1556    (instance.vcpu as f64 - SYSTEM_RESERVE_CPU).max(0.25)
1557}
1558
1559fn allocatable_memory_bytes(instance: &InstanceTypeSpec) -> u64 {
1560    instance
1561        .memory_bytes
1562        .saturating_sub(system_reserve_memory_bytes(instance.memory_bytes))
1563        .max(256 * MI)
1564}
1565
1566fn system_reserve_memory_bytes(memory_bytes: u64) -> u64 {
1567    if memory_bytes < 4 * GI {
1568        256 * MI
1569    } else if memory_bytes < 16 * GI {
1570        512 * MI
1571    } else {
1572        GI
1573    }
1574}
1575
1576// ---------------------------------------------------------------------------
1577// Tests
1578// ---------------------------------------------------------------------------
1579
1580#[cfg(test)]
1581mod tests {
1582    use super::*;
1583    use crate::BinaryTarget;
1584
1585    // -- Parsing tests --
1586
1587    #[test]
1588    fn test_parse_cpu_plain() {
1589        assert_eq!(parse_cpu("1").unwrap(), 1.0);
1590        assert_eq!(parse_cpu("0.5").unwrap(), 0.5);
1591        assert_eq!(parse_cpu("2.0").unwrap(), 2.0);
1592        assert_eq!(parse_cpu("16").unwrap(), 16.0);
1593    }
1594
1595    #[test]
1596    fn test_parse_cpu_millicore() {
1597        assert_eq!(parse_cpu("500m").unwrap(), 0.5);
1598        assert_eq!(parse_cpu("250m").unwrap(), 0.25);
1599        assert_eq!(parse_cpu("1000m").unwrap(), 1.0);
1600        assert_eq!(parse_cpu("100m").unwrap(), 0.1);
1601    }
1602
1603    #[test]
1604    fn test_parse_cpu_invalid() {
1605        assert!(parse_cpu("").is_err());
1606        assert!(parse_cpu("abc").is_err());
1607        assert!(parse_cpu("m").is_err());
1608    }
1609
1610    #[test]
1611    fn test_parse_memory_binary_suffixes() {
1612        assert_eq!(parse_memory_bytes("1Ki").unwrap(), 1024);
1613        assert_eq!(parse_memory_bytes("1Mi").unwrap(), 1024 * 1024);
1614        assert_eq!(parse_memory_bytes("1Gi").unwrap(), 1024 * 1024 * 1024);
1615        assert_eq!(parse_memory_bytes("4Gi").unwrap(), 4 * 1024 * 1024 * 1024);
1616        assert_eq!(parse_memory_bytes("512Mi").unwrap(), 512 * 1024 * 1024);
1617        assert_eq!(
1618            parse_memory_bytes("1Ti").unwrap(),
1619            1024u64 * 1024 * 1024 * 1024
1620        );
1621    }
1622
1623    #[test]
1624    fn test_parse_memory_decimal_suffixes() {
1625        assert_eq!(parse_memory_bytes("1k").unwrap(), 1000);
1626        assert_eq!(parse_memory_bytes("1M").unwrap(), 1_000_000);
1627        assert_eq!(parse_memory_bytes("1G").unwrap(), 1_000_000_000);
1628        assert_eq!(parse_memory_bytes("1T").unwrap(), 1_000_000_000_000);
1629    }
1630
1631    #[test]
1632    fn test_parse_memory_plain_bytes() {
1633        assert_eq!(parse_memory_bytes("1024").unwrap(), 1024);
1634        assert_eq!(parse_memory_bytes("0").unwrap(), 0);
1635    }
1636
1637    #[test]
1638    fn test_parse_memory_invalid() {
1639        assert!(parse_memory_bytes("").is_err());
1640        assert!(parse_memory_bytes("abc").is_err());
1641        assert!(parse_memory_bytes("Gi").is_err());
1642    }
1643
1644    #[test]
1645    fn test_parse_memory_fractional() {
1646        assert_eq!(parse_memory_bytes("0.5Gi").unwrap(), GI / 2);
1647        assert_eq!(parse_memory_bytes("1.5Gi").unwrap(), GI + GI / 2);
1648    }
1649
1650    // -- Catalog lookup tests --
1651
1652    #[test]
1653    fn test_catalog_has_entries_for_all_cloud_platforms() {
1654        assert!(!catalog_for_platform(Platform::Aws).is_empty());
1655        assert!(!catalog_for_platform(Platform::Gcp).is_empty());
1656        assert!(!catalog_for_platform(Platform::Azure).is_empty());
1657    }
1658
1659    #[test]
1660    fn test_catalog_no_entries_for_non_cloud_platforms() {
1661        assert!(catalog_for_platform(Platform::Local).is_empty());
1662        assert!(catalog_for_platform(Platform::Kubernetes).is_empty());
1663    }
1664
1665    #[test]
1666    fn test_find_known_instance_type() {
1667        let spec =
1668            find_instance_type(Platform::Aws, "m7g.2xlarge").expect("should find m7g.2xlarge");
1669        assert_eq!(spec.vcpu, 8);
1670        assert_eq!(spec.memory_bytes, 32 * GI);
1671        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1672    }
1673
1674    #[test]
1675    fn test_find_aws_c8i_nested_virt_instance_type() {
1676        let spec = find_instance_type(Platform::Aws, "c8i.large").expect("should find c8i.large");
1677        assert_eq!(spec.vcpu, 2);
1678        assert_eq!(spec.memory_bytes, 4 * GI);
1679        assert_eq!(spec.family, InstanceFamily::ComputeOptimized);
1680        assert_eq!(spec.architecture, Architecture::X86_64);
1681        assert!(spec.is_nested_virt_capable());
1682    }
1683
1684    #[test]
1685    fn test_find_unknown_instance_type() {
1686        assert!(find_instance_type(Platform::Aws, "nonexistent.xlarge").is_none());
1687    }
1688
1689    #[test]
1690    fn test_find_wrong_platform() {
1691        assert!(find_instance_type(Platform::Gcp, "m7g.2xlarge").is_none());
1692    }
1693
1694    #[test]
1695    fn test_to_machine_profile() {
1696        let spec = find_instance_type(Platform::Aws, "m7g.2xlarge").unwrap();
1697        let profile = spec.to_machine_profile();
1698        assert_eq!(profile.cpu, "8.0");
1699        assert_eq!(profile.memory_bytes, 32 * GI);
1700        assert_eq!(profile.ephemeral_storage_bytes, 20 * GI);
1701        assert!(profile.gpu.is_none());
1702    }
1703
1704    #[test]
1705    fn test_to_machine_profile_with_gpu() {
1706        let spec = find_instance_type(Platform::Aws, "p4d.24xlarge").unwrap();
1707        let profile = spec.to_machine_profile();
1708        let gpu = profile.gpu.as_ref().expect("should have GPU");
1709        assert_eq!(gpu.gpu_type, "nvidia-a100");
1710        assert_eq!(gpu.count, 8);
1711    }
1712
1713    // -- Selection algorithm tests --
1714
1715    #[test]
1716    fn test_select_burstable_for_small_workload() {
1717        let req = WorkloadRequirements {
1718            total_cpu_at_desired: 1.0,
1719            total_memory_bytes_at_desired: 2 * GI,
1720            total_cpu_at_max: 1.0,
1721            total_memory_bytes_at_max: 2 * GI,
1722            max_cpu_per_container: 0.5,
1723            max_memory_per_container: 1 * GI,
1724            max_ephemeral_storage_bytes: 10 * GI,
1725            gpu: None,
1726            architecture: None,
1727            nested_virt: false,
1728        };
1729        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1730        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1731        assert_eq!(spec.family, InstanceFamily::Burstable);
1732    }
1733
1734    #[test]
1735    fn test_selects_smallest_burstable_machine_with_real_headroom() {
1736        let req = WorkloadRequirements {
1737            total_cpu_at_desired: 1.0,
1738            total_memory_bytes_at_desired: 2 * GI,
1739            total_cpu_at_max: 1.0,
1740            total_memory_bytes_at_max: 2 * GI,
1741            max_cpu_per_container: 1.0,
1742            max_memory_per_container: 2 * GI,
1743            max_ephemeral_storage_bytes: 10 * GI,
1744            gpu: None,
1745            architecture: None,
1746            nested_virt: false,
1747        };
1748
1749        let selection = select_instance_type(Platform::Aws, &req).unwrap();
1750
1751        assert_eq!(selection.instance_type, "t4g.medium");
1752        assert_eq!(selection.min_machines, 1);
1753        assert_eq!(selection.max_machines, 1);
1754    }
1755
1756    #[test]
1757    fn test_select_general_purpose_for_standard_workload() {
1758        // Standard workloads always get GeneralPurpose regardless of CPU:memory ratio
1759        let req = WorkloadRequirements {
1760            total_cpu_at_desired: 20.0,
1761            total_memory_bytes_at_desired: 80 * GI,
1762            total_cpu_at_max: 20.0,
1763            total_memory_bytes_at_max: 80 * GI,
1764            max_cpu_per_container: 2.0,
1765            max_memory_per_container: 8 * GI,
1766            max_ephemeral_storage_bytes: 10 * GI,
1767            gpu: None,
1768            architecture: None,
1769            nested_virt: false,
1770        };
1771        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1772        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1773        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1774    }
1775
1776    #[test]
1777    fn test_select_general_purpose_even_for_cpu_heavy() {
1778        // CPU-heavy workloads still get GeneralPurpose (no more ComputeOptimized auto-select)
1779        let req = WorkloadRequirements {
1780            total_cpu_at_desired: 20.0,
1781            total_memory_bytes_at_desired: 20 * GI,
1782            total_cpu_at_max: 20.0,
1783            total_memory_bytes_at_max: 20 * GI,
1784            max_cpu_per_container: 2.0,
1785            max_memory_per_container: 2 * GI,
1786            max_ephemeral_storage_bytes: 10 * GI,
1787            gpu: None,
1788            architecture: None,
1789            nested_virt: false,
1790        };
1791        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1792        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1793        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1794    }
1795
1796    #[test]
1797    fn test_select_storage_optimized_for_large_ephemeral() {
1798        let req = WorkloadRequirements {
1799            total_cpu_at_desired: 8.0,
1800            total_memory_bytes_at_desired: 32 * GI,
1801            total_cpu_at_max: 8.0,
1802            total_memory_bytes_at_max: 32 * GI,
1803            max_cpu_per_container: 2.0,
1804            max_memory_per_container: 8 * GI,
1805            max_ephemeral_storage_bytes: 500 * GI,
1806            gpu: None,
1807            architecture: Some(Architecture::X86_64),
1808            nested_virt: false,
1809        };
1810        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1811        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1812        assert_eq!(spec.family, InstanceFamily::StorageOptimized);
1813    }
1814
1815    #[test]
1816    fn test_select_configurable_storage_above_fixed_local_catalog() {
1817        let req = WorkloadRequirements {
1818            total_cpu_at_desired: 8.0,
1819            total_memory_bytes_at_desired: 32 * GI,
1820            total_cpu_at_max: 8.0,
1821            total_memory_bytes_at_max: 32 * GI,
1822            max_cpu_per_container: 2.0,
1823            max_memory_per_container: 8 * GI,
1824            max_ephemeral_storage_bytes: 8_000 * GI,
1825            gpu: None,
1826            architecture: Some(Architecture::X86_64),
1827            nested_virt: false,
1828        };
1829        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1830        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1831        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1832        assert_eq!(sel.profile.ephemeral_storage_bytes, 8_000 * GI);
1833    }
1834
1835    #[test]
1836    fn test_configurable_storage_fallback_retains_standard_vcpu_cap() {
1837        let req = WorkloadRequirements {
1838            total_cpu_at_desired: 70.0,
1839            total_memory_bytes_at_desired: 140 * GI,
1840            total_cpu_at_max: 70.0,
1841            total_memory_bytes_at_max: 140 * GI,
1842            max_cpu_per_container: 2.0,
1843            max_memory_per_container: 4 * GI,
1844            max_ephemeral_storage_bytes: 8_000 * GI,
1845            gpu: None,
1846            architecture: Some(Architecture::X86_64),
1847            nested_virt: false,
1848        };
1849
1850        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1851        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1852        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1853        assert!(spec.vcpu <= MAX_STANDARD_VCPU);
1854        assert!(sel.max_machines > 1);
1855    }
1856
1857    #[test]
1858    fn test_nested_virtualization_storage_falls_back_to_configurable_disk() {
1859        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
1860            let req = WorkloadRequirements {
1861                total_cpu_at_desired: 2.0,
1862                total_memory_bytes_at_desired: 8 * GI,
1863                total_cpu_at_max: 2.0,
1864                total_memory_bytes_at_max: 8 * GI,
1865                max_cpu_per_container: 2.0,
1866                max_memory_per_container: 8 * GI,
1867                max_ephemeral_storage_bytes: 500 * GI,
1868                gpu: None,
1869                architecture: Some(Architecture::X86_64),
1870                nested_virt: true,
1871            };
1872
1873            let sel = select_instance_type(platform, &req)
1874                .unwrap_or_else(|error| panic!("{platform} should support fallback: {error}"));
1875            let spec = find_instance_type(platform, sel.instance_type).unwrap();
1876            assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1877            assert!(spec.is_nested_virt_capable());
1878            assert_eq!(sel.profile.ephemeral_storage_bytes, 500 * GI);
1879        }
1880    }
1881
1882    #[test]
1883    fn test_configurable_disk_rejects_capacity_above_provider_limit() {
1884        for (platform, requested) in [
1885            (Platform::Aws, 100 * TI),
1886            (Platform::Gcp, 100 * TI),
1887            (Platform::Azure, 8_000 * GI),
1888        ] {
1889            let req = WorkloadRequirements {
1890                total_cpu_at_desired: 2.0,
1891                total_memory_bytes_at_desired: 8 * GI,
1892                total_cpu_at_max: 2.0,
1893                total_memory_bytes_at_max: 8 * GI,
1894                max_cpu_per_container: 2.0,
1895                max_memory_per_container: 8 * GI,
1896                max_ephemeral_storage_bytes: requested,
1897                gpu: None,
1898                architecture: Some(Architecture::X86_64),
1899                nested_virt: true,
1900            };
1901
1902            assert!(select_instance_type(platform, &req).is_err());
1903        }
1904    }
1905
1906    #[test]
1907    fn test_configurable_disk_limits_account_for_controller_rounding() {
1908        for (platform, disk_limit_gib) in [
1909            (Platform::Aws, 64 * 1024),
1910            (Platform::Gcp, 64 * 1024),
1911            (Platform::Azure, 4_095),
1912        ] {
1913            let max = max_configurable_ephemeral_storage_bytes(platform).unwrap();
1914            let materialized_disk_gib = |requested_bytes: u64| {
1915                let requested_gib = requested_bytes.div_ceil(GI);
1916                (requested_gib * 5).div_ceil(4) + 12
1917            };
1918
1919            assert!(materialized_disk_gib(max) <= disk_limit_gib);
1920            assert!(materialized_disk_gib(max + 1) > disk_limit_gib);
1921        }
1922    }
1923
1924    #[test]
1925    fn test_gpu_local_storage_is_not_treated_as_resizable() {
1926        let spec = find_instance_type(Platform::Aws, "g5.xlarge").unwrap();
1927        assert_eq!(spec.family, InstanceFamily::GpuCompute);
1928        assert!(!spec.has_configurable_ephemeral_storage());
1929    }
1930
1931    #[test]
1932    fn test_select_gpu_instance() {
1933        let req = WorkloadRequirements {
1934            total_cpu_at_desired: 8.0,
1935            total_memory_bytes_at_desired: 32 * GI,
1936            total_cpu_at_max: 8.0,
1937            total_memory_bytes_at_max: 32 * GI,
1938            max_cpu_per_container: 4.0,
1939            max_memory_per_container: 16 * GI,
1940            max_ephemeral_storage_bytes: 10 * GI,
1941            gpu: Some(GpuSpec {
1942                gpu_type: "nvidia-a100".to_string(),
1943                count: 1,
1944            }),
1945            architecture: Some(Architecture::X86_64),
1946            nested_virt: false,
1947        };
1948        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1949        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1950        assert_eq!(spec.family, InstanceFamily::GpuCompute);
1951        assert!(spec.gpu.is_some());
1952    }
1953
1954    #[test]
1955    fn test_select_uses_each_cloud_image_target_architecture() {
1956        let req = WorkloadRequirements {
1957            total_cpu_at_desired: 4.0,
1958            total_memory_bytes_at_desired: 16 * GI,
1959            total_cpu_at_max: 4.0,
1960            total_memory_bytes_at_max: 16 * GI,
1961            max_cpu_per_container: 1.0,
1962            max_memory_per_container: 4 * GI,
1963            max_ephemeral_storage_bytes: 10 * GI,
1964            gpu: None,
1965            architecture: None,
1966            nested_virt: false,
1967        };
1968        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
1969            let sel = select_instance_type(platform, &req)
1970                .unwrap_or_else(|error| panic!("selection failed for {platform}: {error}"));
1971            let spec = find_instance_type(platform, sel.instance_type)
1972                .expect("selected machine should exist in the catalog");
1973            assert_eq!(
1974                Some(spec.architecture),
1975                default_architecture(platform),
1976                "machine architecture must match the image target for {platform}"
1977            );
1978        }
1979    }
1980
1981    #[test]
1982    fn test_machine_count_reasonable() {
1983        // Single container: 1 CPU, 2Gi, maxReplicas=20
1984        let req = WorkloadRequirements {
1985            total_cpu_at_desired: 20.0,
1986            total_memory_bytes_at_desired: 40 * GI,
1987            total_cpu_at_max: 20.0,
1988            total_memory_bytes_at_max: 40 * GI,
1989            max_cpu_per_container: 1.0,
1990            max_memory_per_container: 2 * GI,
1991            max_ephemeral_storage_bytes: 10 * GI,
1992            gpu: None,
1993            architecture: None,
1994            nested_virt: false,
1995        };
1996        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1997        assert!(sel.min_machines >= 1);
1998        assert!(sel.max_machines <= MAX_MACHINES_PER_CLUSTER);
1999        assert!(sel.max_machines >= sel.min_machines);
2000    }
2001
2002    #[test]
2003    fn test_instance_size_capped_at_8_vcpu() {
2004        // Even with very large containers, instance size is capped at 8 vCPUs
2005        let req = WorkloadRequirements {
2006            total_cpu_at_desired: 70.0,
2007            total_memory_bytes_at_desired: 140 * GI,
2008            total_cpu_at_max: 70.0,
2009            total_memory_bytes_at_max: 140 * GI,
2010            max_cpu_per_container: 2.0,
2011            max_memory_per_container: 4 * GI,
2012            max_ephemeral_storage_bytes: 10 * GI,
2013            gpu: None,
2014            architecture: None,
2015            nested_virt: false,
2016        };
2017        let sel = select_instance_type(Platform::Gcp, &req).unwrap();
2018        let spec = find_instance_type(Platform::Gcp, sel.instance_type).unwrap();
2019        assert!(
2020            spec.vcpu <= MAX_STANDARD_VCPU,
2021            "selected {} with {} vCPUs, expected <= {}",
2022            spec.name,
2023            spec.vcpu,
2024            MAX_STANDARD_VCPU
2025        );
2026        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
2027        // Should scale horizontally instead
2028        assert!(sel.max_machines > 1);
2029    }
2030
2031    #[test]
2032    fn test_larger_autoscaled_workload_gets_reasonable_instance() {
2033        // Simulates a larger autoscaled workload: 4 containers, each 2 CPU / 4 GiB
2034        // maxReplicas: 10, 10, 10, 5
2035        let req = WorkloadRequirements {
2036            total_cpu_at_desired: 70.0,
2037            total_memory_bytes_at_desired: 140 * GI,
2038            total_cpu_at_max: 70.0,              // 2*10 + 2*10 + 2*10 + 2*5
2039            total_memory_bytes_at_max: 140 * GI, // 4*10 + 4*10 + 4*10 + 4*5
2040            max_cpu_per_container: 2.0,
2041            max_memory_per_container: 4 * GI,
2042            max_ephemeral_storage_bytes: 20 * GI,
2043            gpu: None,
2044            architecture: None,
2045            nested_virt: false,
2046        };
2047        let sel = select_instance_type(Platform::Gcp, &req).unwrap();
2048        // Should pick n2-standard-8 (8 vCPU, 32 GiB) — NOT c3-standard-44
2049        assert_eq!(sel.instance_type, "n2-standard-8");
2050        assert!(sel.max_machines >= 2);
2051    }
2052
2053    /// When `nested_virt` is set on the workload, the selector must
2054    /// restrict to nested-virt-capable families. On AWS that means an m8i
2055    /// (or other 8th-gen Intel) entry, never a Graviton (`*7g`, `t4g`) or
2056    /// burstable. Without this filter the launch template gets created
2057    /// with `CpuOptions.NestedVirtualization=enabled` paired with an
2058    /// instance type AWS rejects at RunInstances.
2059    #[test]
2060    fn test_select_aws_picks_m8i_when_nested_virt_required() {
2061        let req = WorkloadRequirements {
2062            total_cpu_at_desired: 4.0,
2063            total_memory_bytes_at_desired: 8 * GI,
2064            total_cpu_at_max: 4.0,
2065            total_memory_bytes_at_max: 8 * GI,
2066            max_cpu_per_container: 4.0,
2067            max_memory_per_container: 8 * GI,
2068            max_ephemeral_storage_bytes: 10 * GI,
2069            gpu: None,
2070            architecture: Some(Architecture::X86_64),
2071            nested_virt: true,
2072        };
2073        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2074        assert!(
2075            sel.instance_type.starts_with("m8i.")
2076                || sel.instance_type.starts_with("c8i.")
2077                || sel.instance_type.starts_with("r8i."),
2078            "expected an m8i/c8i/r8i instance, got {}",
2079            sel.instance_type
2080        );
2081        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
2082        assert!(spec.is_nested_virt_capable());
2083    }
2084
2085    #[test]
2086    fn test_select_gcp_picks_n2_when_nested_virt_required() {
2087        let req = WorkloadRequirements {
2088            total_cpu_at_desired: 4.0,
2089            total_memory_bytes_at_desired: 8 * GI,
2090            total_cpu_at_max: 4.0,
2091            total_memory_bytes_at_max: 8 * GI,
2092            max_cpu_per_container: 4.0,
2093            max_memory_per_container: 8 * GI,
2094            max_ephemeral_storage_bytes: 0,
2095            architecture: Some(Architecture::X86_64),
2096            gpu: None,
2097            nested_virt: true,
2098        };
2099
2100        let selection = select_instance_type(Platform::Gcp, &req).unwrap();
2101        assert_eq!(selection.instance_type, "n2-standard-8");
2102        assert!(find_instance_type(Platform::Gcp, selection.instance_type)
2103            .unwrap()
2104            .is_nested_virt_capable());
2105    }
2106
2107    #[test]
2108    fn test_select_azure_picks_dsv5_when_nested_virt_required() {
2109        let req = WorkloadRequirements {
2110            total_cpu_at_desired: 4.0,
2111            total_memory_bytes_at_desired: 8 * GI,
2112            total_cpu_at_max: 4.0,
2113            total_memory_bytes_at_max: 8 * GI,
2114            max_cpu_per_container: 4.0,
2115            max_memory_per_container: 8 * GI,
2116            max_ephemeral_storage_bytes: 0,
2117            architecture: Some(Architecture::X86_64),
2118            gpu: None,
2119            nested_virt: true,
2120        };
2121
2122        let selection = select_instance_type(Platform::Azure, &req).unwrap();
2123        assert_eq!(selection.instance_type, "Standard_D8s_v5");
2124        assert!(find_instance_type(Platform::Azure, selection.instance_type)
2125            .unwrap()
2126            .is_nested_virt_capable());
2127    }
2128
2129    #[test]
2130    fn test_select_aws_defaults_to_image_target_architecture() {
2131        let req = WorkloadRequirements {
2132            total_cpu_at_desired: 4.0,
2133            total_memory_bytes_at_desired: 8 * GI,
2134            total_cpu_at_max: 4.0,
2135            total_memory_bytes_at_max: 8 * GI,
2136            max_cpu_per_container: 4.0,
2137            max_memory_per_container: 8 * GI,
2138            max_ephemeral_storage_bytes: 10 * GI,
2139            gpu: None,
2140            architecture: None,
2141            nested_virt: false,
2142        };
2143        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2144        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
2145        assert_eq!(spec.architecture, Architecture::Arm64);
2146    }
2147
2148    #[test]
2149    fn test_cloud_defaults_match_image_target_architectures() {
2150        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
2151            let target = BinaryTarget::defaults_for_platform(platform)
2152                .into_iter()
2153                .next()
2154                .expect("managed cloud should have a default image target");
2155            let image_architecture = match target.oci_arch() {
2156                "arm64" => Architecture::Arm64,
2157                "amd64" => Architecture::X86_64,
2158                architecture => {
2159                    panic!("unsupported managed-cloud image architecture {architecture}")
2160                }
2161            };
2162
2163            assert_eq!(default_architecture(platform), Some(image_architecture));
2164        }
2165    }
2166
2167    /// ARM remains available when the workload or capacity profile declares it.
2168    #[test]
2169    fn test_select_aws_uses_graviton_for_explicit_arm64() {
2170        let req = WorkloadRequirements {
2171            total_cpu_at_desired: 4.0,
2172            total_memory_bytes_at_desired: 8 * GI,
2173            total_cpu_at_max: 4.0,
2174            total_memory_bytes_at_max: 8 * GI,
2175            max_cpu_per_container: 4.0,
2176            max_memory_per_container: 8 * GI,
2177            max_ephemeral_storage_bytes: 10 * GI,
2178            gpu: None,
2179            architecture: Some(Architecture::Arm64),
2180            nested_virt: false,
2181        };
2182        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2183        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
2184        assert_eq!(spec.architecture, Architecture::Arm64);
2185    }
2186
2187    #[test]
2188    fn test_select_rejects_explicit_architecture_missing_from_cloud_catalog() {
2189        let req = WorkloadRequirements {
2190            total_cpu_at_desired: 1.0,
2191            total_memory_bytes_at_desired: 2 * GI,
2192            total_cpu_at_max: 1.0,
2193            total_memory_bytes_at_max: 2 * GI,
2194            max_cpu_per_container: 1.0,
2195            max_memory_per_container: 2 * GI,
2196            max_ephemeral_storage_bytes: 10 * GI,
2197            gpu: None,
2198            architecture: Some(Architecture::Arm64),
2199            nested_virt: false,
2200        };
2201
2202        let error = select_instance_type(Platform::Gcp, &req)
2203            .expect_err("GCP catalog has no ARM64 machine");
2204
2205        assert!(error.contains("architecture Arm64 is unavailable"));
2206    }
2207
2208    #[test]
2209    fn test_profile_has_required_fields() {
2210        let req = WorkloadRequirements {
2211            total_cpu_at_desired: 4.0,
2212            total_memory_bytes_at_desired: 16 * GI,
2213            total_cpu_at_max: 4.0,
2214            total_memory_bytes_at_max: 16 * GI,
2215            max_cpu_per_container: 1.0,
2216            max_memory_per_container: 4 * GI,
2217            max_ephemeral_storage_bytes: 10 * GI,
2218            gpu: None,
2219            architecture: None,
2220            nested_virt: false,
2221        };
2222        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2223        assert!(!sel.profile.cpu.is_empty());
2224        assert!(sel.profile.memory_bytes > 0);
2225        assert!(sel.profile.ephemeral_storage_bytes > 0);
2226    }
2227
2228    #[test]
2229    fn test_error_for_unsupported_gpu_type() {
2230        let req = WorkloadRequirements {
2231            total_cpu_at_desired: 8.0,
2232            total_memory_bytes_at_desired: 32 * GI,
2233            total_cpu_at_max: 8.0,
2234            total_memory_bytes_at_max: 32 * GI,
2235            max_cpu_per_container: 4.0,
2236            max_memory_per_container: 16 * GI,
2237            max_ephemeral_storage_bytes: 10 * GI,
2238            gpu: Some(GpuSpec {
2239                gpu_type: "amd-mi300".to_string(),
2240                count: 1,
2241            }),
2242            architecture: None,
2243            nested_virt: false,
2244        };
2245        let result = select_instance_type(Platform::Aws, &req);
2246        assert!(result.is_err());
2247    }
2248
2249    #[test]
2250    fn test_catalog_instance_types_sorted_by_vcpu_within_family() {
2251        // Verify that within each (platform, family) group, vcpu is non-decreasing.
2252        // This ensures our "min_by_key(vcpu)" logic works correctly.
2253        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
2254            let entries = catalog_for_platform(platform);
2255            let mut by_family: std::collections::HashMap<_, Vec<_>> =
2256                std::collections::HashMap::new();
2257            for entry in entries {
2258                by_family
2259                    .entry(format!("{:?}", entry.family))
2260                    .or_default()
2261                    .push(entry);
2262            }
2263            for (family, instances) in &by_family {
2264                for window in instances.windows(2) {
2265                    assert!(
2266                        window[0].vcpu <= window[1].vcpu,
2267                        "catalog not sorted by vcpu for {platform}/{family}: {} ({}) > {} ({})",
2268                        window[0].name,
2269                        window[0].vcpu,
2270                        window[1].name,
2271                        window[1].vcpu
2272                    );
2273                }
2274            }
2275        }
2276    }
2277}