Skip to main content

alien_core/
instance_catalog.rs

1//! Instance type catalog and selection algorithm for cloud compute infrastructure.
2//!
3//! This module provides:
4//! - A static catalog of known instance types across AWS, GCP, and Azure
5//! - Resource quantity parsing (CPU strings, Kubernetes-style memory/storage quantities)
6//! - An algorithm to select the optimal instance type for a given workload
7//!
8//! The catalog is the single source of truth for instance type specifications.
9//! It is used by the preflights system to automatically populate `CapacityGroup.instance_type`
10//! and `CapacityGroup.profile` based on the containers in a stack.
11
12use crate::{GpuSpec, MachineProfile, Platform};
13use serde::{Deserialize, Serialize};
14
15// ---------------------------------------------------------------------------
16// Resource quantity parsing
17// ---------------------------------------------------------------------------
18
19/// Parse a CPU quantity string to f64.
20///
21/// Accepts plain numbers ("1", "0.5", "2.0") and millicore suffixes ("500m" = 0.5).
22pub fn parse_cpu(s: &str) -> Result<f64, String> {
23    let s = s.trim();
24    if s.is_empty() {
25        return Err("empty CPU string".to_string());
26    }
27
28    if let Some(millis) = s.strip_suffix('m') {
29        let v: f64 = millis
30            .parse()
31            .map_err(|_| format!("invalid CPU millicore value: '{s}'"))?;
32        Ok(v / 1000.0)
33    } else {
34        s.parse().map_err(|_| format!("invalid CPU value: '{s}'"))
35    }
36}
37
38/// Parse a memory or storage quantity string to bytes.
39///
40/// Supports Kubernetes-style binary suffixes (Ki, Mi, Gi, Ti) and
41/// decimal suffixes (k, M, G, T). Plain numbers are interpreted as bytes.
42pub fn parse_memory_bytes(s: &str) -> Result<u64, String> {
43    let s = s.trim();
44    if s.is_empty() {
45        return Err("empty memory/storage string".to_string());
46    }
47
48    // Binary suffixes (powers of 1024)
49    if let Some(num) = s.strip_suffix("Ti") {
50        let v: f64 = num
51            .parse()
52            .map_err(|_| format!("invalid memory value: '{s}'"))?;
53        return Ok((v * 1024.0 * 1024.0 * 1024.0 * 1024.0) as u64);
54    }
55    if let Some(num) = s.strip_suffix("Gi") {
56        let v: f64 = num
57            .parse()
58            .map_err(|_| format!("invalid memory value: '{s}'"))?;
59        return Ok((v * 1024.0 * 1024.0 * 1024.0) as u64);
60    }
61    if let Some(num) = s.strip_suffix("Mi") {
62        let v: f64 = num
63            .parse()
64            .map_err(|_| format!("invalid memory value: '{s}'"))?;
65        return Ok((v * 1024.0 * 1024.0) as u64);
66    }
67    if let Some(num) = s.strip_suffix("Ki") {
68        let v: f64 = num
69            .parse()
70            .map_err(|_| format!("invalid memory value: '{s}'"))?;
71        return Ok((v * 1024.0) as u64);
72    }
73
74    // Decimal suffixes (powers of 1000)
75    if let Some(num) = s.strip_suffix('T') {
76        let v: f64 = num
77            .parse()
78            .map_err(|_| format!("invalid memory value: '{s}'"))?;
79        return Ok((v * 1_000_000_000_000.0) as u64);
80    }
81    if let Some(num) = s.strip_suffix('G') {
82        let v: f64 = num
83            .parse()
84            .map_err(|_| format!("invalid memory value: '{s}'"))?;
85        return Ok((v * 1_000_000_000.0) as u64);
86    }
87    if let Some(num) = s.strip_suffix('M') {
88        let v: f64 = num
89            .parse()
90            .map_err(|_| format!("invalid memory value: '{s}'"))?;
91        return Ok((v * 1_000_000.0) as u64);
92    }
93    if let Some(num) = s.strip_suffix('k') {
94        let v: f64 = num
95            .parse()
96            .map_err(|_| format!("invalid memory value: '{s}'"))?;
97        return Ok((v * 1000.0) as u64);
98    }
99
100    // Plain bytes
101    s.parse()
102        .map_err(|_| format!("invalid memory value: '{s}'"))
103}
104
105// ---------------------------------------------------------------------------
106// Instance type catalog
107// ---------------------------------------------------------------------------
108
109/// Instance family classification.
110#[derive(Debug, Clone, Copy, PartialEq, Eq)]
111pub enum InstanceFamily {
112    Burstable,
113    GeneralPurpose,
114    ComputeOptimized,
115    MemoryOptimized,
116    StorageOptimized,
117    GpuCompute,
118}
119
120/// CPU architecture.
121#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
122#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
123#[serde(rename_all = "snake_case")]
124pub enum Architecture {
125    Arm64,
126    X86_64,
127}
128
129/// Default machine architecture for images built for a managed cloud.
130pub fn default_architecture(platform: Platform) -> Option<Architecture> {
131    match platform {
132        Platform::Aws => Some(Architecture::Arm64),
133        Platform::Gcp | Platform::Azure => Some(Architecture::X86_64),
134        Platform::Kubernetes | Platform::Machines | Platform::Local | Platform::Test => None,
135    }
136}
137
138/// Static GPU specification for catalog entries (no heap allocation).
139#[derive(Debug, Clone, Copy, PartialEq, Eq)]
140pub struct CatalogGpu {
141    pub gpu_type: &'static str,
142    pub count: u32,
143}
144
145/// A known instance type with its hardware specifications.
146///
147/// All fields are compile-time constants. The catalog is a flat array of these.
148#[derive(Debug, Clone)]
149pub struct InstanceTypeSpec {
150    pub name: &'static str,
151    pub platform: Platform,
152    pub family: InstanceFamily,
153    pub architecture: Architecture,
154    /// vCPU count (hardware total)
155    pub vcpu: u32,
156    /// Memory in bytes (hardware total)
157    pub memory_bytes: u64,
158    /// Ephemeral storage in bytes (hardware total, NVMe for storage-optimized)
159    pub ephemeral_storage_bytes: u64,
160    /// GPU specification (for GPU instances)
161    pub gpu: Option<CatalogGpu>,
162}
163
164impl InstanceTypeSpec {
165    /// Whether ephemeral storage is a provider disk that can be sized at
166    /// deployment time instead of fixed local instance storage.
167    pub fn has_configurable_ephemeral_storage(&self) -> bool {
168        matches!(
169            self.family,
170            InstanceFamily::Burstable
171                | InstanceFamily::GeneralPurpose
172                | InstanceFamily::ComputeOptimized
173                | InstanceFamily::MemoryOptimized
174        )
175    }
176
177    /// Whether this instance type supports nested virtualization.
178    ///
179    /// Classify by documented provider families rather than adding a flag to
180    /// every catalog row. GCP still requires the instance template to opt in;
181    /// Azure exposes the capability automatically on supported VM sizes.
182    pub fn is_nested_virt_capable(&self) -> bool {
183        match self.platform {
184            Platform::Aws => {
185                let name = self.name;
186                name.starts_with("m8i.")
187                    || name.starts_with("c8i.")
188                    || name.starts_with("r8i.")
189                    || name.starts_with("m8i-flex.")
190                    || name.starts_with("c8i-flex.")
191                    || name.starts_with("r8i-flex.")
192            }
193            Platform::Gcp => self.name.starts_with("n2-standard-"),
194            Platform::Azure => {
195                let name = self.name;
196                (name.starts_with("Standard_D") && name.ends_with("s_v5"))
197                    || (name.starts_with("Standard_E") && name.ends_with("s_v5"))
198                    || (name.starts_with("Standard_F") && name.ends_with("s_v2"))
199            }
200            _ => false,
201        }
202    }
203
204    /// Convert this catalog entry into a `MachineProfile` for use in `CapacityGroup`.
205    pub fn to_machine_profile(&self) -> MachineProfile {
206        MachineProfile {
207            cpu: format!("{}.0", self.vcpu),
208            memory_bytes: self.memory_bytes,
209            ephemeral_storage_bytes: self.ephemeral_storage_bytes,
210            architecture: Some(self.architecture),
211            gpu: self.gpu.map(|g| GpuSpec {
212                gpu_type: g.gpu_type.to_string(),
213                count: g.count,
214            }),
215        }
216    }
217
218    /// Convert this entry to the profile controllers should provision for a
219    /// workload. Configurable cloud disks grow to the requested capacity;
220    /// fixed local disks retain their catalog capacity.
221    pub fn to_machine_profile_for_storage(&self, requested_bytes: u64) -> MachineProfile {
222        let mut profile = self.to_machine_profile();
223        if self.has_configurable_ephemeral_storage() {
224            profile.ephemeral_storage_bytes = profile.ephemeral_storage_bytes.max(requested_bytes);
225        }
226        profile
227    }
228}
229
230// Helpers for readable byte constants
231const KI: u64 = 1024;
232const MI: u64 = KI * 1024;
233const GI: u64 = MI * 1024;
234const TI: u64 = GI * 1024;
235
236/// Maximum workload storage that the current cloud controllers can turn into
237/// their provider-backed root disk after adding 25% filesystem headroom and
238/// 12 GiB of runtime overhead. Keep these bounds in sync with the provider
239/// controller contract; rejecting here is preferable to failing after cloud
240/// resources have already been created.
241pub fn max_configurable_ephemeral_storage_bytes(platform: Platform) -> Option<u64> {
242    let disk_limit_gib = match platform {
243        // AWS gp3 and GCP persistent disks currently top out at 64 TiB.
244        Platform::Aws | Platform::Gcp => 64 * 1024,
245        // Azure managed OS disks currently top out at 4,095 GiB.
246        Platform::Azure => 4_095,
247        Platform::Kubernetes | Platform::Machines | Platform::Local | Platform::Test => {
248            return None;
249        }
250    };
251
252    // Controllers first round the requested bytes up to a whole GiB, then
253    // round the 25% headroom up again. Returning a fractional-GiB bound would
254    // therefore admit `bound + 1 byte` and materialize a disk one GiB over the
255    // provider limit.
256    let max_requested_gib = (disk_limit_gib - 12) * 4 / 5;
257    Some(max_requested_gib * GI)
258}
259
260/// The complete instance type catalog.
261///
262/// This is the single source of truth for instance type specifications.
263/// Update this array when adding support for new instance types.
264///
265/// NOTE: Ephemeral storage values for non-NVMe instances are conservative defaults
266/// (EBS-backed root volumes). Storage-optimized instances list their NVMe capacity.
267static CATALOG: &[InstanceTypeSpec] = &[
268    // =========================================================================
269    // AWS — ARM (Graviton) preferred for cost efficiency
270    // =========================================================================
271
272    // Burstable (t4g — ARM Graviton2)
273    InstanceTypeSpec {
274        name: "t4g.micro",
275        platform: Platform::Aws,
276        family: InstanceFamily::Burstable,
277        architecture: Architecture::Arm64,
278        vcpu: 2,
279        memory_bytes: 1 * GI,
280        ephemeral_storage_bytes: 20 * GI,
281        gpu: None,
282    },
283    InstanceTypeSpec {
284        name: "t4g.small",
285        platform: Platform::Aws,
286        family: InstanceFamily::Burstable,
287        architecture: Architecture::Arm64,
288        vcpu: 2,
289        memory_bytes: 2 * GI,
290        ephemeral_storage_bytes: 20 * GI,
291        gpu: None,
292    },
293    InstanceTypeSpec {
294        name: "t4g.medium",
295        platform: Platform::Aws,
296        family: InstanceFamily::Burstable,
297        architecture: Architecture::Arm64,
298        vcpu: 2,
299        memory_bytes: 4 * GI,
300        ephemeral_storage_bytes: 20 * GI,
301        gpu: None,
302    },
303    InstanceTypeSpec {
304        name: "t4g.large",
305        platform: Platform::Aws,
306        family: InstanceFamily::Burstable,
307        architecture: Architecture::Arm64,
308        vcpu: 2,
309        memory_bytes: 8 * GI,
310        ephemeral_storage_bytes: 20 * GI,
311        gpu: None,
312    },
313    InstanceTypeSpec {
314        name: "t3.xlarge",
315        platform: Platform::Aws,
316        family: InstanceFamily::Burstable,
317        architecture: Architecture::X86_64,
318        vcpu: 4,
319        memory_bytes: 16 * GI,
320        ephemeral_storage_bytes: 20 * GI,
321        gpu: None,
322    },
323    InstanceTypeSpec {
324        name: "t4g.xlarge",
325        platform: Platform::Aws,
326        family: InstanceFamily::Burstable,
327        architecture: Architecture::Arm64,
328        vcpu: 4,
329        memory_bytes: 16 * GI,
330        ephemeral_storage_bytes: 20 * GI,
331        gpu: None,
332    },
333    // General Purpose (m7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
334    InstanceTypeSpec {
335        name: "m7g.medium",
336        platform: Platform::Aws,
337        family: InstanceFamily::GeneralPurpose,
338        architecture: Architecture::Arm64,
339        vcpu: 1,
340        memory_bytes: 4 * GI,
341        ephemeral_storage_bytes: 20 * GI,
342        gpu: None,
343    },
344    InstanceTypeSpec {
345        name: "m7i.large",
346        platform: Platform::Aws,
347        family: InstanceFamily::GeneralPurpose,
348        architecture: Architecture::X86_64,
349        vcpu: 2,
350        memory_bytes: 8 * GI,
351        ephemeral_storage_bytes: 20 * GI,
352        gpu: None,
353    },
354    InstanceTypeSpec {
355        name: "m7g.large",
356        platform: Platform::Aws,
357        family: InstanceFamily::GeneralPurpose,
358        architecture: Architecture::Arm64,
359        vcpu: 2,
360        memory_bytes: 8 * GI,
361        ephemeral_storage_bytes: 20 * GI,
362        gpu: None,
363    },
364    // 8th-gen Intel AWS families accept
365    // `CpuOptions.NestedVirtualization=enabled`. The catalog filter in
366    // `select_instance_type` includes these entries only when the
367    // workload requests nested virt, so ordinary workloads continue to
368    // pick the cost-efficient Graviton (m7g) above. The pairwise
369    // interleave keeps the per-family vCPU-non-decreasing invariant
370    // (see `test_catalog_instance_types_sorted_by_vcpu_within_family`).
371    InstanceTypeSpec {
372        name: "m8i.large",
373        platform: Platform::Aws,
374        family: InstanceFamily::GeneralPurpose,
375        architecture: Architecture::X86_64,
376        vcpu: 2,
377        memory_bytes: 8 * GI,
378        ephemeral_storage_bytes: 20 * GI,
379        gpu: None,
380    },
381    InstanceTypeSpec {
382        name: "m7i.xlarge",
383        platform: Platform::Aws,
384        family: InstanceFamily::GeneralPurpose,
385        architecture: Architecture::X86_64,
386        vcpu: 4,
387        memory_bytes: 16 * GI,
388        ephemeral_storage_bytes: 20 * GI,
389        gpu: None,
390    },
391    InstanceTypeSpec {
392        name: "m7g.xlarge",
393        platform: Platform::Aws,
394        family: InstanceFamily::GeneralPurpose,
395        architecture: Architecture::Arm64,
396        vcpu: 4,
397        memory_bytes: 16 * GI,
398        ephemeral_storage_bytes: 20 * GI,
399        gpu: None,
400    },
401    InstanceTypeSpec {
402        name: "m8i.xlarge",
403        platform: Platform::Aws,
404        family: InstanceFamily::GeneralPurpose,
405        architecture: Architecture::X86_64,
406        vcpu: 4,
407        memory_bytes: 16 * GI,
408        ephemeral_storage_bytes: 20 * GI,
409        gpu: None,
410    },
411    InstanceTypeSpec {
412        name: "m7i.2xlarge",
413        platform: Platform::Aws,
414        family: InstanceFamily::GeneralPurpose,
415        architecture: Architecture::X86_64,
416        vcpu: 8,
417        memory_bytes: 32 * GI,
418        ephemeral_storage_bytes: 20 * GI,
419        gpu: None,
420    },
421    InstanceTypeSpec {
422        name: "m7g.2xlarge",
423        platform: Platform::Aws,
424        family: InstanceFamily::GeneralPurpose,
425        architecture: Architecture::Arm64,
426        vcpu: 8,
427        memory_bytes: 32 * GI,
428        ephemeral_storage_bytes: 20 * GI,
429        gpu: None,
430    },
431    InstanceTypeSpec {
432        name: "m8i.2xlarge",
433        platform: Platform::Aws,
434        family: InstanceFamily::GeneralPurpose,
435        architecture: Architecture::X86_64,
436        vcpu: 8,
437        memory_bytes: 32 * GI,
438        ephemeral_storage_bytes: 20 * GI,
439        gpu: None,
440    },
441    InstanceTypeSpec {
442        name: "m7i.4xlarge",
443        platform: Platform::Aws,
444        family: InstanceFamily::GeneralPurpose,
445        architecture: Architecture::X86_64,
446        vcpu: 16,
447        memory_bytes: 64 * GI,
448        ephemeral_storage_bytes: 20 * GI,
449        gpu: None,
450    },
451    InstanceTypeSpec {
452        name: "m7g.4xlarge",
453        platform: Platform::Aws,
454        family: InstanceFamily::GeneralPurpose,
455        architecture: Architecture::Arm64,
456        vcpu: 16,
457        memory_bytes: 64 * GI,
458        ephemeral_storage_bytes: 20 * GI,
459        gpu: None,
460    },
461    InstanceTypeSpec {
462        name: "m8i.4xlarge",
463        platform: Platform::Aws,
464        family: InstanceFamily::GeneralPurpose,
465        architecture: Architecture::X86_64,
466        vcpu: 16,
467        memory_bytes: 64 * GI,
468        ephemeral_storage_bytes: 20 * GI,
469        gpu: None,
470    },
471    // Compute Optimized (c7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
472    InstanceTypeSpec {
473        name: "c7g.medium",
474        platform: Platform::Aws,
475        family: InstanceFamily::ComputeOptimized,
476        architecture: Architecture::Arm64,
477        vcpu: 1,
478        memory_bytes: 2 * GI,
479        ephemeral_storage_bytes: 20 * GI,
480        gpu: None,
481    },
482    InstanceTypeSpec {
483        name: "c7g.large",
484        platform: Platform::Aws,
485        family: InstanceFamily::ComputeOptimized,
486        architecture: Architecture::Arm64,
487        vcpu: 2,
488        memory_bytes: 4 * GI,
489        ephemeral_storage_bytes: 20 * GI,
490        gpu: None,
491    },
492    InstanceTypeSpec {
493        name: "c8i.large",
494        platform: Platform::Aws,
495        family: InstanceFamily::ComputeOptimized,
496        architecture: Architecture::X86_64,
497        vcpu: 2,
498        memory_bytes: 4 * GI,
499        ephemeral_storage_bytes: 20 * GI,
500        gpu: None,
501    },
502    InstanceTypeSpec {
503        name: "c7g.xlarge",
504        platform: Platform::Aws,
505        family: InstanceFamily::ComputeOptimized,
506        architecture: Architecture::Arm64,
507        vcpu: 4,
508        memory_bytes: 8 * GI,
509        ephemeral_storage_bytes: 20 * GI,
510        gpu: None,
511    },
512    InstanceTypeSpec {
513        name: "c8i.xlarge",
514        platform: Platform::Aws,
515        family: InstanceFamily::ComputeOptimized,
516        architecture: Architecture::X86_64,
517        vcpu: 4,
518        memory_bytes: 8 * GI,
519        ephemeral_storage_bytes: 20 * GI,
520        gpu: None,
521    },
522    InstanceTypeSpec {
523        name: "c7g.2xlarge",
524        platform: Platform::Aws,
525        family: InstanceFamily::ComputeOptimized,
526        architecture: Architecture::Arm64,
527        vcpu: 8,
528        memory_bytes: 16 * GI,
529        ephemeral_storage_bytes: 20 * GI,
530        gpu: None,
531    },
532    InstanceTypeSpec {
533        name: "c8i.2xlarge",
534        platform: Platform::Aws,
535        family: InstanceFamily::ComputeOptimized,
536        architecture: Architecture::X86_64,
537        vcpu: 8,
538        memory_bytes: 16 * GI,
539        ephemeral_storage_bytes: 20 * GI,
540        gpu: None,
541    },
542    InstanceTypeSpec {
543        name: "c7g.4xlarge",
544        platform: Platform::Aws,
545        family: InstanceFamily::ComputeOptimized,
546        architecture: Architecture::Arm64,
547        vcpu: 16,
548        memory_bytes: 32 * GI,
549        ephemeral_storage_bytes: 20 * GI,
550        gpu: None,
551    },
552    InstanceTypeSpec {
553        name: "c8i.4xlarge",
554        platform: Platform::Aws,
555        family: InstanceFamily::ComputeOptimized,
556        architecture: Architecture::X86_64,
557        vcpu: 16,
558        memory_bytes: 32 * GI,
559        ephemeral_storage_bytes: 20 * GI,
560        gpu: None,
561    },
562    // Memory Optimized (r7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
563    InstanceTypeSpec {
564        name: "r7g.medium",
565        platform: Platform::Aws,
566        family: InstanceFamily::MemoryOptimized,
567        architecture: Architecture::Arm64,
568        vcpu: 1,
569        memory_bytes: 8 * GI,
570        ephemeral_storage_bytes: 20 * GI,
571        gpu: None,
572    },
573    InstanceTypeSpec {
574        name: "r7g.large",
575        platform: Platform::Aws,
576        family: InstanceFamily::MemoryOptimized,
577        architecture: Architecture::Arm64,
578        vcpu: 2,
579        memory_bytes: 16 * GI,
580        ephemeral_storage_bytes: 20 * GI,
581        gpu: None,
582    },
583    InstanceTypeSpec {
584        name: "r7g.xlarge",
585        platform: Platform::Aws,
586        family: InstanceFamily::MemoryOptimized,
587        architecture: Architecture::Arm64,
588        vcpu: 4,
589        memory_bytes: 32 * GI,
590        ephemeral_storage_bytes: 20 * GI,
591        gpu: None,
592    },
593    InstanceTypeSpec {
594        name: "r7g.2xlarge",
595        platform: Platform::Aws,
596        family: InstanceFamily::MemoryOptimized,
597        architecture: Architecture::Arm64,
598        vcpu: 8,
599        memory_bytes: 64 * GI,
600        ephemeral_storage_bytes: 20 * GI,
601        gpu: None,
602    },
603    InstanceTypeSpec {
604        name: "r7g.4xlarge",
605        platform: Platform::Aws,
606        family: InstanceFamily::MemoryOptimized,
607        architecture: Architecture::Arm64,
608        vcpu: 16,
609        memory_bytes: 128 * GI,
610        ephemeral_storage_bytes: 20 * GI,
611        gpu: None,
612    },
613    // Storage Optimized (i4i — x86_64, NVMe)
614    InstanceTypeSpec {
615        name: "i4i.xlarge",
616        platform: Platform::Aws,
617        family: InstanceFamily::StorageOptimized,
618        architecture: Architecture::X86_64,
619        vcpu: 4,
620        memory_bytes: 32 * GI,
621        ephemeral_storage_bytes: 937 * GI,
622        gpu: None,
623    },
624    InstanceTypeSpec {
625        name: "i4i.2xlarge",
626        platform: Platform::Aws,
627        family: InstanceFamily::StorageOptimized,
628        architecture: Architecture::X86_64,
629        vcpu: 8,
630        memory_bytes: 64 * GI,
631        ephemeral_storage_bytes: 1875 * GI,
632        gpu: None,
633    },
634    InstanceTypeSpec {
635        name: "i4i.4xlarge",
636        platform: Platform::Aws,
637        family: InstanceFamily::StorageOptimized,
638        architecture: Architecture::X86_64,
639        vcpu: 16,
640        memory_bytes: 128 * GI,
641        ephemeral_storage_bytes: 3750 * GI,
642        gpu: None,
643    },
644    InstanceTypeSpec {
645        name: "i4i.8xlarge",
646        platform: Platform::Aws,
647        family: InstanceFamily::StorageOptimized,
648        architecture: Architecture::X86_64,
649        vcpu: 32,
650        memory_bytes: 256 * GI,
651        ephemeral_storage_bytes: 7500 * GI,
652        gpu: None,
653    },
654    // GPU — NVIDIA T4 (g5 — x86_64)
655    InstanceTypeSpec {
656        name: "g5.xlarge",
657        platform: Platform::Aws,
658        family: InstanceFamily::GpuCompute,
659        architecture: Architecture::X86_64,
660        vcpu: 4,
661        memory_bytes: 16 * GI,
662        ephemeral_storage_bytes: 250 * GI,
663        gpu: Some(CatalogGpu {
664            gpu_type: "nvidia-t4",
665            count: 1,
666        }),
667    },
668    InstanceTypeSpec {
669        name: "g5.2xlarge",
670        platform: Platform::Aws,
671        family: InstanceFamily::GpuCompute,
672        architecture: Architecture::X86_64,
673        vcpu: 8,
674        memory_bytes: 32 * GI,
675        ephemeral_storage_bytes: 450 * GI,
676        gpu: Some(CatalogGpu {
677            gpu_type: "nvidia-t4",
678            count: 1,
679        }),
680    },
681    // GPU — NVIDIA A100 (p4d — x86_64)
682    InstanceTypeSpec {
683        name: "p4d.24xlarge",
684        platform: Platform::Aws,
685        family: InstanceFamily::GpuCompute,
686        architecture: Architecture::X86_64,
687        vcpu: 96,
688        memory_bytes: 1152 * GI,
689        ephemeral_storage_bytes: 8000 * GI,
690        gpu: Some(CatalogGpu {
691            gpu_type: "nvidia-a100",
692            count: 8,
693        }),
694    },
695    // GPU — NVIDIA H100 (p5 — x86_64)
696    InstanceTypeSpec {
697        name: "p5.48xlarge",
698        platform: Platform::Aws,
699        family: InstanceFamily::GpuCompute,
700        architecture: Architecture::X86_64,
701        vcpu: 192,
702        memory_bytes: 2048 * GI,
703        ephemeral_storage_bytes: 8000 * GI,
704        gpu: Some(CatalogGpu {
705            gpu_type: "nvidia-h100",
706            count: 8,
707        }),
708    },
709    // =========================================================================
710    // GCP
711    // =========================================================================
712
713    // Burstable (e2)
714    InstanceTypeSpec {
715        name: "e2-micro",
716        platform: Platform::Gcp,
717        family: InstanceFamily::Burstable,
718        architecture: Architecture::X86_64,
719        vcpu: 2,
720        memory_bytes: 1 * GI,
721        ephemeral_storage_bytes: 20 * GI,
722        gpu: None,
723    },
724    InstanceTypeSpec {
725        name: "e2-small",
726        platform: Platform::Gcp,
727        family: InstanceFamily::Burstable,
728        architecture: Architecture::X86_64,
729        vcpu: 2,
730        memory_bytes: 2 * GI,
731        ephemeral_storage_bytes: 20 * GI,
732        gpu: None,
733    },
734    InstanceTypeSpec {
735        name: "e2-medium",
736        platform: Platform::Gcp,
737        family: InstanceFamily::Burstable,
738        architecture: Architecture::X86_64,
739        vcpu: 2,
740        memory_bytes: 4 * GI,
741        ephemeral_storage_bytes: 20 * GI,
742        gpu: None,
743    },
744    // General Purpose (n2-standard, up to 16 vCPU)
745    InstanceTypeSpec {
746        name: "n2-standard-2",
747        platform: Platform::Gcp,
748        family: InstanceFamily::GeneralPurpose,
749        architecture: Architecture::X86_64,
750        vcpu: 2,
751        memory_bytes: 8 * GI,
752        ephemeral_storage_bytes: 20 * GI,
753        gpu: None,
754    },
755    InstanceTypeSpec {
756        name: "n2-standard-4",
757        platform: Platform::Gcp,
758        family: InstanceFamily::GeneralPurpose,
759        architecture: Architecture::X86_64,
760        vcpu: 4,
761        memory_bytes: 16 * GI,
762        ephemeral_storage_bytes: 20 * GI,
763        gpu: None,
764    },
765    InstanceTypeSpec {
766        name: "n2-standard-8",
767        platform: Platform::Gcp,
768        family: InstanceFamily::GeneralPurpose,
769        architecture: Architecture::X86_64,
770        vcpu: 8,
771        memory_bytes: 32 * GI,
772        ephemeral_storage_bytes: 20 * GI,
773        gpu: None,
774    },
775    InstanceTypeSpec {
776        name: "n2-standard-16",
777        platform: Platform::Gcp,
778        family: InstanceFamily::GeneralPurpose,
779        architecture: Architecture::X86_64,
780        vcpu: 16,
781        memory_bytes: 64 * GI,
782        ephemeral_storage_bytes: 20 * GI,
783        gpu: None,
784    },
785    // Compute Optimized (c3-standard, up to 8 vCPU)
786    InstanceTypeSpec {
787        name: "c3-standard-4",
788        platform: Platform::Gcp,
789        family: InstanceFamily::ComputeOptimized,
790        architecture: Architecture::X86_64,
791        vcpu: 4,
792        memory_bytes: 8 * GI,
793        ephemeral_storage_bytes: 20 * GI,
794        gpu: None,
795    },
796    InstanceTypeSpec {
797        name: "c3-standard-8",
798        platform: Platform::Gcp,
799        family: InstanceFamily::ComputeOptimized,
800        architecture: Architecture::X86_64,
801        vcpu: 8,
802        memory_bytes: 16 * GI,
803        ephemeral_storage_bytes: 20 * GI,
804        gpu: None,
805    },
806    // Memory Optimized (n2-highmem, up to 8 vCPU)
807    InstanceTypeSpec {
808        name: "n2-highmem-2",
809        platform: Platform::Gcp,
810        family: InstanceFamily::MemoryOptimized,
811        architecture: Architecture::X86_64,
812        vcpu: 2,
813        memory_bytes: 16 * GI,
814        ephemeral_storage_bytes: 20 * GI,
815        gpu: None,
816    },
817    InstanceTypeSpec {
818        name: "n2-highmem-4",
819        platform: Platform::Gcp,
820        family: InstanceFamily::MemoryOptimized,
821        architecture: Architecture::X86_64,
822        vcpu: 4,
823        memory_bytes: 32 * GI,
824        ephemeral_storage_bytes: 20 * GI,
825        gpu: None,
826    },
827    InstanceTypeSpec {
828        name: "n2-highmem-8",
829        platform: Platform::Gcp,
830        family: InstanceFamily::MemoryOptimized,
831        architecture: Architecture::X86_64,
832        vcpu: 8,
833        memory_bytes: 64 * GI,
834        ephemeral_storage_bytes: 20 * GI,
835        gpu: None,
836    },
837    InstanceTypeSpec {
838        name: "n2-highmem-16",
839        platform: Platform::Gcp,
840        family: InstanceFamily::MemoryOptimized,
841        architecture: Architecture::X86_64,
842        vcpu: 16,
843        memory_bytes: 128 * GI,
844        ephemeral_storage_bytes: 20 * GI,
845        gpu: None,
846    },
847    InstanceTypeSpec {
848        name: "n2-highmem-32",
849        platform: Platform::Gcp,
850        family: InstanceFamily::MemoryOptimized,
851        architecture: Architecture::X86_64,
852        vcpu: 32,
853        memory_bytes: 256 * GI,
854        ephemeral_storage_bytes: 20 * GI,
855        gpu: None,
856    },
857    // Storage Optimized (c3d-standard with local SSD)
858    InstanceTypeSpec {
859        name: "c3d-standard-8",
860        platform: Platform::Gcp,
861        family: InstanceFamily::StorageOptimized,
862        architecture: Architecture::X86_64,
863        vcpu: 8,
864        memory_bytes: 32 * GI,
865        ephemeral_storage_bytes: 480 * GI,
866        gpu: None,
867    },
868    InstanceTypeSpec {
869        name: "c3d-standard-16",
870        platform: Platform::Gcp,
871        family: InstanceFamily::StorageOptimized,
872        architecture: Architecture::X86_64,
873        vcpu: 16,
874        memory_bytes: 64 * GI,
875        ephemeral_storage_bytes: 960 * GI,
876        gpu: None,
877    },
878    InstanceTypeSpec {
879        name: "c3d-standard-30",
880        platform: Platform::Gcp,
881        family: InstanceFamily::StorageOptimized,
882        architecture: Architecture::X86_64,
883        vcpu: 30,
884        memory_bytes: 120 * GI,
885        ephemeral_storage_bytes: 1920 * GI,
886        gpu: None,
887    },
888    // GPU — NVIDIA T4 (n1-standard + T4)
889    InstanceTypeSpec {
890        name: "n1-standard-4-t4",
891        platform: Platform::Gcp,
892        family: InstanceFamily::GpuCompute,
893        architecture: Architecture::X86_64,
894        vcpu: 4,
895        memory_bytes: 15 * GI,
896        ephemeral_storage_bytes: 100 * GI,
897        gpu: Some(CatalogGpu {
898            gpu_type: "nvidia-t4",
899            count: 1,
900        }),
901    },
902    // GPU — NVIDIA A100 (a2-highgpu)
903    InstanceTypeSpec {
904        name: "a2-highgpu-1g",
905        platform: Platform::Gcp,
906        family: InstanceFamily::GpuCompute,
907        architecture: Architecture::X86_64,
908        vcpu: 12,
909        memory_bytes: 85 * GI,
910        ephemeral_storage_bytes: 100 * GI,
911        gpu: Some(CatalogGpu {
912            gpu_type: "nvidia-a100",
913            count: 1,
914        }),
915    },
916    InstanceTypeSpec {
917        name: "a2-highgpu-8g",
918        platform: Platform::Gcp,
919        family: InstanceFamily::GpuCompute,
920        architecture: Architecture::X86_64,
921        vcpu: 96,
922        memory_bytes: 1360 * GI,
923        ephemeral_storage_bytes: 100 * GI,
924        gpu: Some(CatalogGpu {
925            gpu_type: "nvidia-a100",
926            count: 8,
927        }),
928    },
929    // GPU — NVIDIA H100 (a3-highgpu)
930    InstanceTypeSpec {
931        name: "a3-highgpu-8g",
932        platform: Platform::Gcp,
933        family: InstanceFamily::GpuCompute,
934        architecture: Architecture::X86_64,
935        vcpu: 208,
936        memory_bytes: 1872 * GI,
937        ephemeral_storage_bytes: 100 * GI,
938        gpu: Some(CatalogGpu {
939            gpu_type: "nvidia-h100",
940            count: 8,
941        }),
942    },
943    // =========================================================================
944    // Azure
945    // =========================================================================
946
947    // Burstable (B-series v2)
948    InstanceTypeSpec {
949        name: "Standard_B1s",
950        platform: Platform::Azure,
951        family: InstanceFamily::Burstable,
952        architecture: Architecture::X86_64,
953        vcpu: 1,
954        memory_bytes: 1 * GI,
955        ephemeral_storage_bytes: 20 * GI,
956        gpu: None,
957    },
958    InstanceTypeSpec {
959        name: "Standard_B2s",
960        platform: Platform::Azure,
961        family: InstanceFamily::Burstable,
962        architecture: Architecture::X86_64,
963        vcpu: 2,
964        memory_bytes: 4 * GI,
965        ephemeral_storage_bytes: 20 * GI,
966        gpu: None,
967    },
968    InstanceTypeSpec {
969        name: "Standard_B2ms",
970        platform: Platform::Azure,
971        family: InstanceFamily::Burstable,
972        architecture: Architecture::X86_64,
973        vcpu: 2,
974        memory_bytes: 8 * GI,
975        ephemeral_storage_bytes: 20 * GI,
976        gpu: None,
977    },
978    InstanceTypeSpec {
979        name: "Standard_B4ms",
980        platform: Platform::Azure,
981        family: InstanceFamily::Burstable,
982        architecture: Architecture::X86_64,
983        vcpu: 4,
984        memory_bytes: 16 * GI,
985        ephemeral_storage_bytes: 20 * GI,
986        gpu: None,
987    },
988    // General Purpose (Dv5-series, up to 16 vCPU)
989    InstanceTypeSpec {
990        name: "Standard_D2s_v5",
991        platform: Platform::Azure,
992        family: InstanceFamily::GeneralPurpose,
993        architecture: Architecture::X86_64,
994        vcpu: 2,
995        memory_bytes: 8 * GI,
996        ephemeral_storage_bytes: 20 * GI,
997        gpu: None,
998    },
999    InstanceTypeSpec {
1000        name: "Standard_D4s_v5",
1001        platform: Platform::Azure,
1002        family: InstanceFamily::GeneralPurpose,
1003        architecture: Architecture::X86_64,
1004        vcpu: 4,
1005        memory_bytes: 16 * GI,
1006        ephemeral_storage_bytes: 20 * GI,
1007        gpu: None,
1008    },
1009    InstanceTypeSpec {
1010        name: "Standard_D8s_v5",
1011        platform: Platform::Azure,
1012        family: InstanceFamily::GeneralPurpose,
1013        architecture: Architecture::X86_64,
1014        vcpu: 8,
1015        memory_bytes: 32 * GI,
1016        ephemeral_storage_bytes: 20 * GI,
1017        gpu: None,
1018    },
1019    InstanceTypeSpec {
1020        name: "Standard_D16s_v5",
1021        platform: Platform::Azure,
1022        family: InstanceFamily::GeneralPurpose,
1023        architecture: Architecture::X86_64,
1024        vcpu: 16,
1025        memory_bytes: 64 * GI,
1026        ephemeral_storage_bytes: 20 * GI,
1027        gpu: None,
1028    },
1029    // Compute Optimized (Fv2-series, up to 16 vCPU)
1030    InstanceTypeSpec {
1031        name: "Standard_F2s_v2",
1032        platform: Platform::Azure,
1033        family: InstanceFamily::ComputeOptimized,
1034        architecture: Architecture::X86_64,
1035        vcpu: 2,
1036        memory_bytes: 4 * GI,
1037        ephemeral_storage_bytes: 20 * GI,
1038        gpu: None,
1039    },
1040    InstanceTypeSpec {
1041        name: "Standard_F4s_v2",
1042        platform: Platform::Azure,
1043        family: InstanceFamily::ComputeOptimized,
1044        architecture: Architecture::X86_64,
1045        vcpu: 4,
1046        memory_bytes: 8 * GI,
1047        ephemeral_storage_bytes: 20 * GI,
1048        gpu: None,
1049    },
1050    InstanceTypeSpec {
1051        name: "Standard_F8s_v2",
1052        platform: Platform::Azure,
1053        family: InstanceFamily::ComputeOptimized,
1054        architecture: Architecture::X86_64,
1055        vcpu: 8,
1056        memory_bytes: 16 * GI,
1057        ephemeral_storage_bytes: 20 * GI,
1058        gpu: None,
1059    },
1060    InstanceTypeSpec {
1061        name: "Standard_F16s_v2",
1062        platform: Platform::Azure,
1063        family: InstanceFamily::ComputeOptimized,
1064        architecture: Architecture::X86_64,
1065        vcpu: 16,
1066        memory_bytes: 32 * GI,
1067        ephemeral_storage_bytes: 20 * GI,
1068        gpu: None,
1069    },
1070    // Memory Optimized (Ev5-series, up to 16 vCPU)
1071    InstanceTypeSpec {
1072        name: "Standard_E2s_v5",
1073        platform: Platform::Azure,
1074        family: InstanceFamily::MemoryOptimized,
1075        architecture: Architecture::X86_64,
1076        vcpu: 2,
1077        memory_bytes: 16 * GI,
1078        ephemeral_storage_bytes: 20 * GI,
1079        gpu: None,
1080    },
1081    InstanceTypeSpec {
1082        name: "Standard_E4s_v5",
1083        platform: Platform::Azure,
1084        family: InstanceFamily::MemoryOptimized,
1085        architecture: Architecture::X86_64,
1086        vcpu: 4,
1087        memory_bytes: 32 * GI,
1088        ephemeral_storage_bytes: 20 * GI,
1089        gpu: None,
1090    },
1091    InstanceTypeSpec {
1092        name: "Standard_E8s_v5",
1093        platform: Platform::Azure,
1094        family: InstanceFamily::MemoryOptimized,
1095        architecture: Architecture::X86_64,
1096        vcpu: 8,
1097        memory_bytes: 64 * GI,
1098        ephemeral_storage_bytes: 20 * GI,
1099        gpu: None,
1100    },
1101    InstanceTypeSpec {
1102        name: "Standard_E16s_v5",
1103        platform: Platform::Azure,
1104        family: InstanceFamily::MemoryOptimized,
1105        architecture: Architecture::X86_64,
1106        vcpu: 16,
1107        memory_bytes: 128 * GI,
1108        ephemeral_storage_bytes: 20 * GI,
1109        gpu: None,
1110    },
1111    // Storage Optimized (Lsv3-series with NVMe)
1112    InstanceTypeSpec {
1113        name: "Standard_L8s_v3",
1114        platform: Platform::Azure,
1115        family: InstanceFamily::StorageOptimized,
1116        architecture: Architecture::X86_64,
1117        vcpu: 8,
1118        memory_bytes: 64 * GI,
1119        ephemeral_storage_bytes: 1788 * GI,
1120        gpu: None,
1121    },
1122    InstanceTypeSpec {
1123        name: "Standard_L16s_v3",
1124        platform: Platform::Azure,
1125        family: InstanceFamily::StorageOptimized,
1126        architecture: Architecture::X86_64,
1127        vcpu: 16,
1128        memory_bytes: 128 * GI,
1129        ephemeral_storage_bytes: 3576 * GI,
1130        gpu: None,
1131    },
1132    InstanceTypeSpec {
1133        name: "Standard_L32s_v3",
1134        platform: Platform::Azure,
1135        family: InstanceFamily::StorageOptimized,
1136        architecture: Architecture::X86_64,
1137        vcpu: 32,
1138        memory_bytes: 256 * GI,
1139        ephemeral_storage_bytes: 7154 * GI,
1140        gpu: None,
1141    },
1142    // GPU — NVIDIA T4 (NCasT4_v3-series)
1143    InstanceTypeSpec {
1144        name: "Standard_NC4as_T4_v3",
1145        platform: Platform::Azure,
1146        family: InstanceFamily::GpuCompute,
1147        architecture: Architecture::X86_64,
1148        vcpu: 4,
1149        memory_bytes: 28 * GI,
1150        ephemeral_storage_bytes: 176 * GI,
1151        gpu: Some(CatalogGpu {
1152            gpu_type: "nvidia-t4",
1153            count: 1,
1154        }),
1155    },
1156    // GPU — NVIDIA A100 (NC A100 v4-series)
1157    InstanceTypeSpec {
1158        name: "Standard_NC24ads_A100_v4",
1159        platform: Platform::Azure,
1160        family: InstanceFamily::GpuCompute,
1161        architecture: Architecture::X86_64,
1162        vcpu: 24,
1163        memory_bytes: 220 * GI,
1164        ephemeral_storage_bytes: 958 * GI,
1165        gpu: Some(CatalogGpu {
1166            gpu_type: "nvidia-a100",
1167            count: 1,
1168        }),
1169    },
1170    InstanceTypeSpec {
1171        name: "Standard_NC96ads_A100_v4",
1172        platform: Platform::Azure,
1173        family: InstanceFamily::GpuCompute,
1174        architecture: Architecture::X86_64,
1175        vcpu: 96,
1176        memory_bytes: 880 * GI,
1177        ephemeral_storage_bytes: 3916 * GI,
1178        gpu: Some(CatalogGpu {
1179            gpu_type: "nvidia-a100",
1180            count: 4,
1181        }),
1182    },
1183    // GPU — NVIDIA H100 (ND H100 v5-series)
1184    InstanceTypeSpec {
1185        name: "Standard_ND96isr_H100_v5",
1186        platform: Platform::Azure,
1187        family: InstanceFamily::GpuCompute,
1188        architecture: Architecture::X86_64,
1189        vcpu: 96,
1190        memory_bytes: 1900 * GI,
1191        ephemeral_storage_bytes: 1000 * GI,
1192        gpu: Some(CatalogGpu {
1193            gpu_type: "nvidia-h100",
1194            count: 8,
1195        }),
1196    },
1197];
1198
1199// ---------------------------------------------------------------------------
1200// Catalog lookup
1201// ---------------------------------------------------------------------------
1202
1203/// Get all instance types for a given platform.
1204pub fn catalog_for_platform(platform: Platform) -> Vec<&'static InstanceTypeSpec> {
1205    CATALOG
1206        .iter()
1207        .filter(|spec| spec.platform == platform)
1208        .collect()
1209}
1210
1211/// Find a specific instance type by name and platform.
1212pub fn find_instance_type(platform: Platform, name: &str) -> Option<&'static InstanceTypeSpec> {
1213    CATALOG
1214        .iter()
1215        .find(|spec| spec.platform == platform && spec.name == name)
1216}
1217
1218// ---------------------------------------------------------------------------
1219// Instance type selection
1220// ---------------------------------------------------------------------------
1221
1222/// Aggregated resource requirements from all containers in a capacity group.
1223#[derive(Debug, Clone)]
1224pub struct WorkloadRequirements {
1225    /// Total CPU needed at desired scale (sum of desired CPU * desired_replicas per container)
1226    pub total_cpu_at_desired: f64,
1227    /// Total memory needed at desired scale (sum of desired memory * desired_replicas per container)
1228    pub total_memory_bytes_at_desired: u64,
1229    /// Total CPU needed at maximum scale (sum of desired CPU * max_replicas per container)
1230    pub total_cpu_at_max: f64,
1231    /// Total memory needed at maximum scale (sum of desired memory * max_replicas per container)
1232    pub total_memory_bytes_at_max: u64,
1233    /// Largest CPU request among all individual containers (single replica)
1234    pub max_cpu_per_container: f64,
1235    /// Largest memory request among all individual containers (single replica)
1236    pub max_memory_per_container: u64,
1237    /// Maximum ephemeral storage any single container requires
1238    pub max_ephemeral_storage_bytes: u64,
1239    /// GPU requirement (if any container needs GPU)
1240    pub gpu: Option<GpuSpec>,
1241    /// Required CPU architecture, when source explicitly constrains it.
1242    pub architecture: Option<Architecture>,
1243    /// If true, only instance types that expose nested virtualization (VT-x/EPT)
1244    /// to guest VMs are eligible. Required by workloads that run QEMU/KVM
1245    /// inside a container.
1246    pub nested_virt: bool,
1247}
1248
1249/// Result of instance type selection.
1250#[derive(Debug, Clone)]
1251pub struct InstanceSelection {
1252    /// Selected instance type name (e.g., "m7g.2xlarge")
1253    pub instance_type: &'static str,
1254    /// Machine profile derived from the instance type
1255    pub profile: MachineProfile,
1256    /// Recommended minimum number of machines
1257    pub min_machines: u32,
1258    /// Recommended maximum number of machines
1259    pub max_machines: u32,
1260}
1261
1262/// Ephemeral storage threshold above which storage-optimized instances are selected.
1263const STORAGE_OPTIMIZED_THRESHOLD: u64 = 200 * GI;
1264
1265/// Maximum number of machines per cluster.
1266const MAX_MACHINES_PER_CLUSTER: u32 = 10;
1267
1268/// Hard cap on vCPUs for non-GPU/non-storage workloads. Equivalent to AWS 2xlarge.
1269/// Beyond this, horizontal scaling is always preferred over bigger machines.
1270const MAX_STANDARD_VCPU: u32 = 8;
1271
1272/// Runtime CPU reserved for system processes on each managed container machine.
1273const SYSTEM_RESERVE_CPU: f64 = 0.5;
1274
1275/// Runtime planning headroom for total desired/max workload.
1276const WORKLOAD_HEADROOM_FACTOR: f64 = 1.15;
1277
1278/// Select the best instance type for a workload on a given platform.
1279///
1280/// The algorithm:
1281/// 1. GPU workloads: Match by GPU type, find smallest instance with enough GPUs.
1282/// 2. Storage-heavy workloads (>200Gi ephemeral): Use storage-optimized instances.
1283/// 3. All other workloads: Size the machine to fit a small HA-friendly baseline,
1284///    capped at 8 vCPUs. Use GeneralPurpose family for broad availability and
1285///    reasonable cost. Scale horizontally for more capacity.
1286///
1287/// Returns an error if no suitable instance type is found.
1288pub fn select_instance_type(
1289    platform: Platform,
1290    requirements: &WorkloadRequirements,
1291) -> Result<InstanceSelection, String> {
1292    let architecture = requirements
1293        .architecture
1294        .or_else(|| default_architecture(platform))
1295        .ok_or_else(|| format!("platform {platform} has no default compute architecture"))?;
1296
1297    // Determine which family to use. Nested virt isn't available on
1298    // burstable hardware on any cloud, so a workload that classifies as
1299    // Burstable but needs nested virt must be upgraded to GeneralPurpose
1300    // (the family that actually has nested-virt-capable entries).
1301    let raw_family = select_family(requirements);
1302    let family = if requirements.nested_virt && raw_family == InstanceFamily::Burstable {
1303        InstanceFamily::GeneralPurpose
1304    } else {
1305        raw_family
1306    };
1307
1308    let candidates: Vec<&InstanceTypeSpec> = CATALOG
1309        .iter()
1310        .filter(|spec| spec.platform == platform && spec.family == family)
1311        .filter(|spec| {
1312            if requirements.nested_virt {
1313                spec.is_nested_virt_capable()
1314            } else {
1315                platform != Platform::Aws || !spec.is_nested_virt_capable()
1316            }
1317        })
1318        .filter(|spec| spec.architecture == architecture)
1319        .collect();
1320
1321    // A storage-heavy workload may have no fixed-local candidate for the
1322    // requested architecture/nested-virtualization contract. That is exactly
1323    // when we must consider a provider-backed disk on a general-purpose VM;
1324    // rejecting here made the fallback below unreachable.
1325    if candidates.is_empty() && family != InstanceFamily::StorageOptimized {
1326        let family_has_other_architecture = CATALOG.iter().any(|spec| {
1327            spec.platform == platform
1328                && spec.family == family
1329                && if requirements.nested_virt {
1330                    spec.is_nested_virt_capable()
1331                } else {
1332                    platform != Platform::Aws || !spec.is_nested_virt_capable()
1333                }
1334        });
1335        if family_has_other_architecture {
1336            return Err(format!(
1337                "architecture {architecture:?} is unavailable for this workload on platform {platform}"
1338            ));
1339        }
1340        return Err(if requirements.nested_virt {
1341            format!(
1342                "no nested-virt-capable {family:?} instance types in catalog for platform {platform}"
1343            )
1344        } else {
1345            format!("no {family:?} instance types in catalog for platform {platform}")
1346        });
1347    }
1348
1349    // For GPU workloads, filter by GPU type
1350    let candidates = if let Some(ref gpu) = requirements.gpu {
1351        let filtered: Vec<&InstanceTypeSpec> = candidates
1352            .into_iter()
1353            .filter(|spec| {
1354                spec.gpu.as_ref().map_or(false, |g| {
1355                    g.gpu_type == gpu.gpu_type && g.count >= gpu.count
1356                })
1357            })
1358            .collect();
1359        if filtered.is_empty() {
1360            return Err(format!(
1361                "no instance type for GPU type '{}' x{} on platform {platform}",
1362                gpu.gpu_type, gpu.count
1363            ));
1364        }
1365        filtered
1366    } else {
1367        candidates
1368    };
1369
1370    // For storage workloads, filter by ephemeral storage capacity
1371    let candidates = if family == InstanceFamily::StorageOptimized {
1372        let filtered: Vec<&InstanceTypeSpec> = candidates
1373            .into_iter()
1374            .filter(|spec| spec.ephemeral_storage_bytes >= requirements.max_ephemeral_storage_bytes)
1375            .collect();
1376        if filtered.is_empty() {
1377            // Fixed-local NVMe is preferred for large requests, but provider
1378            // disks on ordinary cloud machines remain configurable. Fall back
1379            // when no fixed-local catalog entry can satisfy the request.
1380            CATALOG
1381                .iter()
1382                .filter(|spec| {
1383                    spec.platform == platform
1384                        && spec.family == InstanceFamily::GeneralPurpose
1385                        && spec.has_configurable_ephemeral_storage()
1386                        && max_configurable_ephemeral_storage_bytes(platform)
1387                            .is_some_and(|max| requirements.max_ephemeral_storage_bytes <= max)
1388                })
1389                .filter(|spec| {
1390                    if requirements.nested_virt {
1391                        spec.is_nested_virt_capable()
1392                    } else {
1393                        platform != Platform::Aws || !spec.is_nested_virt_capable()
1394                    }
1395                })
1396                .filter(|spec| spec.architecture == architecture)
1397                .collect()
1398        } else {
1399            filtered
1400        }
1401    } else {
1402        candidates
1403    };
1404
1405    if candidates.is_empty() {
1406        return Err(format!(
1407            "architecture {architecture:?} is unavailable for this workload on platform {platform}"
1408        ));
1409    }
1410
1411    // Apply the policy for the candidates we will actually select from. A
1412    // storage-heavy request can fall back from fixed-local storage machines to
1413    // general-purpose machines with provider-backed disks; those machines must
1414    // retain the normal horizontal-scaling cap.
1415    let effective_family = candidates[0].family;
1416    let vcpu_cap = if effective_family == InstanceFamily::GpuCompute
1417        || effective_family == InstanceFamily::StorageOptimized
1418    {
1419        u32::MAX
1420    } else {
1421        MAX_STANDARD_VCPU
1422    };
1423
1424    let desired_target_machines = desired_target_machines(requirements);
1425    let target_cpu = requirements
1426        .max_cpu_per_container
1427        .max(requirements.total_cpu_at_desired / desired_target_machines as f64)
1428        * WORKLOAD_HEADROOM_FACTOR;
1429    let target_memory = (requirements.max_memory_per_container as f64)
1430        .max(requirements.total_memory_bytes_at_desired as f64 / desired_target_machines as f64)
1431        * WORKLOAD_HEADROOM_FACTOR;
1432
1433    // Find the smallest instance whose allocatable capacity meets the workload
1434    // target after host reserve and workload headroom. Machine count already
1435    // accounts for multiple replicas; requiring space for an arbitrary second
1436    // copy here would size the same demand twice.
1437    let selected = candidates
1438        .iter()
1439        .filter(|spec| {
1440            spec.vcpu <= vcpu_cap
1441                && allocatable_cpu(spec) >= target_cpu
1442                && allocatable_memory_bytes(spec) as f64 >= target_memory
1443        })
1444        .min_by_key(|spec| spec.vcpu)
1445        .or_else(|| {
1446            // If nothing fits within the cap, pick the largest instance under the cap
1447            candidates
1448                .iter()
1449                .filter(|spec| spec.vcpu <= vcpu_cap)
1450                .max_by_key(|spec| spec.vcpu)
1451        })
1452        .or_else(|| {
1453            // Last resort: pick the smallest available instance (for GPU/storage)
1454            candidates.iter().min_by_key(|spec| spec.vcpu)
1455        })
1456        .ok_or_else(|| format!("no instance types available for platform {platform}"))?;
1457
1458    // Calculate machine counts
1459    let max_machines = compute_max_machines(requirements, selected);
1460    let min_machines = compute_min_machines(requirements, selected, max_machines);
1461
1462    Ok(InstanceSelection {
1463        instance_type: selected.name,
1464        profile: selected.to_machine_profile_for_storage(requirements.max_ephemeral_storage_bytes),
1465        min_machines,
1466        max_machines,
1467    })
1468}
1469
1470/// Select instance family based on workload characteristics.
1471///
1472/// Uses GeneralPurpose for all standard workloads — widely available across
1473/// regions and cost-effective. Only specialized workloads (GPU, large ephemeral
1474/// storage) get specialized families. Very small workloads get burstable.
1475pub fn select_family(requirements: &WorkloadRequirements) -> InstanceFamily {
1476    // GPU workloads always get GPU instances
1477    if requirements.gpu.is_some() {
1478        return InstanceFamily::GpuCompute;
1479    }
1480
1481    // Large ephemeral storage needs NVMe (storage-optimized)
1482    if requirements.max_ephemeral_storage_bytes > STORAGE_OPTIMIZED_THRESHOLD {
1483        return InstanceFamily::StorageOptimized;
1484    }
1485
1486    // Very small workloads use burstable instances
1487    if requirements.total_cpu_at_max < 2.0 {
1488        return InstanceFamily::Burstable;
1489    }
1490
1491    // All other workloads use GeneralPurpose — available everywhere, good pricing
1492    InstanceFamily::GeneralPurpose
1493}
1494
1495/// Calculate maximum machines needed to fit the workload with headroom.
1496fn compute_max_machines(requirements: &WorkloadRequirements, instance: &InstanceTypeSpec) -> u32 {
1497    let cpu_with_headroom = requirements.total_cpu_at_max * WORKLOAD_HEADROOM_FACTOR;
1498    let cpu_machines = (cpu_with_headroom / allocatable_cpu(instance)).ceil() as u32;
1499
1500    let mem_with_headroom =
1501        requirements.total_memory_bytes_at_max as f64 * WORKLOAD_HEADROOM_FACTOR;
1502    let mem_machines =
1503        (mem_with_headroom / allocatable_memory_bytes(instance) as f64).ceil() as u32;
1504
1505    // Take the larger of CPU-based and memory-based, clamped to cluster limit
1506    cpu_machines
1507        .max(mem_machines)
1508        .max(1)
1509        .min(MAX_MACHINES_PER_CLUSTER)
1510}
1511
1512/// Calculate minimum machines for HA.
1513fn compute_min_machines(
1514    requirements: &WorkloadRequirements,
1515    instance: &InstanceTypeSpec,
1516    max_machines: u32,
1517) -> u32 {
1518    let cpu_with_headroom = requirements.total_cpu_at_desired * WORKLOAD_HEADROOM_FACTOR;
1519    let cpu_machines = (cpu_with_headroom / allocatable_cpu(instance)).ceil() as u32;
1520
1521    let mem_with_headroom =
1522        requirements.total_memory_bytes_at_desired as f64 * WORKLOAD_HEADROOM_FACTOR;
1523    let mem_machines =
1524        (mem_with_headroom / allocatable_memory_bytes(instance) as f64).ceil() as u32;
1525
1526    cpu_machines
1527        .max(mem_machines)
1528        .max(1)
1529        .min(2)
1530        .min(max_machines)
1531}
1532
1533fn desired_target_machines(requirements: &WorkloadRequirements) -> u32 {
1534    if requirements.total_cpu_at_desired >= 2.0
1535        || requirements.total_memory_bytes_at_desired >= 4 * GI
1536    {
1537        2
1538    } else {
1539        1
1540    }
1541}
1542
1543fn allocatable_cpu(instance: &InstanceTypeSpec) -> f64 {
1544    (instance.vcpu as f64 - SYSTEM_RESERVE_CPU).max(0.25)
1545}
1546
1547fn allocatable_memory_bytes(instance: &InstanceTypeSpec) -> u64 {
1548    instance
1549        .memory_bytes
1550        .saturating_sub(system_reserve_memory_bytes(instance.memory_bytes))
1551        .max(256 * MI)
1552}
1553
1554fn system_reserve_memory_bytes(memory_bytes: u64) -> u64 {
1555    if memory_bytes < 4 * GI {
1556        256 * MI
1557    } else if memory_bytes < 16 * GI {
1558        512 * MI
1559    } else {
1560        GI
1561    }
1562}
1563
1564// ---------------------------------------------------------------------------
1565// Tests
1566// ---------------------------------------------------------------------------
1567
1568#[cfg(test)]
1569mod tests {
1570    use super::*;
1571    use crate::BinaryTarget;
1572
1573    // -- Parsing tests --
1574
1575    #[test]
1576    fn test_parse_cpu_plain() {
1577        assert_eq!(parse_cpu("1").unwrap(), 1.0);
1578        assert_eq!(parse_cpu("0.5").unwrap(), 0.5);
1579        assert_eq!(parse_cpu("2.0").unwrap(), 2.0);
1580        assert_eq!(parse_cpu("16").unwrap(), 16.0);
1581    }
1582
1583    #[test]
1584    fn test_parse_cpu_millicore() {
1585        assert_eq!(parse_cpu("500m").unwrap(), 0.5);
1586        assert_eq!(parse_cpu("250m").unwrap(), 0.25);
1587        assert_eq!(parse_cpu("1000m").unwrap(), 1.0);
1588        assert_eq!(parse_cpu("100m").unwrap(), 0.1);
1589    }
1590
1591    #[test]
1592    fn test_parse_cpu_invalid() {
1593        assert!(parse_cpu("").is_err());
1594        assert!(parse_cpu("abc").is_err());
1595        assert!(parse_cpu("m").is_err());
1596    }
1597
1598    #[test]
1599    fn test_parse_memory_binary_suffixes() {
1600        assert_eq!(parse_memory_bytes("1Ki").unwrap(), 1024);
1601        assert_eq!(parse_memory_bytes("1Mi").unwrap(), 1024 * 1024);
1602        assert_eq!(parse_memory_bytes("1Gi").unwrap(), 1024 * 1024 * 1024);
1603        assert_eq!(parse_memory_bytes("4Gi").unwrap(), 4 * 1024 * 1024 * 1024);
1604        assert_eq!(parse_memory_bytes("512Mi").unwrap(), 512 * 1024 * 1024);
1605        assert_eq!(
1606            parse_memory_bytes("1Ti").unwrap(),
1607            1024u64 * 1024 * 1024 * 1024
1608        );
1609    }
1610
1611    #[test]
1612    fn test_parse_memory_decimal_suffixes() {
1613        assert_eq!(parse_memory_bytes("1k").unwrap(), 1000);
1614        assert_eq!(parse_memory_bytes("1M").unwrap(), 1_000_000);
1615        assert_eq!(parse_memory_bytes("1G").unwrap(), 1_000_000_000);
1616        assert_eq!(parse_memory_bytes("1T").unwrap(), 1_000_000_000_000);
1617    }
1618
1619    #[test]
1620    fn test_parse_memory_plain_bytes() {
1621        assert_eq!(parse_memory_bytes("1024").unwrap(), 1024);
1622        assert_eq!(parse_memory_bytes("0").unwrap(), 0);
1623    }
1624
1625    #[test]
1626    fn test_parse_memory_invalid() {
1627        assert!(parse_memory_bytes("").is_err());
1628        assert!(parse_memory_bytes("abc").is_err());
1629        assert!(parse_memory_bytes("Gi").is_err());
1630    }
1631
1632    #[test]
1633    fn test_parse_memory_fractional() {
1634        assert_eq!(parse_memory_bytes("0.5Gi").unwrap(), GI / 2);
1635        assert_eq!(parse_memory_bytes("1.5Gi").unwrap(), GI + GI / 2);
1636    }
1637
1638    // -- Catalog lookup tests --
1639
1640    #[test]
1641    fn test_catalog_has_entries_for_all_cloud_platforms() {
1642        assert!(!catalog_for_platform(Platform::Aws).is_empty());
1643        assert!(!catalog_for_platform(Platform::Gcp).is_empty());
1644        assert!(!catalog_for_platform(Platform::Azure).is_empty());
1645    }
1646
1647    #[test]
1648    fn test_catalog_no_entries_for_non_cloud_platforms() {
1649        assert!(catalog_for_platform(Platform::Local).is_empty());
1650        assert!(catalog_for_platform(Platform::Kubernetes).is_empty());
1651    }
1652
1653    #[test]
1654    fn test_find_known_instance_type() {
1655        let spec =
1656            find_instance_type(Platform::Aws, "m7g.2xlarge").expect("should find m7g.2xlarge");
1657        assert_eq!(spec.vcpu, 8);
1658        assert_eq!(spec.memory_bytes, 32 * GI);
1659        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1660    }
1661
1662    #[test]
1663    fn test_find_aws_c8i_nested_virt_instance_type() {
1664        let spec = find_instance_type(Platform::Aws, "c8i.large").expect("should find c8i.large");
1665        assert_eq!(spec.vcpu, 2);
1666        assert_eq!(spec.memory_bytes, 4 * GI);
1667        assert_eq!(spec.family, InstanceFamily::ComputeOptimized);
1668        assert_eq!(spec.architecture, Architecture::X86_64);
1669        assert!(spec.is_nested_virt_capable());
1670    }
1671
1672    #[test]
1673    fn test_find_unknown_instance_type() {
1674        assert!(find_instance_type(Platform::Aws, "nonexistent.xlarge").is_none());
1675    }
1676
1677    #[test]
1678    fn test_find_wrong_platform() {
1679        assert!(find_instance_type(Platform::Gcp, "m7g.2xlarge").is_none());
1680    }
1681
1682    #[test]
1683    fn test_to_machine_profile() {
1684        let spec = find_instance_type(Platform::Aws, "m7g.2xlarge").unwrap();
1685        let profile = spec.to_machine_profile();
1686        assert_eq!(profile.cpu, "8.0");
1687        assert_eq!(profile.memory_bytes, 32 * GI);
1688        assert_eq!(profile.ephemeral_storage_bytes, 20 * GI);
1689        assert!(profile.gpu.is_none());
1690    }
1691
1692    #[test]
1693    fn test_to_machine_profile_with_gpu() {
1694        let spec = find_instance_type(Platform::Aws, "p4d.24xlarge").unwrap();
1695        let profile = spec.to_machine_profile();
1696        let gpu = profile.gpu.as_ref().expect("should have GPU");
1697        assert_eq!(gpu.gpu_type, "nvidia-a100");
1698        assert_eq!(gpu.count, 8);
1699    }
1700
1701    // -- Selection algorithm tests --
1702
1703    #[test]
1704    fn test_select_burstable_for_small_workload() {
1705        let req = WorkloadRequirements {
1706            total_cpu_at_desired: 1.0,
1707            total_memory_bytes_at_desired: 2 * GI,
1708            total_cpu_at_max: 1.0,
1709            total_memory_bytes_at_max: 2 * GI,
1710            max_cpu_per_container: 0.5,
1711            max_memory_per_container: 1 * GI,
1712            max_ephemeral_storage_bytes: 10 * GI,
1713            gpu: None,
1714            architecture: None,
1715            nested_virt: false,
1716        };
1717        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1718        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1719        assert_eq!(spec.family, InstanceFamily::Burstable);
1720    }
1721
1722    #[test]
1723    fn test_selects_smallest_burstable_machine_with_real_headroom() {
1724        let req = WorkloadRequirements {
1725            total_cpu_at_desired: 1.0,
1726            total_memory_bytes_at_desired: 2 * GI,
1727            total_cpu_at_max: 1.0,
1728            total_memory_bytes_at_max: 2 * GI,
1729            max_cpu_per_container: 1.0,
1730            max_memory_per_container: 2 * GI,
1731            max_ephemeral_storage_bytes: 10 * GI,
1732            gpu: None,
1733            architecture: None,
1734            nested_virt: false,
1735        };
1736
1737        let selection = select_instance_type(Platform::Aws, &req).unwrap();
1738
1739        assert_eq!(selection.instance_type, "t4g.medium");
1740        assert_eq!(selection.min_machines, 1);
1741        assert_eq!(selection.max_machines, 1);
1742    }
1743
1744    #[test]
1745    fn test_select_general_purpose_for_standard_workload() {
1746        // Standard workloads always get GeneralPurpose regardless of CPU:memory ratio
1747        let req = WorkloadRequirements {
1748            total_cpu_at_desired: 20.0,
1749            total_memory_bytes_at_desired: 80 * GI,
1750            total_cpu_at_max: 20.0,
1751            total_memory_bytes_at_max: 80 * GI,
1752            max_cpu_per_container: 2.0,
1753            max_memory_per_container: 8 * GI,
1754            max_ephemeral_storage_bytes: 10 * GI,
1755            gpu: None,
1756            architecture: None,
1757            nested_virt: false,
1758        };
1759        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1760        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1761        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1762    }
1763
1764    #[test]
1765    fn test_select_general_purpose_even_for_cpu_heavy() {
1766        // CPU-heavy workloads still get GeneralPurpose (no more ComputeOptimized auto-select)
1767        let req = WorkloadRequirements {
1768            total_cpu_at_desired: 20.0,
1769            total_memory_bytes_at_desired: 20 * GI,
1770            total_cpu_at_max: 20.0,
1771            total_memory_bytes_at_max: 20 * GI,
1772            max_cpu_per_container: 2.0,
1773            max_memory_per_container: 2 * GI,
1774            max_ephemeral_storage_bytes: 10 * GI,
1775            gpu: None,
1776            architecture: None,
1777            nested_virt: false,
1778        };
1779        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1780        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1781        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1782    }
1783
1784    #[test]
1785    fn test_select_storage_optimized_for_large_ephemeral() {
1786        let req = WorkloadRequirements {
1787            total_cpu_at_desired: 8.0,
1788            total_memory_bytes_at_desired: 32 * GI,
1789            total_cpu_at_max: 8.0,
1790            total_memory_bytes_at_max: 32 * GI,
1791            max_cpu_per_container: 2.0,
1792            max_memory_per_container: 8 * GI,
1793            max_ephemeral_storage_bytes: 500 * GI,
1794            gpu: None,
1795            architecture: Some(Architecture::X86_64),
1796            nested_virt: false,
1797        };
1798        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1799        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1800        assert_eq!(spec.family, InstanceFamily::StorageOptimized);
1801    }
1802
1803    #[test]
1804    fn test_select_configurable_storage_above_fixed_local_catalog() {
1805        let req = WorkloadRequirements {
1806            total_cpu_at_desired: 8.0,
1807            total_memory_bytes_at_desired: 32 * GI,
1808            total_cpu_at_max: 8.0,
1809            total_memory_bytes_at_max: 32 * GI,
1810            max_cpu_per_container: 2.0,
1811            max_memory_per_container: 8 * GI,
1812            max_ephemeral_storage_bytes: 8_000 * GI,
1813            gpu: None,
1814            architecture: Some(Architecture::X86_64),
1815            nested_virt: false,
1816        };
1817        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1818        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1819        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1820        assert_eq!(sel.profile.ephemeral_storage_bytes, 8_000 * GI);
1821    }
1822
1823    #[test]
1824    fn test_configurable_storage_fallback_retains_standard_vcpu_cap() {
1825        let req = WorkloadRequirements {
1826            total_cpu_at_desired: 70.0,
1827            total_memory_bytes_at_desired: 140 * GI,
1828            total_cpu_at_max: 70.0,
1829            total_memory_bytes_at_max: 140 * GI,
1830            max_cpu_per_container: 2.0,
1831            max_memory_per_container: 4 * GI,
1832            max_ephemeral_storage_bytes: 8_000 * GI,
1833            gpu: None,
1834            architecture: Some(Architecture::X86_64),
1835            nested_virt: false,
1836        };
1837
1838        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1839        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1840        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1841        assert!(spec.vcpu <= MAX_STANDARD_VCPU);
1842        assert!(sel.max_machines > 1);
1843    }
1844
1845    #[test]
1846    fn test_nested_virtualization_storage_falls_back_to_configurable_disk() {
1847        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
1848            let req = WorkloadRequirements {
1849                total_cpu_at_desired: 2.0,
1850                total_memory_bytes_at_desired: 8 * GI,
1851                total_cpu_at_max: 2.0,
1852                total_memory_bytes_at_max: 8 * GI,
1853                max_cpu_per_container: 2.0,
1854                max_memory_per_container: 8 * GI,
1855                max_ephemeral_storage_bytes: 500 * GI,
1856                gpu: None,
1857                architecture: Some(Architecture::X86_64),
1858                nested_virt: true,
1859            };
1860
1861            let sel = select_instance_type(platform, &req)
1862                .unwrap_or_else(|error| panic!("{platform} should support fallback: {error}"));
1863            let spec = find_instance_type(platform, sel.instance_type).unwrap();
1864            assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1865            assert!(spec.is_nested_virt_capable());
1866            assert_eq!(sel.profile.ephemeral_storage_bytes, 500 * GI);
1867        }
1868    }
1869
1870    #[test]
1871    fn test_configurable_disk_rejects_capacity_above_provider_limit() {
1872        for (platform, requested) in [
1873            (Platform::Aws, 100 * TI),
1874            (Platform::Gcp, 100 * TI),
1875            (Platform::Azure, 8_000 * GI),
1876        ] {
1877            let req = WorkloadRequirements {
1878                total_cpu_at_desired: 2.0,
1879                total_memory_bytes_at_desired: 8 * GI,
1880                total_cpu_at_max: 2.0,
1881                total_memory_bytes_at_max: 8 * GI,
1882                max_cpu_per_container: 2.0,
1883                max_memory_per_container: 8 * GI,
1884                max_ephemeral_storage_bytes: requested,
1885                gpu: None,
1886                architecture: Some(Architecture::X86_64),
1887                nested_virt: true,
1888            };
1889
1890            assert!(select_instance_type(platform, &req).is_err());
1891        }
1892    }
1893
1894    #[test]
1895    fn test_configurable_disk_limits_account_for_controller_rounding() {
1896        for (platform, disk_limit_gib) in [
1897            (Platform::Aws, 64 * 1024),
1898            (Platform::Gcp, 64 * 1024),
1899            (Platform::Azure, 4_095),
1900        ] {
1901            let max = max_configurable_ephemeral_storage_bytes(platform).unwrap();
1902            let materialized_disk_gib = |requested_bytes: u64| {
1903                let requested_gib = requested_bytes.div_ceil(GI);
1904                (requested_gib * 5).div_ceil(4) + 12
1905            };
1906
1907            assert!(materialized_disk_gib(max) <= disk_limit_gib);
1908            assert!(materialized_disk_gib(max + 1) > disk_limit_gib);
1909        }
1910    }
1911
1912    #[test]
1913    fn test_gpu_local_storage_is_not_treated_as_resizable() {
1914        let spec = find_instance_type(Platform::Aws, "g5.xlarge").unwrap();
1915        assert_eq!(spec.family, InstanceFamily::GpuCompute);
1916        assert!(!spec.has_configurable_ephemeral_storage());
1917    }
1918
1919    #[test]
1920    fn test_select_gpu_instance() {
1921        let req = WorkloadRequirements {
1922            total_cpu_at_desired: 8.0,
1923            total_memory_bytes_at_desired: 32 * GI,
1924            total_cpu_at_max: 8.0,
1925            total_memory_bytes_at_max: 32 * GI,
1926            max_cpu_per_container: 4.0,
1927            max_memory_per_container: 16 * GI,
1928            max_ephemeral_storage_bytes: 10 * GI,
1929            gpu: Some(GpuSpec {
1930                gpu_type: "nvidia-a100".to_string(),
1931                count: 1,
1932            }),
1933            architecture: Some(Architecture::X86_64),
1934            nested_virt: false,
1935        };
1936        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1937        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1938        assert_eq!(spec.family, InstanceFamily::GpuCompute);
1939        assert!(spec.gpu.is_some());
1940    }
1941
1942    #[test]
1943    fn test_select_uses_each_cloud_image_target_architecture() {
1944        let req = WorkloadRequirements {
1945            total_cpu_at_desired: 4.0,
1946            total_memory_bytes_at_desired: 16 * GI,
1947            total_cpu_at_max: 4.0,
1948            total_memory_bytes_at_max: 16 * GI,
1949            max_cpu_per_container: 1.0,
1950            max_memory_per_container: 4 * GI,
1951            max_ephemeral_storage_bytes: 10 * GI,
1952            gpu: None,
1953            architecture: None,
1954            nested_virt: false,
1955        };
1956        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
1957            let sel = select_instance_type(platform, &req)
1958                .unwrap_or_else(|error| panic!("selection failed for {platform}: {error}"));
1959            let spec = find_instance_type(platform, sel.instance_type)
1960                .expect("selected machine should exist in the catalog");
1961            assert_eq!(
1962                Some(spec.architecture),
1963                default_architecture(platform),
1964                "machine architecture must match the image target for {platform}"
1965            );
1966        }
1967    }
1968
1969    #[test]
1970    fn test_machine_count_reasonable() {
1971        // Single container: 1 CPU, 2Gi, maxReplicas=20
1972        let req = WorkloadRequirements {
1973            total_cpu_at_desired: 20.0,
1974            total_memory_bytes_at_desired: 40 * GI,
1975            total_cpu_at_max: 20.0,
1976            total_memory_bytes_at_max: 40 * GI,
1977            max_cpu_per_container: 1.0,
1978            max_memory_per_container: 2 * GI,
1979            max_ephemeral_storage_bytes: 10 * GI,
1980            gpu: None,
1981            architecture: None,
1982            nested_virt: false,
1983        };
1984        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1985        assert!(sel.min_machines >= 1);
1986        assert!(sel.max_machines <= MAX_MACHINES_PER_CLUSTER);
1987        assert!(sel.max_machines >= sel.min_machines);
1988    }
1989
1990    #[test]
1991    fn test_instance_size_capped_at_8_vcpu() {
1992        // Even with very large containers, instance size is capped at 8 vCPUs
1993        let req = WorkloadRequirements {
1994            total_cpu_at_desired: 70.0,
1995            total_memory_bytes_at_desired: 140 * GI,
1996            total_cpu_at_max: 70.0,
1997            total_memory_bytes_at_max: 140 * GI,
1998            max_cpu_per_container: 2.0,
1999            max_memory_per_container: 4 * GI,
2000            max_ephemeral_storage_bytes: 10 * GI,
2001            gpu: None,
2002            architecture: None,
2003            nested_virt: false,
2004        };
2005        let sel = select_instance_type(Platform::Gcp, &req).unwrap();
2006        let spec = find_instance_type(Platform::Gcp, sel.instance_type).unwrap();
2007        assert!(
2008            spec.vcpu <= MAX_STANDARD_VCPU,
2009            "selected {} with {} vCPUs, expected <= {}",
2010            spec.name,
2011            spec.vcpu,
2012            MAX_STANDARD_VCPU
2013        );
2014        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
2015        // Should scale horizontally instead
2016        assert!(sel.max_machines > 1);
2017    }
2018
2019    #[test]
2020    fn test_larger_autoscaled_workload_gets_reasonable_instance() {
2021        // Simulates a larger autoscaled workload: 4 containers, each 2 CPU / 4 GiB
2022        // maxReplicas: 10, 10, 10, 5
2023        let req = WorkloadRequirements {
2024            total_cpu_at_desired: 70.0,
2025            total_memory_bytes_at_desired: 140 * GI,
2026            total_cpu_at_max: 70.0,              // 2*10 + 2*10 + 2*10 + 2*5
2027            total_memory_bytes_at_max: 140 * GI, // 4*10 + 4*10 + 4*10 + 4*5
2028            max_cpu_per_container: 2.0,
2029            max_memory_per_container: 4 * GI,
2030            max_ephemeral_storage_bytes: 20 * GI,
2031            gpu: None,
2032            architecture: None,
2033            nested_virt: false,
2034        };
2035        let sel = select_instance_type(Platform::Gcp, &req).unwrap();
2036        // Should pick n2-standard-8 (8 vCPU, 32 GiB) — NOT c3-standard-44
2037        assert_eq!(sel.instance_type, "n2-standard-8");
2038        assert!(sel.max_machines >= 2);
2039    }
2040
2041    /// When `nested_virt` is set on the workload, the selector must
2042    /// restrict to nested-virt-capable families. On AWS that means an m8i
2043    /// (or other 8th-gen Intel) entry, never a Graviton (`*7g`, `t4g`) or
2044    /// burstable. Without this filter the launch template gets created
2045    /// with `CpuOptions.NestedVirtualization=enabled` paired with an
2046    /// instance type AWS rejects at RunInstances.
2047    #[test]
2048    fn test_select_aws_picks_m8i_when_nested_virt_required() {
2049        let req = WorkloadRequirements {
2050            total_cpu_at_desired: 4.0,
2051            total_memory_bytes_at_desired: 8 * GI,
2052            total_cpu_at_max: 4.0,
2053            total_memory_bytes_at_max: 8 * GI,
2054            max_cpu_per_container: 4.0,
2055            max_memory_per_container: 8 * GI,
2056            max_ephemeral_storage_bytes: 10 * GI,
2057            gpu: None,
2058            architecture: Some(Architecture::X86_64),
2059            nested_virt: true,
2060        };
2061        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2062        assert!(
2063            sel.instance_type.starts_with("m8i.")
2064                || sel.instance_type.starts_with("c8i.")
2065                || sel.instance_type.starts_with("r8i."),
2066            "expected an m8i/c8i/r8i instance, got {}",
2067            sel.instance_type
2068        );
2069        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
2070        assert!(spec.is_nested_virt_capable());
2071    }
2072
2073    #[test]
2074    fn test_select_gcp_picks_n2_when_nested_virt_required() {
2075        let req = WorkloadRequirements {
2076            total_cpu_at_desired: 4.0,
2077            total_memory_bytes_at_desired: 8 * GI,
2078            total_cpu_at_max: 4.0,
2079            total_memory_bytes_at_max: 8 * GI,
2080            max_cpu_per_container: 4.0,
2081            max_memory_per_container: 8 * GI,
2082            max_ephemeral_storage_bytes: 0,
2083            architecture: Some(Architecture::X86_64),
2084            gpu: None,
2085            nested_virt: true,
2086        };
2087
2088        let selection = select_instance_type(Platform::Gcp, &req).unwrap();
2089        assert_eq!(selection.instance_type, "n2-standard-8");
2090        assert!(find_instance_type(Platform::Gcp, selection.instance_type)
2091            .unwrap()
2092            .is_nested_virt_capable());
2093    }
2094
2095    #[test]
2096    fn test_select_azure_picks_dsv5_when_nested_virt_required() {
2097        let req = WorkloadRequirements {
2098            total_cpu_at_desired: 4.0,
2099            total_memory_bytes_at_desired: 8 * GI,
2100            total_cpu_at_max: 4.0,
2101            total_memory_bytes_at_max: 8 * GI,
2102            max_cpu_per_container: 4.0,
2103            max_memory_per_container: 8 * GI,
2104            max_ephemeral_storage_bytes: 0,
2105            architecture: Some(Architecture::X86_64),
2106            gpu: None,
2107            nested_virt: true,
2108        };
2109
2110        let selection = select_instance_type(Platform::Azure, &req).unwrap();
2111        assert_eq!(selection.instance_type, "Standard_D8s_v5");
2112        assert!(find_instance_type(Platform::Azure, selection.instance_type)
2113            .unwrap()
2114            .is_nested_virt_capable());
2115    }
2116
2117    #[test]
2118    fn test_select_aws_defaults_to_image_target_architecture() {
2119        let req = WorkloadRequirements {
2120            total_cpu_at_desired: 4.0,
2121            total_memory_bytes_at_desired: 8 * GI,
2122            total_cpu_at_max: 4.0,
2123            total_memory_bytes_at_max: 8 * GI,
2124            max_cpu_per_container: 4.0,
2125            max_memory_per_container: 8 * GI,
2126            max_ephemeral_storage_bytes: 10 * GI,
2127            gpu: None,
2128            architecture: None,
2129            nested_virt: false,
2130        };
2131        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2132        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
2133        assert_eq!(spec.architecture, Architecture::Arm64);
2134    }
2135
2136    #[test]
2137    fn test_cloud_defaults_match_image_target_architectures() {
2138        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
2139            let target = BinaryTarget::defaults_for_platform(platform)
2140                .into_iter()
2141                .next()
2142                .expect("managed cloud should have a default image target");
2143            let image_architecture = match target.oci_arch() {
2144                "arm64" => Architecture::Arm64,
2145                "amd64" => Architecture::X86_64,
2146                architecture => {
2147                    panic!("unsupported managed-cloud image architecture {architecture}")
2148                }
2149            };
2150
2151            assert_eq!(default_architecture(platform), Some(image_architecture));
2152        }
2153    }
2154
2155    /// ARM remains available when the workload or capacity profile declares it.
2156    #[test]
2157    fn test_select_aws_uses_graviton_for_explicit_arm64() {
2158        let req = WorkloadRequirements {
2159            total_cpu_at_desired: 4.0,
2160            total_memory_bytes_at_desired: 8 * GI,
2161            total_cpu_at_max: 4.0,
2162            total_memory_bytes_at_max: 8 * GI,
2163            max_cpu_per_container: 4.0,
2164            max_memory_per_container: 8 * GI,
2165            max_ephemeral_storage_bytes: 10 * GI,
2166            gpu: None,
2167            architecture: Some(Architecture::Arm64),
2168            nested_virt: false,
2169        };
2170        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2171        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
2172        assert_eq!(spec.architecture, Architecture::Arm64);
2173    }
2174
2175    #[test]
2176    fn test_select_rejects_explicit_architecture_missing_from_cloud_catalog() {
2177        let req = WorkloadRequirements {
2178            total_cpu_at_desired: 1.0,
2179            total_memory_bytes_at_desired: 2 * GI,
2180            total_cpu_at_max: 1.0,
2181            total_memory_bytes_at_max: 2 * GI,
2182            max_cpu_per_container: 1.0,
2183            max_memory_per_container: 2 * GI,
2184            max_ephemeral_storage_bytes: 10 * GI,
2185            gpu: None,
2186            architecture: Some(Architecture::Arm64),
2187            nested_virt: false,
2188        };
2189
2190        let error = select_instance_type(Platform::Gcp, &req)
2191            .expect_err("GCP catalog has no ARM64 machine");
2192
2193        assert!(error.contains("architecture Arm64 is unavailable"));
2194    }
2195
2196    #[test]
2197    fn test_profile_has_required_fields() {
2198        let req = WorkloadRequirements {
2199            total_cpu_at_desired: 4.0,
2200            total_memory_bytes_at_desired: 16 * GI,
2201            total_cpu_at_max: 4.0,
2202            total_memory_bytes_at_max: 16 * GI,
2203            max_cpu_per_container: 1.0,
2204            max_memory_per_container: 4 * GI,
2205            max_ephemeral_storage_bytes: 10 * GI,
2206            gpu: None,
2207            architecture: None,
2208            nested_virt: false,
2209        };
2210        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2211        assert!(!sel.profile.cpu.is_empty());
2212        assert!(sel.profile.memory_bytes > 0);
2213        assert!(sel.profile.ephemeral_storage_bytes > 0);
2214    }
2215
2216    #[test]
2217    fn test_error_for_unsupported_gpu_type() {
2218        let req = WorkloadRequirements {
2219            total_cpu_at_desired: 8.0,
2220            total_memory_bytes_at_desired: 32 * GI,
2221            total_cpu_at_max: 8.0,
2222            total_memory_bytes_at_max: 32 * GI,
2223            max_cpu_per_container: 4.0,
2224            max_memory_per_container: 16 * GI,
2225            max_ephemeral_storage_bytes: 10 * GI,
2226            gpu: Some(GpuSpec {
2227                gpu_type: "amd-mi300".to_string(),
2228                count: 1,
2229            }),
2230            architecture: None,
2231            nested_virt: false,
2232        };
2233        let result = select_instance_type(Platform::Aws, &req);
2234        assert!(result.is_err());
2235    }
2236
2237    #[test]
2238    fn test_catalog_instance_types_sorted_by_vcpu_within_family() {
2239        // Verify that within each (platform, family) group, vcpu is non-decreasing.
2240        // This ensures our "min_by_key(vcpu)" logic works correctly.
2241        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
2242            let entries = catalog_for_platform(platform);
2243            let mut by_family: std::collections::HashMap<_, Vec<_>> =
2244                std::collections::HashMap::new();
2245            for entry in entries {
2246                by_family
2247                    .entry(format!("{:?}", entry.family))
2248                    .or_default()
2249                    .push(entry);
2250            }
2251            for (family, instances) in &by_family {
2252                for window in instances.windows(2) {
2253                    assert!(
2254                        window[0].vcpu <= window[1].vcpu,
2255                        "catalog not sorted by vcpu for {platform}/{family}: {} ({}) > {} ({})",
2256                        window[0].name,
2257                        window[0].vcpu,
2258                        window[1].name,
2259                        window[1].vcpu
2260                    );
2261                }
2262            }
2263        }
2264    }
2265}