Skip to main content

alien_core/
instance_catalog.rs

1//! Instance type catalog and selection algorithm for cloud compute infrastructure.
2//!
3//! This module provides:
4//! - A static catalog of known instance types across AWS, GCP, and Azure
5//! - Resource quantity parsing (CPU strings, Kubernetes-style memory/storage quantities)
6//! - An algorithm to select the optimal instance type for a given workload
7//!
8//! The catalog is the single source of truth for instance type specifications.
9//! It is used by the preflights system to automatically populate `CapacityGroup.instance_type`
10//! and `CapacityGroup.profile` based on the containers in a stack.
11
12use crate::{GpuSpec, MachineProfile, Platform};
13use serde::{Deserialize, Serialize};
14
15// ---------------------------------------------------------------------------
16// Resource quantity parsing
17// ---------------------------------------------------------------------------
18
19/// Parse a CPU quantity string to f64.
20///
21/// Accepts plain numbers ("1", "0.5", "2.0") and millicore suffixes ("500m" = 0.5).
22pub fn parse_cpu(s: &str) -> Result<f64, String> {
23    let s = s.trim();
24    if s.is_empty() {
25        return Err("empty CPU string".to_string());
26    }
27
28    if let Some(millis) = s.strip_suffix('m') {
29        let v: f64 = millis
30            .parse()
31            .map_err(|_| format!("invalid CPU millicore value: '{s}'"))?;
32        Ok(v / 1000.0)
33    } else {
34        s.parse().map_err(|_| format!("invalid CPU value: '{s}'"))
35    }
36}
37
38/// Parse a memory or storage quantity string to bytes.
39///
40/// Supports Kubernetes-style binary suffixes (Ki, Mi, Gi, Ti) and
41/// decimal suffixes (k, M, G, T). Plain numbers are interpreted as bytes.
42pub fn parse_memory_bytes(s: &str) -> Result<u64, String> {
43    let s = s.trim();
44    if s.is_empty() {
45        return Err("empty memory/storage string".to_string());
46    }
47
48    // Binary suffixes (powers of 1024)
49    if let Some(num) = s.strip_suffix("Ti") {
50        let v: f64 = num
51            .parse()
52            .map_err(|_| format!("invalid memory value: '{s}'"))?;
53        return Ok((v * 1024.0 * 1024.0 * 1024.0 * 1024.0) as u64);
54    }
55    if let Some(num) = s.strip_suffix("Gi") {
56        let v: f64 = num
57            .parse()
58            .map_err(|_| format!("invalid memory value: '{s}'"))?;
59        return Ok((v * 1024.0 * 1024.0 * 1024.0) as u64);
60    }
61    if let Some(num) = s.strip_suffix("Mi") {
62        let v: f64 = num
63            .parse()
64            .map_err(|_| format!("invalid memory value: '{s}'"))?;
65        return Ok((v * 1024.0 * 1024.0) as u64);
66    }
67    if let Some(num) = s.strip_suffix("Ki") {
68        let v: f64 = num
69            .parse()
70            .map_err(|_| format!("invalid memory value: '{s}'"))?;
71        return Ok((v * 1024.0) as u64);
72    }
73
74    // Decimal suffixes (powers of 1000)
75    if let Some(num) = s.strip_suffix('T') {
76        let v: f64 = num
77            .parse()
78            .map_err(|_| format!("invalid memory value: '{s}'"))?;
79        return Ok((v * 1_000_000_000_000.0) as u64);
80    }
81    if let Some(num) = s.strip_suffix('G') {
82        let v: f64 = num
83            .parse()
84            .map_err(|_| format!("invalid memory value: '{s}'"))?;
85        return Ok((v * 1_000_000_000.0) as u64);
86    }
87    if let Some(num) = s.strip_suffix('M') {
88        let v: f64 = num
89            .parse()
90            .map_err(|_| format!("invalid memory value: '{s}'"))?;
91        return Ok((v * 1_000_000.0) as u64);
92    }
93    if let Some(num) = s.strip_suffix('k') {
94        let v: f64 = num
95            .parse()
96            .map_err(|_| format!("invalid memory value: '{s}'"))?;
97        return Ok((v * 1000.0) as u64);
98    }
99
100    // Plain bytes
101    s.parse()
102        .map_err(|_| format!("invalid memory value: '{s}'"))
103}
104
105// ---------------------------------------------------------------------------
106// Instance type catalog
107// ---------------------------------------------------------------------------
108
109/// Instance family classification.
110#[derive(Debug, Clone, Copy, PartialEq, Eq)]
111pub enum InstanceFamily {
112    Burstable,
113    GeneralPurpose,
114    ComputeOptimized,
115    MemoryOptimized,
116    StorageOptimized,
117    GpuCompute,
118}
119
120/// CPU architecture.
121#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
122#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
123#[serde(rename_all = "snake_case")]
124pub enum Architecture {
125    Arm64,
126    X86_64,
127}
128
129/// Default machine architecture for images built for a managed cloud.
130pub fn default_architecture(platform: Platform) -> Option<Architecture> {
131    match platform {
132        Platform::Aws => Some(Architecture::Arm64),
133        Platform::Gcp | Platform::Azure => Some(Architecture::X86_64),
134        Platform::Kubernetes | Platform::Machines | Platform::Local | Platform::Test => None,
135    }
136}
137
138/// Static GPU specification for catalog entries (no heap allocation).
139#[derive(Debug, Clone, Copy, PartialEq, Eq)]
140pub struct CatalogGpu {
141    pub gpu_type: &'static str,
142    pub count: u32,
143}
144
145/// A known instance type with its hardware specifications.
146///
147/// All fields are compile-time constants. The catalog is a flat array of these.
148#[derive(Debug, Clone)]
149pub struct InstanceTypeSpec {
150    pub name: &'static str,
151    pub platform: Platform,
152    pub family: InstanceFamily,
153    pub architecture: Architecture,
154    /// vCPU count (hardware total)
155    pub vcpu: u32,
156    /// Memory in bytes (hardware total)
157    pub memory_bytes: u64,
158    /// Ephemeral storage in bytes (hardware total, NVMe for storage-optimized)
159    pub ephemeral_storage_bytes: u64,
160    /// GPU specification (for GPU instances)
161    pub gpu: Option<CatalogGpu>,
162}
163
164impl InstanceTypeSpec {
165    /// Whether this instance type supports nested virtualization.
166    ///
167    /// Classify by documented provider families rather than adding a flag to
168    /// every catalog row. GCP still requires the instance template to opt in;
169    /// Azure exposes the capability automatically on supported VM sizes.
170    pub fn is_nested_virt_capable(&self) -> bool {
171        match self.platform {
172            Platform::Aws => {
173                let name = self.name;
174                name.starts_with("m8i.")
175                    || name.starts_with("c8i.")
176                    || name.starts_with("r8i.")
177                    || name.starts_with("m8i-flex.")
178                    || name.starts_with("c8i-flex.")
179                    || name.starts_with("r8i-flex.")
180            }
181            Platform::Gcp => self.name.starts_with("n2-standard-"),
182            Platform::Azure => {
183                let name = self.name;
184                (name.starts_with("Standard_D") && name.ends_with("s_v5"))
185                    || (name.starts_with("Standard_E") && name.ends_with("s_v5"))
186                    || (name.starts_with("Standard_F") && name.ends_with("s_v2"))
187            }
188            _ => false,
189        }
190    }
191
192    /// Convert this catalog entry into a `MachineProfile` for use in `CapacityGroup`.
193    pub fn to_machine_profile(&self) -> MachineProfile {
194        MachineProfile {
195            cpu: format!("{}.0", self.vcpu),
196            memory_bytes: self.memory_bytes,
197            ephemeral_storage_bytes: self.ephemeral_storage_bytes,
198            architecture: Some(self.architecture),
199            gpu: self.gpu.map(|g| GpuSpec {
200                gpu_type: g.gpu_type.to_string(),
201                count: g.count,
202            }),
203        }
204    }
205}
206
207// Helpers for readable byte constants
208const KI: u64 = 1024;
209const MI: u64 = KI * 1024;
210const GI: u64 = MI * 1024;
211
212/// The complete instance type catalog.
213///
214/// This is the single source of truth for instance type specifications.
215/// Update this array when adding support for new instance types.
216///
217/// NOTE: Ephemeral storage values for non-NVMe instances are conservative defaults
218/// (EBS-backed root volumes). Storage-optimized instances list their NVMe capacity.
219static CATALOG: &[InstanceTypeSpec] = &[
220    // =========================================================================
221    // AWS — ARM (Graviton) preferred for cost efficiency
222    // =========================================================================
223
224    // Burstable (t4g — ARM Graviton2)
225    InstanceTypeSpec {
226        name: "t4g.micro",
227        platform: Platform::Aws,
228        family: InstanceFamily::Burstable,
229        architecture: Architecture::Arm64,
230        vcpu: 2,
231        memory_bytes: 1 * GI,
232        ephemeral_storage_bytes: 20 * GI,
233        gpu: None,
234    },
235    InstanceTypeSpec {
236        name: "t4g.small",
237        platform: Platform::Aws,
238        family: InstanceFamily::Burstable,
239        architecture: Architecture::Arm64,
240        vcpu: 2,
241        memory_bytes: 2 * GI,
242        ephemeral_storage_bytes: 20 * GI,
243        gpu: None,
244    },
245    InstanceTypeSpec {
246        name: "t4g.medium",
247        platform: Platform::Aws,
248        family: InstanceFamily::Burstable,
249        architecture: Architecture::Arm64,
250        vcpu: 2,
251        memory_bytes: 4 * GI,
252        ephemeral_storage_bytes: 20 * GI,
253        gpu: None,
254    },
255    InstanceTypeSpec {
256        name: "t4g.large",
257        platform: Platform::Aws,
258        family: InstanceFamily::Burstable,
259        architecture: Architecture::Arm64,
260        vcpu: 2,
261        memory_bytes: 8 * GI,
262        ephemeral_storage_bytes: 20 * GI,
263        gpu: None,
264    },
265    InstanceTypeSpec {
266        name: "t3.xlarge",
267        platform: Platform::Aws,
268        family: InstanceFamily::Burstable,
269        architecture: Architecture::X86_64,
270        vcpu: 4,
271        memory_bytes: 16 * GI,
272        ephemeral_storage_bytes: 20 * GI,
273        gpu: None,
274    },
275    InstanceTypeSpec {
276        name: "t4g.xlarge",
277        platform: Platform::Aws,
278        family: InstanceFamily::Burstable,
279        architecture: Architecture::Arm64,
280        vcpu: 4,
281        memory_bytes: 16 * GI,
282        ephemeral_storage_bytes: 20 * GI,
283        gpu: None,
284    },
285    // General Purpose (m7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
286    InstanceTypeSpec {
287        name: "m7g.medium",
288        platform: Platform::Aws,
289        family: InstanceFamily::GeneralPurpose,
290        architecture: Architecture::Arm64,
291        vcpu: 1,
292        memory_bytes: 4 * GI,
293        ephemeral_storage_bytes: 20 * GI,
294        gpu: None,
295    },
296    InstanceTypeSpec {
297        name: "m7i.large",
298        platform: Platform::Aws,
299        family: InstanceFamily::GeneralPurpose,
300        architecture: Architecture::X86_64,
301        vcpu: 2,
302        memory_bytes: 8 * GI,
303        ephemeral_storage_bytes: 20 * GI,
304        gpu: None,
305    },
306    InstanceTypeSpec {
307        name: "m7g.large",
308        platform: Platform::Aws,
309        family: InstanceFamily::GeneralPurpose,
310        architecture: Architecture::Arm64,
311        vcpu: 2,
312        memory_bytes: 8 * GI,
313        ephemeral_storage_bytes: 20 * GI,
314        gpu: None,
315    },
316    // 8th-gen Intel AWS families accept
317    // `CpuOptions.NestedVirtualization=enabled`. The catalog filter in
318    // `select_instance_type` includes these entries only when the
319    // workload requests nested virt, so ordinary workloads continue to
320    // pick the cost-efficient Graviton (m7g) above. The pairwise
321    // interleave keeps the per-family vCPU-non-decreasing invariant
322    // (see `test_catalog_instance_types_sorted_by_vcpu_within_family`).
323    InstanceTypeSpec {
324        name: "m8i.large",
325        platform: Platform::Aws,
326        family: InstanceFamily::GeneralPurpose,
327        architecture: Architecture::X86_64,
328        vcpu: 2,
329        memory_bytes: 8 * GI,
330        ephemeral_storage_bytes: 20 * GI,
331        gpu: None,
332    },
333    InstanceTypeSpec {
334        name: "m7i.xlarge",
335        platform: Platform::Aws,
336        family: InstanceFamily::GeneralPurpose,
337        architecture: Architecture::X86_64,
338        vcpu: 4,
339        memory_bytes: 16 * GI,
340        ephemeral_storage_bytes: 20 * GI,
341        gpu: None,
342    },
343    InstanceTypeSpec {
344        name: "m7g.xlarge",
345        platform: Platform::Aws,
346        family: InstanceFamily::GeneralPurpose,
347        architecture: Architecture::Arm64,
348        vcpu: 4,
349        memory_bytes: 16 * GI,
350        ephemeral_storage_bytes: 20 * GI,
351        gpu: None,
352    },
353    InstanceTypeSpec {
354        name: "m8i.xlarge",
355        platform: Platform::Aws,
356        family: InstanceFamily::GeneralPurpose,
357        architecture: Architecture::X86_64,
358        vcpu: 4,
359        memory_bytes: 16 * GI,
360        ephemeral_storage_bytes: 20 * GI,
361        gpu: None,
362    },
363    InstanceTypeSpec {
364        name: "m7i.2xlarge",
365        platform: Platform::Aws,
366        family: InstanceFamily::GeneralPurpose,
367        architecture: Architecture::X86_64,
368        vcpu: 8,
369        memory_bytes: 32 * GI,
370        ephemeral_storage_bytes: 20 * GI,
371        gpu: None,
372    },
373    InstanceTypeSpec {
374        name: "m7g.2xlarge",
375        platform: Platform::Aws,
376        family: InstanceFamily::GeneralPurpose,
377        architecture: Architecture::Arm64,
378        vcpu: 8,
379        memory_bytes: 32 * GI,
380        ephemeral_storage_bytes: 20 * GI,
381        gpu: None,
382    },
383    InstanceTypeSpec {
384        name: "m8i.2xlarge",
385        platform: Platform::Aws,
386        family: InstanceFamily::GeneralPurpose,
387        architecture: Architecture::X86_64,
388        vcpu: 8,
389        memory_bytes: 32 * GI,
390        ephemeral_storage_bytes: 20 * GI,
391        gpu: None,
392    },
393    InstanceTypeSpec {
394        name: "m7i.4xlarge",
395        platform: Platform::Aws,
396        family: InstanceFamily::GeneralPurpose,
397        architecture: Architecture::X86_64,
398        vcpu: 16,
399        memory_bytes: 64 * GI,
400        ephemeral_storage_bytes: 20 * GI,
401        gpu: None,
402    },
403    InstanceTypeSpec {
404        name: "m7g.4xlarge",
405        platform: Platform::Aws,
406        family: InstanceFamily::GeneralPurpose,
407        architecture: Architecture::Arm64,
408        vcpu: 16,
409        memory_bytes: 64 * GI,
410        ephemeral_storage_bytes: 20 * GI,
411        gpu: None,
412    },
413    InstanceTypeSpec {
414        name: "m8i.4xlarge",
415        platform: Platform::Aws,
416        family: InstanceFamily::GeneralPurpose,
417        architecture: Architecture::X86_64,
418        vcpu: 16,
419        memory_bytes: 64 * GI,
420        ephemeral_storage_bytes: 20 * GI,
421        gpu: None,
422    },
423    // Compute Optimized (c7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
424    InstanceTypeSpec {
425        name: "c7g.medium",
426        platform: Platform::Aws,
427        family: InstanceFamily::ComputeOptimized,
428        architecture: Architecture::Arm64,
429        vcpu: 1,
430        memory_bytes: 2 * GI,
431        ephemeral_storage_bytes: 20 * GI,
432        gpu: None,
433    },
434    InstanceTypeSpec {
435        name: "c7g.large",
436        platform: Platform::Aws,
437        family: InstanceFamily::ComputeOptimized,
438        architecture: Architecture::Arm64,
439        vcpu: 2,
440        memory_bytes: 4 * GI,
441        ephemeral_storage_bytes: 20 * GI,
442        gpu: None,
443    },
444    InstanceTypeSpec {
445        name: "c8i.large",
446        platform: Platform::Aws,
447        family: InstanceFamily::ComputeOptimized,
448        architecture: Architecture::X86_64,
449        vcpu: 2,
450        memory_bytes: 4 * GI,
451        ephemeral_storage_bytes: 20 * GI,
452        gpu: None,
453    },
454    InstanceTypeSpec {
455        name: "c7g.xlarge",
456        platform: Platform::Aws,
457        family: InstanceFamily::ComputeOptimized,
458        architecture: Architecture::Arm64,
459        vcpu: 4,
460        memory_bytes: 8 * GI,
461        ephemeral_storage_bytes: 20 * GI,
462        gpu: None,
463    },
464    InstanceTypeSpec {
465        name: "c8i.xlarge",
466        platform: Platform::Aws,
467        family: InstanceFamily::ComputeOptimized,
468        architecture: Architecture::X86_64,
469        vcpu: 4,
470        memory_bytes: 8 * GI,
471        ephemeral_storage_bytes: 20 * GI,
472        gpu: None,
473    },
474    InstanceTypeSpec {
475        name: "c7g.2xlarge",
476        platform: Platform::Aws,
477        family: InstanceFamily::ComputeOptimized,
478        architecture: Architecture::Arm64,
479        vcpu: 8,
480        memory_bytes: 16 * GI,
481        ephemeral_storage_bytes: 20 * GI,
482        gpu: None,
483    },
484    InstanceTypeSpec {
485        name: "c8i.2xlarge",
486        platform: Platform::Aws,
487        family: InstanceFamily::ComputeOptimized,
488        architecture: Architecture::X86_64,
489        vcpu: 8,
490        memory_bytes: 16 * GI,
491        ephemeral_storage_bytes: 20 * GI,
492        gpu: None,
493    },
494    InstanceTypeSpec {
495        name: "c7g.4xlarge",
496        platform: Platform::Aws,
497        family: InstanceFamily::ComputeOptimized,
498        architecture: Architecture::Arm64,
499        vcpu: 16,
500        memory_bytes: 32 * GI,
501        ephemeral_storage_bytes: 20 * GI,
502        gpu: None,
503    },
504    InstanceTypeSpec {
505        name: "c8i.4xlarge",
506        platform: Platform::Aws,
507        family: InstanceFamily::ComputeOptimized,
508        architecture: Architecture::X86_64,
509        vcpu: 16,
510        memory_bytes: 32 * GI,
511        ephemeral_storage_bytes: 20 * GI,
512        gpu: None,
513    },
514    // Memory Optimized (r7g — ARM Graviton3, up to 2xlarge / 8 vCPU)
515    InstanceTypeSpec {
516        name: "r7g.medium",
517        platform: Platform::Aws,
518        family: InstanceFamily::MemoryOptimized,
519        architecture: Architecture::Arm64,
520        vcpu: 1,
521        memory_bytes: 8 * GI,
522        ephemeral_storage_bytes: 20 * GI,
523        gpu: None,
524    },
525    InstanceTypeSpec {
526        name: "r7g.large",
527        platform: Platform::Aws,
528        family: InstanceFamily::MemoryOptimized,
529        architecture: Architecture::Arm64,
530        vcpu: 2,
531        memory_bytes: 16 * GI,
532        ephemeral_storage_bytes: 20 * GI,
533        gpu: None,
534    },
535    InstanceTypeSpec {
536        name: "r7g.xlarge",
537        platform: Platform::Aws,
538        family: InstanceFamily::MemoryOptimized,
539        architecture: Architecture::Arm64,
540        vcpu: 4,
541        memory_bytes: 32 * GI,
542        ephemeral_storage_bytes: 20 * GI,
543        gpu: None,
544    },
545    InstanceTypeSpec {
546        name: "r7g.2xlarge",
547        platform: Platform::Aws,
548        family: InstanceFamily::MemoryOptimized,
549        architecture: Architecture::Arm64,
550        vcpu: 8,
551        memory_bytes: 64 * GI,
552        ephemeral_storage_bytes: 20 * GI,
553        gpu: None,
554    },
555    InstanceTypeSpec {
556        name: "r7g.4xlarge",
557        platform: Platform::Aws,
558        family: InstanceFamily::MemoryOptimized,
559        architecture: Architecture::Arm64,
560        vcpu: 16,
561        memory_bytes: 128 * GI,
562        ephemeral_storage_bytes: 20 * GI,
563        gpu: None,
564    },
565    // Storage Optimized (i4i — x86_64, NVMe)
566    InstanceTypeSpec {
567        name: "i4i.xlarge",
568        platform: Platform::Aws,
569        family: InstanceFamily::StorageOptimized,
570        architecture: Architecture::X86_64,
571        vcpu: 4,
572        memory_bytes: 32 * GI,
573        ephemeral_storage_bytes: 937 * GI,
574        gpu: None,
575    },
576    InstanceTypeSpec {
577        name: "i4i.2xlarge",
578        platform: Platform::Aws,
579        family: InstanceFamily::StorageOptimized,
580        architecture: Architecture::X86_64,
581        vcpu: 8,
582        memory_bytes: 64 * GI,
583        ephemeral_storage_bytes: 1875 * GI,
584        gpu: None,
585    },
586    InstanceTypeSpec {
587        name: "i4i.4xlarge",
588        platform: Platform::Aws,
589        family: InstanceFamily::StorageOptimized,
590        architecture: Architecture::X86_64,
591        vcpu: 16,
592        memory_bytes: 128 * GI,
593        ephemeral_storage_bytes: 3750 * GI,
594        gpu: None,
595    },
596    InstanceTypeSpec {
597        name: "i4i.8xlarge",
598        platform: Platform::Aws,
599        family: InstanceFamily::StorageOptimized,
600        architecture: Architecture::X86_64,
601        vcpu: 32,
602        memory_bytes: 256 * GI,
603        ephemeral_storage_bytes: 7500 * GI,
604        gpu: None,
605    },
606    // GPU — NVIDIA T4 (g5 — x86_64)
607    InstanceTypeSpec {
608        name: "g5.xlarge",
609        platform: Platform::Aws,
610        family: InstanceFamily::GpuCompute,
611        architecture: Architecture::X86_64,
612        vcpu: 4,
613        memory_bytes: 16 * GI,
614        ephemeral_storage_bytes: 250 * GI,
615        gpu: Some(CatalogGpu {
616            gpu_type: "nvidia-t4",
617            count: 1,
618        }),
619    },
620    InstanceTypeSpec {
621        name: "g5.2xlarge",
622        platform: Platform::Aws,
623        family: InstanceFamily::GpuCompute,
624        architecture: Architecture::X86_64,
625        vcpu: 8,
626        memory_bytes: 32 * GI,
627        ephemeral_storage_bytes: 450 * GI,
628        gpu: Some(CatalogGpu {
629            gpu_type: "nvidia-t4",
630            count: 1,
631        }),
632    },
633    // GPU — NVIDIA A100 (p4d — x86_64)
634    InstanceTypeSpec {
635        name: "p4d.24xlarge",
636        platform: Platform::Aws,
637        family: InstanceFamily::GpuCompute,
638        architecture: Architecture::X86_64,
639        vcpu: 96,
640        memory_bytes: 1152 * GI,
641        ephemeral_storage_bytes: 8000 * GI,
642        gpu: Some(CatalogGpu {
643            gpu_type: "nvidia-a100",
644            count: 8,
645        }),
646    },
647    // GPU — NVIDIA H100 (p5 — x86_64)
648    InstanceTypeSpec {
649        name: "p5.48xlarge",
650        platform: Platform::Aws,
651        family: InstanceFamily::GpuCompute,
652        architecture: Architecture::X86_64,
653        vcpu: 192,
654        memory_bytes: 2048 * GI,
655        ephemeral_storage_bytes: 8000 * GI,
656        gpu: Some(CatalogGpu {
657            gpu_type: "nvidia-h100",
658            count: 8,
659        }),
660    },
661    // =========================================================================
662    // GCP
663    // =========================================================================
664
665    // Burstable (e2)
666    InstanceTypeSpec {
667        name: "e2-micro",
668        platform: Platform::Gcp,
669        family: InstanceFamily::Burstable,
670        architecture: Architecture::X86_64,
671        vcpu: 2,
672        memory_bytes: 1 * GI,
673        ephemeral_storage_bytes: 20 * GI,
674        gpu: None,
675    },
676    InstanceTypeSpec {
677        name: "e2-small",
678        platform: Platform::Gcp,
679        family: InstanceFamily::Burstable,
680        architecture: Architecture::X86_64,
681        vcpu: 2,
682        memory_bytes: 2 * GI,
683        ephemeral_storage_bytes: 20 * GI,
684        gpu: None,
685    },
686    InstanceTypeSpec {
687        name: "e2-medium",
688        platform: Platform::Gcp,
689        family: InstanceFamily::Burstable,
690        architecture: Architecture::X86_64,
691        vcpu: 2,
692        memory_bytes: 4 * GI,
693        ephemeral_storage_bytes: 20 * GI,
694        gpu: None,
695    },
696    // General Purpose (n2-standard, up to 16 vCPU)
697    InstanceTypeSpec {
698        name: "n2-standard-2",
699        platform: Platform::Gcp,
700        family: InstanceFamily::GeneralPurpose,
701        architecture: Architecture::X86_64,
702        vcpu: 2,
703        memory_bytes: 8 * GI,
704        ephemeral_storage_bytes: 20 * GI,
705        gpu: None,
706    },
707    InstanceTypeSpec {
708        name: "n2-standard-4",
709        platform: Platform::Gcp,
710        family: InstanceFamily::GeneralPurpose,
711        architecture: Architecture::X86_64,
712        vcpu: 4,
713        memory_bytes: 16 * GI,
714        ephemeral_storage_bytes: 20 * GI,
715        gpu: None,
716    },
717    InstanceTypeSpec {
718        name: "n2-standard-8",
719        platform: Platform::Gcp,
720        family: InstanceFamily::GeneralPurpose,
721        architecture: Architecture::X86_64,
722        vcpu: 8,
723        memory_bytes: 32 * GI,
724        ephemeral_storage_bytes: 20 * GI,
725        gpu: None,
726    },
727    InstanceTypeSpec {
728        name: "n2-standard-16",
729        platform: Platform::Gcp,
730        family: InstanceFamily::GeneralPurpose,
731        architecture: Architecture::X86_64,
732        vcpu: 16,
733        memory_bytes: 64 * GI,
734        ephemeral_storage_bytes: 20 * GI,
735        gpu: None,
736    },
737    // Compute Optimized (c3-standard, up to 8 vCPU)
738    InstanceTypeSpec {
739        name: "c3-standard-4",
740        platform: Platform::Gcp,
741        family: InstanceFamily::ComputeOptimized,
742        architecture: Architecture::X86_64,
743        vcpu: 4,
744        memory_bytes: 8 * GI,
745        ephemeral_storage_bytes: 20 * GI,
746        gpu: None,
747    },
748    InstanceTypeSpec {
749        name: "c3-standard-8",
750        platform: Platform::Gcp,
751        family: InstanceFamily::ComputeOptimized,
752        architecture: Architecture::X86_64,
753        vcpu: 8,
754        memory_bytes: 16 * GI,
755        ephemeral_storage_bytes: 20 * GI,
756        gpu: None,
757    },
758    // Memory Optimized (n2-highmem, up to 8 vCPU)
759    InstanceTypeSpec {
760        name: "n2-highmem-2",
761        platform: Platform::Gcp,
762        family: InstanceFamily::MemoryOptimized,
763        architecture: Architecture::X86_64,
764        vcpu: 2,
765        memory_bytes: 16 * GI,
766        ephemeral_storage_bytes: 20 * GI,
767        gpu: None,
768    },
769    InstanceTypeSpec {
770        name: "n2-highmem-4",
771        platform: Platform::Gcp,
772        family: InstanceFamily::MemoryOptimized,
773        architecture: Architecture::X86_64,
774        vcpu: 4,
775        memory_bytes: 32 * GI,
776        ephemeral_storage_bytes: 20 * GI,
777        gpu: None,
778    },
779    InstanceTypeSpec {
780        name: "n2-highmem-8",
781        platform: Platform::Gcp,
782        family: InstanceFamily::MemoryOptimized,
783        architecture: Architecture::X86_64,
784        vcpu: 8,
785        memory_bytes: 64 * GI,
786        ephemeral_storage_bytes: 20 * GI,
787        gpu: None,
788    },
789    InstanceTypeSpec {
790        name: "n2-highmem-16",
791        platform: Platform::Gcp,
792        family: InstanceFamily::MemoryOptimized,
793        architecture: Architecture::X86_64,
794        vcpu: 16,
795        memory_bytes: 128 * GI,
796        ephemeral_storage_bytes: 20 * GI,
797        gpu: None,
798    },
799    InstanceTypeSpec {
800        name: "n2-highmem-32",
801        platform: Platform::Gcp,
802        family: InstanceFamily::MemoryOptimized,
803        architecture: Architecture::X86_64,
804        vcpu: 32,
805        memory_bytes: 256 * GI,
806        ephemeral_storage_bytes: 20 * GI,
807        gpu: None,
808    },
809    // Storage Optimized (c3d-standard with local SSD)
810    InstanceTypeSpec {
811        name: "c3d-standard-8",
812        platform: Platform::Gcp,
813        family: InstanceFamily::StorageOptimized,
814        architecture: Architecture::X86_64,
815        vcpu: 8,
816        memory_bytes: 32 * GI,
817        ephemeral_storage_bytes: 480 * GI,
818        gpu: None,
819    },
820    InstanceTypeSpec {
821        name: "c3d-standard-16",
822        platform: Platform::Gcp,
823        family: InstanceFamily::StorageOptimized,
824        architecture: Architecture::X86_64,
825        vcpu: 16,
826        memory_bytes: 64 * GI,
827        ephemeral_storage_bytes: 960 * GI,
828        gpu: None,
829    },
830    InstanceTypeSpec {
831        name: "c3d-standard-30",
832        platform: Platform::Gcp,
833        family: InstanceFamily::StorageOptimized,
834        architecture: Architecture::X86_64,
835        vcpu: 30,
836        memory_bytes: 120 * GI,
837        ephemeral_storage_bytes: 1920 * GI,
838        gpu: None,
839    },
840    // GPU — NVIDIA T4 (n1-standard + T4)
841    InstanceTypeSpec {
842        name: "n1-standard-4-t4",
843        platform: Platform::Gcp,
844        family: InstanceFamily::GpuCompute,
845        architecture: Architecture::X86_64,
846        vcpu: 4,
847        memory_bytes: 15 * GI,
848        ephemeral_storage_bytes: 100 * GI,
849        gpu: Some(CatalogGpu {
850            gpu_type: "nvidia-t4",
851            count: 1,
852        }),
853    },
854    // GPU — NVIDIA A100 (a2-highgpu)
855    InstanceTypeSpec {
856        name: "a2-highgpu-1g",
857        platform: Platform::Gcp,
858        family: InstanceFamily::GpuCompute,
859        architecture: Architecture::X86_64,
860        vcpu: 12,
861        memory_bytes: 85 * GI,
862        ephemeral_storage_bytes: 100 * GI,
863        gpu: Some(CatalogGpu {
864            gpu_type: "nvidia-a100",
865            count: 1,
866        }),
867    },
868    InstanceTypeSpec {
869        name: "a2-highgpu-8g",
870        platform: Platform::Gcp,
871        family: InstanceFamily::GpuCompute,
872        architecture: Architecture::X86_64,
873        vcpu: 96,
874        memory_bytes: 1360 * GI,
875        ephemeral_storage_bytes: 100 * GI,
876        gpu: Some(CatalogGpu {
877            gpu_type: "nvidia-a100",
878            count: 8,
879        }),
880    },
881    // GPU — NVIDIA H100 (a3-highgpu)
882    InstanceTypeSpec {
883        name: "a3-highgpu-8g",
884        platform: Platform::Gcp,
885        family: InstanceFamily::GpuCompute,
886        architecture: Architecture::X86_64,
887        vcpu: 208,
888        memory_bytes: 1872 * GI,
889        ephemeral_storage_bytes: 100 * GI,
890        gpu: Some(CatalogGpu {
891            gpu_type: "nvidia-h100",
892            count: 8,
893        }),
894    },
895    // =========================================================================
896    // Azure
897    // =========================================================================
898
899    // Burstable (B-series v2)
900    InstanceTypeSpec {
901        name: "Standard_B1s",
902        platform: Platform::Azure,
903        family: InstanceFamily::Burstable,
904        architecture: Architecture::X86_64,
905        vcpu: 1,
906        memory_bytes: 1 * GI,
907        ephemeral_storage_bytes: 20 * GI,
908        gpu: None,
909    },
910    InstanceTypeSpec {
911        name: "Standard_B2s",
912        platform: Platform::Azure,
913        family: InstanceFamily::Burstable,
914        architecture: Architecture::X86_64,
915        vcpu: 2,
916        memory_bytes: 4 * GI,
917        ephemeral_storage_bytes: 20 * GI,
918        gpu: None,
919    },
920    InstanceTypeSpec {
921        name: "Standard_B2ms",
922        platform: Platform::Azure,
923        family: InstanceFamily::Burstable,
924        architecture: Architecture::X86_64,
925        vcpu: 2,
926        memory_bytes: 8 * GI,
927        ephemeral_storage_bytes: 20 * GI,
928        gpu: None,
929    },
930    InstanceTypeSpec {
931        name: "Standard_B4ms",
932        platform: Platform::Azure,
933        family: InstanceFamily::Burstable,
934        architecture: Architecture::X86_64,
935        vcpu: 4,
936        memory_bytes: 16 * GI,
937        ephemeral_storage_bytes: 20 * GI,
938        gpu: None,
939    },
940    // General Purpose (Dv5-series, up to 16 vCPU)
941    InstanceTypeSpec {
942        name: "Standard_D2s_v5",
943        platform: Platform::Azure,
944        family: InstanceFamily::GeneralPurpose,
945        architecture: Architecture::X86_64,
946        vcpu: 2,
947        memory_bytes: 8 * GI,
948        ephemeral_storage_bytes: 20 * GI,
949        gpu: None,
950    },
951    InstanceTypeSpec {
952        name: "Standard_D4s_v5",
953        platform: Platform::Azure,
954        family: InstanceFamily::GeneralPurpose,
955        architecture: Architecture::X86_64,
956        vcpu: 4,
957        memory_bytes: 16 * GI,
958        ephemeral_storage_bytes: 20 * GI,
959        gpu: None,
960    },
961    InstanceTypeSpec {
962        name: "Standard_D8s_v5",
963        platform: Platform::Azure,
964        family: InstanceFamily::GeneralPurpose,
965        architecture: Architecture::X86_64,
966        vcpu: 8,
967        memory_bytes: 32 * GI,
968        ephemeral_storage_bytes: 20 * GI,
969        gpu: None,
970    },
971    InstanceTypeSpec {
972        name: "Standard_D16s_v5",
973        platform: Platform::Azure,
974        family: InstanceFamily::GeneralPurpose,
975        architecture: Architecture::X86_64,
976        vcpu: 16,
977        memory_bytes: 64 * GI,
978        ephemeral_storage_bytes: 20 * GI,
979        gpu: None,
980    },
981    // Compute Optimized (Fv2-series, up to 16 vCPU)
982    InstanceTypeSpec {
983        name: "Standard_F2s_v2",
984        platform: Platform::Azure,
985        family: InstanceFamily::ComputeOptimized,
986        architecture: Architecture::X86_64,
987        vcpu: 2,
988        memory_bytes: 4 * GI,
989        ephemeral_storage_bytes: 20 * GI,
990        gpu: None,
991    },
992    InstanceTypeSpec {
993        name: "Standard_F4s_v2",
994        platform: Platform::Azure,
995        family: InstanceFamily::ComputeOptimized,
996        architecture: Architecture::X86_64,
997        vcpu: 4,
998        memory_bytes: 8 * GI,
999        ephemeral_storage_bytes: 20 * GI,
1000        gpu: None,
1001    },
1002    InstanceTypeSpec {
1003        name: "Standard_F8s_v2",
1004        platform: Platform::Azure,
1005        family: InstanceFamily::ComputeOptimized,
1006        architecture: Architecture::X86_64,
1007        vcpu: 8,
1008        memory_bytes: 16 * GI,
1009        ephemeral_storage_bytes: 20 * GI,
1010        gpu: None,
1011    },
1012    InstanceTypeSpec {
1013        name: "Standard_F16s_v2",
1014        platform: Platform::Azure,
1015        family: InstanceFamily::ComputeOptimized,
1016        architecture: Architecture::X86_64,
1017        vcpu: 16,
1018        memory_bytes: 32 * GI,
1019        ephemeral_storage_bytes: 20 * GI,
1020        gpu: None,
1021    },
1022    // Memory Optimized (Ev5-series, up to 16 vCPU)
1023    InstanceTypeSpec {
1024        name: "Standard_E2s_v5",
1025        platform: Platform::Azure,
1026        family: InstanceFamily::MemoryOptimized,
1027        architecture: Architecture::X86_64,
1028        vcpu: 2,
1029        memory_bytes: 16 * GI,
1030        ephemeral_storage_bytes: 20 * GI,
1031        gpu: None,
1032    },
1033    InstanceTypeSpec {
1034        name: "Standard_E4s_v5",
1035        platform: Platform::Azure,
1036        family: InstanceFamily::MemoryOptimized,
1037        architecture: Architecture::X86_64,
1038        vcpu: 4,
1039        memory_bytes: 32 * GI,
1040        ephemeral_storage_bytes: 20 * GI,
1041        gpu: None,
1042    },
1043    InstanceTypeSpec {
1044        name: "Standard_E8s_v5",
1045        platform: Platform::Azure,
1046        family: InstanceFamily::MemoryOptimized,
1047        architecture: Architecture::X86_64,
1048        vcpu: 8,
1049        memory_bytes: 64 * GI,
1050        ephemeral_storage_bytes: 20 * GI,
1051        gpu: None,
1052    },
1053    InstanceTypeSpec {
1054        name: "Standard_E16s_v5",
1055        platform: Platform::Azure,
1056        family: InstanceFamily::MemoryOptimized,
1057        architecture: Architecture::X86_64,
1058        vcpu: 16,
1059        memory_bytes: 128 * GI,
1060        ephemeral_storage_bytes: 20 * GI,
1061        gpu: None,
1062    },
1063    // Storage Optimized (Lsv3-series with NVMe)
1064    InstanceTypeSpec {
1065        name: "Standard_L8s_v3",
1066        platform: Platform::Azure,
1067        family: InstanceFamily::StorageOptimized,
1068        architecture: Architecture::X86_64,
1069        vcpu: 8,
1070        memory_bytes: 64 * GI,
1071        ephemeral_storage_bytes: 1788 * GI,
1072        gpu: None,
1073    },
1074    InstanceTypeSpec {
1075        name: "Standard_L16s_v3",
1076        platform: Platform::Azure,
1077        family: InstanceFamily::StorageOptimized,
1078        architecture: Architecture::X86_64,
1079        vcpu: 16,
1080        memory_bytes: 128 * GI,
1081        ephemeral_storage_bytes: 3576 * GI,
1082        gpu: None,
1083    },
1084    InstanceTypeSpec {
1085        name: "Standard_L32s_v3",
1086        platform: Platform::Azure,
1087        family: InstanceFamily::StorageOptimized,
1088        architecture: Architecture::X86_64,
1089        vcpu: 32,
1090        memory_bytes: 256 * GI,
1091        ephemeral_storage_bytes: 7154 * GI,
1092        gpu: None,
1093    },
1094    // GPU — NVIDIA T4 (NCasT4_v3-series)
1095    InstanceTypeSpec {
1096        name: "Standard_NC4as_T4_v3",
1097        platform: Platform::Azure,
1098        family: InstanceFamily::GpuCompute,
1099        architecture: Architecture::X86_64,
1100        vcpu: 4,
1101        memory_bytes: 28 * GI,
1102        ephemeral_storage_bytes: 176 * GI,
1103        gpu: Some(CatalogGpu {
1104            gpu_type: "nvidia-t4",
1105            count: 1,
1106        }),
1107    },
1108    // GPU — NVIDIA A100 (NC A100 v4-series)
1109    InstanceTypeSpec {
1110        name: "Standard_NC24ads_A100_v4",
1111        platform: Platform::Azure,
1112        family: InstanceFamily::GpuCompute,
1113        architecture: Architecture::X86_64,
1114        vcpu: 24,
1115        memory_bytes: 220 * GI,
1116        ephemeral_storage_bytes: 958 * GI,
1117        gpu: Some(CatalogGpu {
1118            gpu_type: "nvidia-a100",
1119            count: 1,
1120        }),
1121    },
1122    InstanceTypeSpec {
1123        name: "Standard_NC96ads_A100_v4",
1124        platform: Platform::Azure,
1125        family: InstanceFamily::GpuCompute,
1126        architecture: Architecture::X86_64,
1127        vcpu: 96,
1128        memory_bytes: 880 * GI,
1129        ephemeral_storage_bytes: 3916 * GI,
1130        gpu: Some(CatalogGpu {
1131            gpu_type: "nvidia-a100",
1132            count: 4,
1133        }),
1134    },
1135    // GPU — NVIDIA H100 (ND H100 v5-series)
1136    InstanceTypeSpec {
1137        name: "Standard_ND96isr_H100_v5",
1138        platform: Platform::Azure,
1139        family: InstanceFamily::GpuCompute,
1140        architecture: Architecture::X86_64,
1141        vcpu: 96,
1142        memory_bytes: 1900 * GI,
1143        ephemeral_storage_bytes: 1000 * GI,
1144        gpu: Some(CatalogGpu {
1145            gpu_type: "nvidia-h100",
1146            count: 8,
1147        }),
1148    },
1149];
1150
1151// ---------------------------------------------------------------------------
1152// Catalog lookup
1153// ---------------------------------------------------------------------------
1154
1155/// Get all instance types for a given platform.
1156pub fn catalog_for_platform(platform: Platform) -> Vec<&'static InstanceTypeSpec> {
1157    CATALOG
1158        .iter()
1159        .filter(|spec| spec.platform == platform)
1160        .collect()
1161}
1162
1163/// Find a specific instance type by name and platform.
1164pub fn find_instance_type(platform: Platform, name: &str) -> Option<&'static InstanceTypeSpec> {
1165    CATALOG
1166        .iter()
1167        .find(|spec| spec.platform == platform && spec.name == name)
1168}
1169
1170// ---------------------------------------------------------------------------
1171// Instance type selection
1172// ---------------------------------------------------------------------------
1173
1174/// Aggregated resource requirements from all containers in a capacity group.
1175#[derive(Debug, Clone)]
1176pub struct WorkloadRequirements {
1177    /// Total CPU needed at desired scale (sum of desired CPU * desired_replicas per container)
1178    pub total_cpu_at_desired: f64,
1179    /// Total memory needed at desired scale (sum of desired memory * desired_replicas per container)
1180    pub total_memory_bytes_at_desired: u64,
1181    /// Total CPU needed at maximum scale (sum of desired CPU * max_replicas per container)
1182    pub total_cpu_at_max: f64,
1183    /// Total memory needed at maximum scale (sum of desired memory * max_replicas per container)
1184    pub total_memory_bytes_at_max: u64,
1185    /// Largest CPU request among all individual containers (single replica)
1186    pub max_cpu_per_container: f64,
1187    /// Largest memory request among all individual containers (single replica)
1188    pub max_memory_per_container: u64,
1189    /// Maximum ephemeral storage any single container requires
1190    pub max_ephemeral_storage_bytes: u64,
1191    /// GPU requirement (if any container needs GPU)
1192    pub gpu: Option<GpuSpec>,
1193    /// Required CPU architecture, when source explicitly constrains it.
1194    pub architecture: Option<Architecture>,
1195    /// If true, only instance types that expose nested virtualization (VT-x/EPT)
1196    /// to guest VMs are eligible. Required by workloads that run QEMU/KVM
1197    /// inside a container.
1198    pub nested_virt: bool,
1199}
1200
1201/// Result of instance type selection.
1202#[derive(Debug, Clone)]
1203pub struct InstanceSelection {
1204    /// Selected instance type name (e.g., "m7g.2xlarge")
1205    pub instance_type: &'static str,
1206    /// Machine profile derived from the instance type
1207    pub profile: MachineProfile,
1208    /// Recommended minimum number of machines
1209    pub min_machines: u32,
1210    /// Recommended maximum number of machines
1211    pub max_machines: u32,
1212}
1213
1214/// Ephemeral storage threshold above which storage-optimized instances are selected.
1215const STORAGE_OPTIMIZED_THRESHOLD: u64 = 200 * GI;
1216
1217/// Maximum number of machines per cluster.
1218const MAX_MACHINES_PER_CLUSTER: u32 = 10;
1219
1220/// Hard cap on vCPUs for non-GPU/non-storage workloads. Equivalent to AWS 2xlarge.
1221/// Beyond this, horizontal scaling is always preferred over bigger machines.
1222const MAX_STANDARD_VCPU: u32 = 8;
1223
1224/// Runtime CPU reserved for system processes on each managed container machine.
1225const SYSTEM_RESERVE_CPU: f64 = 0.5;
1226
1227/// Runtime planning headroom for total desired/max workload.
1228const WORKLOAD_HEADROOM_FACTOR: f64 = 1.15;
1229
1230/// Select the best instance type for a workload on a given platform.
1231///
1232/// The algorithm:
1233/// 1. GPU workloads: Match by GPU type, find smallest instance with enough GPUs.
1234/// 2. Storage-heavy workloads (>200Gi ephemeral): Use storage-optimized instances.
1235/// 3. All other workloads: Size the machine to fit a small HA-friendly baseline,
1236///    capped at 8 vCPUs. Use GeneralPurpose family for broad availability and
1237///    reasonable cost. Scale horizontally for more capacity.
1238///
1239/// Returns an error if no suitable instance type is found.
1240pub fn select_instance_type(
1241    platform: Platform,
1242    requirements: &WorkloadRequirements,
1243) -> Result<InstanceSelection, String> {
1244    // Determine which family to use. Nested virt isn't available on
1245    // burstable hardware on any cloud, so a workload that classifies as
1246    // Burstable but needs nested virt must be upgraded to GeneralPurpose
1247    // (the family that actually has nested-virt-capable entries).
1248    let raw_family = select_family(requirements);
1249    let family = if requirements.nested_virt && raw_family == InstanceFamily::Burstable {
1250        InstanceFamily::GeneralPurpose
1251    } else {
1252        raw_family
1253    };
1254
1255    let candidates: Vec<&InstanceTypeSpec> = CATALOG
1256        .iter()
1257        .filter(|spec| spec.platform == platform && spec.family == family)
1258        .filter(|spec| {
1259            if requirements.nested_virt {
1260                spec.is_nested_virt_capable()
1261            } else {
1262                platform != Platform::Aws || !spec.is_nested_virt_capable()
1263            }
1264        })
1265        .collect();
1266
1267    if candidates.is_empty() {
1268        return Err(if requirements.nested_virt {
1269            format!(
1270                "no nested-virt-capable {family:?} instance types in catalog for platform {platform}"
1271            )
1272        } else {
1273            format!("no {family:?} instance types in catalog for platform {platform}")
1274        });
1275    }
1276
1277    // For GPU workloads, filter by GPU type
1278    let candidates = if let Some(ref gpu) = requirements.gpu {
1279        let filtered: Vec<&InstanceTypeSpec> = candidates
1280            .into_iter()
1281            .filter(|spec| {
1282                spec.gpu.as_ref().map_or(false, |g| {
1283                    g.gpu_type == gpu.gpu_type && g.count >= gpu.count
1284                })
1285            })
1286            .collect();
1287        if filtered.is_empty() {
1288            return Err(format!(
1289                "no instance type for GPU type '{}' x{} on platform {platform}",
1290                gpu.gpu_type, gpu.count
1291            ));
1292        }
1293        filtered
1294    } else {
1295        candidates
1296    };
1297
1298    // For storage workloads, filter by ephemeral storage capacity
1299    let candidates = if family == InstanceFamily::StorageOptimized {
1300        let filtered: Vec<&InstanceTypeSpec> = candidates
1301            .into_iter()
1302            .filter(|spec| spec.ephemeral_storage_bytes >= requirements.max_ephemeral_storage_bytes)
1303            .collect();
1304        if filtered.is_empty() {
1305            return Err(format!(
1306                "no storage-optimized instance with >= {} bytes ephemeral storage on platform {platform}",
1307                requirements.max_ephemeral_storage_bytes
1308            ));
1309        }
1310        filtered
1311    } else {
1312        candidates
1313    };
1314
1315    let architecture = requirements
1316        .architecture
1317        .or_else(|| default_architecture(platform))
1318        .ok_or_else(|| format!("platform {platform} has no default compute architecture"))?;
1319    let candidates: Vec<&InstanceTypeSpec> = candidates
1320        .into_iter()
1321        .filter(|spec| spec.architecture == architecture)
1322        .collect();
1323    if candidates.is_empty() {
1324        return Err(format!(
1325            "architecture {architecture:?} is unavailable for this workload on platform {platform}"
1326        ));
1327    }
1328
1329    // Cap at MAX_STANDARD_VCPU for non-GPU/non-storage workloads
1330    let vcpu_cap =
1331        if family == InstanceFamily::GpuCompute || family == InstanceFamily::StorageOptimized {
1332            u32::MAX
1333        } else {
1334            MAX_STANDARD_VCPU
1335        };
1336
1337    let desired_target_machines = desired_target_machines(requirements);
1338    let target_cpu = requirements
1339        .max_cpu_per_container
1340        .max(requirements.total_cpu_at_desired / desired_target_machines as f64)
1341        * WORKLOAD_HEADROOM_FACTOR;
1342    let target_memory = (requirements.max_memory_per_container as f64)
1343        .max(requirements.total_memory_bytes_at_desired as f64 / desired_target_machines as f64)
1344        * WORKLOAD_HEADROOM_FACTOR;
1345
1346    // Find the smallest instance whose allocatable capacity meets the workload
1347    // target after host reserve and workload headroom. Machine count already
1348    // accounts for multiple replicas; requiring space for an arbitrary second
1349    // copy here would size the same demand twice.
1350    let selected = candidates
1351        .iter()
1352        .filter(|spec| {
1353            spec.vcpu <= vcpu_cap
1354                && allocatable_cpu(spec) >= target_cpu
1355                && allocatable_memory_bytes(spec) as f64 >= target_memory
1356        })
1357        .min_by_key(|spec| spec.vcpu)
1358        .or_else(|| {
1359            // If nothing fits within the cap, pick the largest instance under the cap
1360            candidates
1361                .iter()
1362                .filter(|spec| spec.vcpu <= vcpu_cap)
1363                .max_by_key(|spec| spec.vcpu)
1364        })
1365        .or_else(|| {
1366            // Last resort: pick the smallest available instance (for GPU/storage)
1367            candidates.iter().min_by_key(|spec| spec.vcpu)
1368        })
1369        .ok_or_else(|| format!("no instance types available for platform {platform}"))?;
1370
1371    // Calculate machine counts
1372    let max_machines = compute_max_machines(requirements, selected);
1373    let min_machines = compute_min_machines(requirements, selected, max_machines);
1374
1375    Ok(InstanceSelection {
1376        instance_type: selected.name,
1377        profile: selected.to_machine_profile(),
1378        min_machines,
1379        max_machines,
1380    })
1381}
1382
1383/// Select instance family based on workload characteristics.
1384///
1385/// Uses GeneralPurpose for all standard workloads — widely available across
1386/// regions and cost-effective. Only specialized workloads (GPU, large ephemeral
1387/// storage) get specialized families. Very small workloads get burstable.
1388pub fn select_family(requirements: &WorkloadRequirements) -> InstanceFamily {
1389    // GPU workloads always get GPU instances
1390    if requirements.gpu.is_some() {
1391        return InstanceFamily::GpuCompute;
1392    }
1393
1394    // Large ephemeral storage needs NVMe (storage-optimized)
1395    if requirements.max_ephemeral_storage_bytes > STORAGE_OPTIMIZED_THRESHOLD {
1396        return InstanceFamily::StorageOptimized;
1397    }
1398
1399    // Very small workloads use burstable instances
1400    if requirements.total_cpu_at_max < 2.0 {
1401        return InstanceFamily::Burstable;
1402    }
1403
1404    // All other workloads use GeneralPurpose — available everywhere, good pricing
1405    InstanceFamily::GeneralPurpose
1406}
1407
1408/// Calculate maximum machines needed to fit the workload with headroom.
1409fn compute_max_machines(requirements: &WorkloadRequirements, instance: &InstanceTypeSpec) -> u32 {
1410    let cpu_with_headroom = requirements.total_cpu_at_max * WORKLOAD_HEADROOM_FACTOR;
1411    let cpu_machines = (cpu_with_headroom / allocatable_cpu(instance)).ceil() as u32;
1412
1413    let mem_with_headroom =
1414        requirements.total_memory_bytes_at_max as f64 * WORKLOAD_HEADROOM_FACTOR;
1415    let mem_machines =
1416        (mem_with_headroom / allocatable_memory_bytes(instance) as f64).ceil() as u32;
1417
1418    // Take the larger of CPU-based and memory-based, clamped to cluster limit
1419    cpu_machines
1420        .max(mem_machines)
1421        .max(1)
1422        .min(MAX_MACHINES_PER_CLUSTER)
1423}
1424
1425/// Calculate minimum machines for HA.
1426fn compute_min_machines(
1427    requirements: &WorkloadRequirements,
1428    instance: &InstanceTypeSpec,
1429    max_machines: u32,
1430) -> u32 {
1431    let cpu_with_headroom = requirements.total_cpu_at_desired * WORKLOAD_HEADROOM_FACTOR;
1432    let cpu_machines = (cpu_with_headroom / allocatable_cpu(instance)).ceil() as u32;
1433
1434    let mem_with_headroom =
1435        requirements.total_memory_bytes_at_desired as f64 * WORKLOAD_HEADROOM_FACTOR;
1436    let mem_machines =
1437        (mem_with_headroom / allocatable_memory_bytes(instance) as f64).ceil() as u32;
1438
1439    cpu_machines
1440        .max(mem_machines)
1441        .max(1)
1442        .min(2)
1443        .min(max_machines)
1444}
1445
1446fn desired_target_machines(requirements: &WorkloadRequirements) -> u32 {
1447    if requirements.total_cpu_at_desired >= 2.0
1448        || requirements.total_memory_bytes_at_desired >= 4 * GI
1449    {
1450        2
1451    } else {
1452        1
1453    }
1454}
1455
1456fn allocatable_cpu(instance: &InstanceTypeSpec) -> f64 {
1457    (instance.vcpu as f64 - SYSTEM_RESERVE_CPU).max(0.25)
1458}
1459
1460fn allocatable_memory_bytes(instance: &InstanceTypeSpec) -> u64 {
1461    instance
1462        .memory_bytes
1463        .saturating_sub(system_reserve_memory_bytes(instance.memory_bytes))
1464        .max(256 * MI)
1465}
1466
1467fn system_reserve_memory_bytes(memory_bytes: u64) -> u64 {
1468    if memory_bytes < 4 * GI {
1469        256 * MI
1470    } else if memory_bytes < 16 * GI {
1471        512 * MI
1472    } else {
1473        GI
1474    }
1475}
1476
1477// ---------------------------------------------------------------------------
1478// Tests
1479// ---------------------------------------------------------------------------
1480
1481#[cfg(test)]
1482mod tests {
1483    use super::*;
1484    use crate::BinaryTarget;
1485
1486    // -- Parsing tests --
1487
1488    #[test]
1489    fn test_parse_cpu_plain() {
1490        assert_eq!(parse_cpu("1").unwrap(), 1.0);
1491        assert_eq!(parse_cpu("0.5").unwrap(), 0.5);
1492        assert_eq!(parse_cpu("2.0").unwrap(), 2.0);
1493        assert_eq!(parse_cpu("16").unwrap(), 16.0);
1494    }
1495
1496    #[test]
1497    fn test_parse_cpu_millicore() {
1498        assert_eq!(parse_cpu("500m").unwrap(), 0.5);
1499        assert_eq!(parse_cpu("250m").unwrap(), 0.25);
1500        assert_eq!(parse_cpu("1000m").unwrap(), 1.0);
1501        assert_eq!(parse_cpu("100m").unwrap(), 0.1);
1502    }
1503
1504    #[test]
1505    fn test_parse_cpu_invalid() {
1506        assert!(parse_cpu("").is_err());
1507        assert!(parse_cpu("abc").is_err());
1508        assert!(parse_cpu("m").is_err());
1509    }
1510
1511    #[test]
1512    fn test_parse_memory_binary_suffixes() {
1513        assert_eq!(parse_memory_bytes("1Ki").unwrap(), 1024);
1514        assert_eq!(parse_memory_bytes("1Mi").unwrap(), 1024 * 1024);
1515        assert_eq!(parse_memory_bytes("1Gi").unwrap(), 1024 * 1024 * 1024);
1516        assert_eq!(parse_memory_bytes("4Gi").unwrap(), 4 * 1024 * 1024 * 1024);
1517        assert_eq!(parse_memory_bytes("512Mi").unwrap(), 512 * 1024 * 1024);
1518        assert_eq!(
1519            parse_memory_bytes("1Ti").unwrap(),
1520            1024u64 * 1024 * 1024 * 1024
1521        );
1522    }
1523
1524    #[test]
1525    fn test_parse_memory_decimal_suffixes() {
1526        assert_eq!(parse_memory_bytes("1k").unwrap(), 1000);
1527        assert_eq!(parse_memory_bytes("1M").unwrap(), 1_000_000);
1528        assert_eq!(parse_memory_bytes("1G").unwrap(), 1_000_000_000);
1529        assert_eq!(parse_memory_bytes("1T").unwrap(), 1_000_000_000_000);
1530    }
1531
1532    #[test]
1533    fn test_parse_memory_plain_bytes() {
1534        assert_eq!(parse_memory_bytes("1024").unwrap(), 1024);
1535        assert_eq!(parse_memory_bytes("0").unwrap(), 0);
1536    }
1537
1538    #[test]
1539    fn test_parse_memory_invalid() {
1540        assert!(parse_memory_bytes("").is_err());
1541        assert!(parse_memory_bytes("abc").is_err());
1542        assert!(parse_memory_bytes("Gi").is_err());
1543    }
1544
1545    #[test]
1546    fn test_parse_memory_fractional() {
1547        assert_eq!(parse_memory_bytes("0.5Gi").unwrap(), GI / 2);
1548        assert_eq!(parse_memory_bytes("1.5Gi").unwrap(), GI + GI / 2);
1549    }
1550
1551    // -- Catalog lookup tests --
1552
1553    #[test]
1554    fn test_catalog_has_entries_for_all_cloud_platforms() {
1555        assert!(!catalog_for_platform(Platform::Aws).is_empty());
1556        assert!(!catalog_for_platform(Platform::Gcp).is_empty());
1557        assert!(!catalog_for_platform(Platform::Azure).is_empty());
1558    }
1559
1560    #[test]
1561    fn test_catalog_no_entries_for_non_cloud_platforms() {
1562        assert!(catalog_for_platform(Platform::Local).is_empty());
1563        assert!(catalog_for_platform(Platform::Kubernetes).is_empty());
1564    }
1565
1566    #[test]
1567    fn test_find_known_instance_type() {
1568        let spec =
1569            find_instance_type(Platform::Aws, "m7g.2xlarge").expect("should find m7g.2xlarge");
1570        assert_eq!(spec.vcpu, 8);
1571        assert_eq!(spec.memory_bytes, 32 * GI);
1572        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1573    }
1574
1575    #[test]
1576    fn test_find_aws_c8i_nested_virt_instance_type() {
1577        let spec = find_instance_type(Platform::Aws, "c8i.large").expect("should find c8i.large");
1578        assert_eq!(spec.vcpu, 2);
1579        assert_eq!(spec.memory_bytes, 4 * GI);
1580        assert_eq!(spec.family, InstanceFamily::ComputeOptimized);
1581        assert_eq!(spec.architecture, Architecture::X86_64);
1582        assert!(spec.is_nested_virt_capable());
1583    }
1584
1585    #[test]
1586    fn test_find_unknown_instance_type() {
1587        assert!(find_instance_type(Platform::Aws, "nonexistent.xlarge").is_none());
1588    }
1589
1590    #[test]
1591    fn test_find_wrong_platform() {
1592        assert!(find_instance_type(Platform::Gcp, "m7g.2xlarge").is_none());
1593    }
1594
1595    #[test]
1596    fn test_to_machine_profile() {
1597        let spec = find_instance_type(Platform::Aws, "m7g.2xlarge").unwrap();
1598        let profile = spec.to_machine_profile();
1599        assert_eq!(profile.cpu, "8.0");
1600        assert_eq!(profile.memory_bytes, 32 * GI);
1601        assert_eq!(profile.ephemeral_storage_bytes, 20 * GI);
1602        assert!(profile.gpu.is_none());
1603    }
1604
1605    #[test]
1606    fn test_to_machine_profile_with_gpu() {
1607        let spec = find_instance_type(Platform::Aws, "p4d.24xlarge").unwrap();
1608        let profile = spec.to_machine_profile();
1609        let gpu = profile.gpu.as_ref().expect("should have GPU");
1610        assert_eq!(gpu.gpu_type, "nvidia-a100");
1611        assert_eq!(gpu.count, 8);
1612    }
1613
1614    // -- Selection algorithm tests --
1615
1616    #[test]
1617    fn test_select_burstable_for_small_workload() {
1618        let req = WorkloadRequirements {
1619            total_cpu_at_desired: 1.0,
1620            total_memory_bytes_at_desired: 2 * GI,
1621            total_cpu_at_max: 1.0,
1622            total_memory_bytes_at_max: 2 * GI,
1623            max_cpu_per_container: 0.5,
1624            max_memory_per_container: 1 * GI,
1625            max_ephemeral_storage_bytes: 10 * GI,
1626            gpu: None,
1627            architecture: None,
1628            nested_virt: false,
1629        };
1630        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1631        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1632        assert_eq!(spec.family, InstanceFamily::Burstable);
1633    }
1634
1635    #[test]
1636    fn test_selects_smallest_burstable_machine_with_real_headroom() {
1637        let req = WorkloadRequirements {
1638            total_cpu_at_desired: 1.0,
1639            total_memory_bytes_at_desired: 2 * GI,
1640            total_cpu_at_max: 1.0,
1641            total_memory_bytes_at_max: 2 * GI,
1642            max_cpu_per_container: 1.0,
1643            max_memory_per_container: 2 * GI,
1644            max_ephemeral_storage_bytes: 10 * GI,
1645            gpu: None,
1646            architecture: None,
1647            nested_virt: false,
1648        };
1649
1650        let selection = select_instance_type(Platform::Aws, &req).unwrap();
1651
1652        assert_eq!(selection.instance_type, "t4g.medium");
1653        assert_eq!(selection.min_machines, 1);
1654        assert_eq!(selection.max_machines, 1);
1655    }
1656
1657    #[test]
1658    fn test_select_general_purpose_for_standard_workload() {
1659        // Standard workloads always get GeneralPurpose regardless of CPU:memory ratio
1660        let req = WorkloadRequirements {
1661            total_cpu_at_desired: 20.0,
1662            total_memory_bytes_at_desired: 80 * GI,
1663            total_cpu_at_max: 20.0,
1664            total_memory_bytes_at_max: 80 * GI,
1665            max_cpu_per_container: 2.0,
1666            max_memory_per_container: 8 * GI,
1667            max_ephemeral_storage_bytes: 10 * GI,
1668            gpu: None,
1669            architecture: None,
1670            nested_virt: false,
1671        };
1672        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1673        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1674        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1675    }
1676
1677    #[test]
1678    fn test_select_general_purpose_even_for_cpu_heavy() {
1679        // CPU-heavy workloads still get GeneralPurpose (no more ComputeOptimized auto-select)
1680        let req = WorkloadRequirements {
1681            total_cpu_at_desired: 20.0,
1682            total_memory_bytes_at_desired: 20 * GI,
1683            total_cpu_at_max: 20.0,
1684            total_memory_bytes_at_max: 20 * GI,
1685            max_cpu_per_container: 2.0,
1686            max_memory_per_container: 2 * GI,
1687            max_ephemeral_storage_bytes: 10 * GI,
1688            gpu: None,
1689            architecture: None,
1690            nested_virt: false,
1691        };
1692        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1693        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1694        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1695    }
1696
1697    #[test]
1698    fn test_select_storage_optimized_for_large_ephemeral() {
1699        let req = WorkloadRequirements {
1700            total_cpu_at_desired: 8.0,
1701            total_memory_bytes_at_desired: 32 * GI,
1702            total_cpu_at_max: 8.0,
1703            total_memory_bytes_at_max: 32 * GI,
1704            max_cpu_per_container: 2.0,
1705            max_memory_per_container: 8 * GI,
1706            max_ephemeral_storage_bytes: 500 * GI,
1707            gpu: None,
1708            architecture: Some(Architecture::X86_64),
1709            nested_virt: false,
1710        };
1711        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1712        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1713        assert_eq!(spec.family, InstanceFamily::StorageOptimized);
1714    }
1715
1716    #[test]
1717    fn test_select_gpu_instance() {
1718        let req = WorkloadRequirements {
1719            total_cpu_at_desired: 8.0,
1720            total_memory_bytes_at_desired: 32 * GI,
1721            total_cpu_at_max: 8.0,
1722            total_memory_bytes_at_max: 32 * GI,
1723            max_cpu_per_container: 4.0,
1724            max_memory_per_container: 16 * GI,
1725            max_ephemeral_storage_bytes: 10 * GI,
1726            gpu: Some(GpuSpec {
1727                gpu_type: "nvidia-a100".to_string(),
1728                count: 1,
1729            }),
1730            architecture: Some(Architecture::X86_64),
1731            nested_virt: false,
1732        };
1733        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1734        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1735        assert_eq!(spec.family, InstanceFamily::GpuCompute);
1736        assert!(spec.gpu.is_some());
1737    }
1738
1739    #[test]
1740    fn test_select_uses_each_cloud_image_target_architecture() {
1741        let req = WorkloadRequirements {
1742            total_cpu_at_desired: 4.0,
1743            total_memory_bytes_at_desired: 16 * GI,
1744            total_cpu_at_max: 4.0,
1745            total_memory_bytes_at_max: 16 * GI,
1746            max_cpu_per_container: 1.0,
1747            max_memory_per_container: 4 * GI,
1748            max_ephemeral_storage_bytes: 10 * GI,
1749            gpu: None,
1750            architecture: None,
1751            nested_virt: false,
1752        };
1753        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
1754            let sel = select_instance_type(platform, &req)
1755                .unwrap_or_else(|error| panic!("selection failed for {platform}: {error}"));
1756            let spec = find_instance_type(platform, sel.instance_type)
1757                .expect("selected machine should exist in the catalog");
1758            assert_eq!(
1759                Some(spec.architecture),
1760                default_architecture(platform),
1761                "machine architecture must match the image target for {platform}"
1762            );
1763        }
1764    }
1765
1766    #[test]
1767    fn test_machine_count_reasonable() {
1768        // Single container: 1 CPU, 2Gi, maxReplicas=20
1769        let req = WorkloadRequirements {
1770            total_cpu_at_desired: 20.0,
1771            total_memory_bytes_at_desired: 40 * GI,
1772            total_cpu_at_max: 20.0,
1773            total_memory_bytes_at_max: 40 * GI,
1774            max_cpu_per_container: 1.0,
1775            max_memory_per_container: 2 * GI,
1776            max_ephemeral_storage_bytes: 10 * GI,
1777            gpu: None,
1778            architecture: None,
1779            nested_virt: false,
1780        };
1781        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1782        assert!(sel.min_machines >= 1);
1783        assert!(sel.max_machines <= MAX_MACHINES_PER_CLUSTER);
1784        assert!(sel.max_machines >= sel.min_machines);
1785    }
1786
1787    #[test]
1788    fn test_instance_size_capped_at_8_vcpu() {
1789        // Even with very large containers, instance size is capped at 8 vCPUs
1790        let req = WorkloadRequirements {
1791            total_cpu_at_desired: 70.0,
1792            total_memory_bytes_at_desired: 140 * GI,
1793            total_cpu_at_max: 70.0,
1794            total_memory_bytes_at_max: 140 * GI,
1795            max_cpu_per_container: 2.0,
1796            max_memory_per_container: 4 * GI,
1797            max_ephemeral_storage_bytes: 10 * GI,
1798            gpu: None,
1799            architecture: None,
1800            nested_virt: false,
1801        };
1802        let sel = select_instance_type(Platform::Gcp, &req).unwrap();
1803        let spec = find_instance_type(Platform::Gcp, sel.instance_type).unwrap();
1804        assert!(
1805            spec.vcpu <= MAX_STANDARD_VCPU,
1806            "selected {} with {} vCPUs, expected <= {}",
1807            spec.name,
1808            spec.vcpu,
1809            MAX_STANDARD_VCPU
1810        );
1811        assert_eq!(spec.family, InstanceFamily::GeneralPurpose);
1812        // Should scale horizontally instead
1813        assert!(sel.max_machines > 1);
1814    }
1815
1816    #[test]
1817    fn test_larger_autoscaled_workload_gets_reasonable_instance() {
1818        // Simulates a larger autoscaled workload: 4 containers, each 2 CPU / 4 GiB
1819        // maxReplicas: 10, 10, 10, 5
1820        let req = WorkloadRequirements {
1821            total_cpu_at_desired: 70.0,
1822            total_memory_bytes_at_desired: 140 * GI,
1823            total_cpu_at_max: 70.0,              // 2*10 + 2*10 + 2*10 + 2*5
1824            total_memory_bytes_at_max: 140 * GI, // 4*10 + 4*10 + 4*10 + 4*5
1825            max_cpu_per_container: 2.0,
1826            max_memory_per_container: 4 * GI,
1827            max_ephemeral_storage_bytes: 20 * GI,
1828            gpu: None,
1829            architecture: None,
1830            nested_virt: false,
1831        };
1832        let sel = select_instance_type(Platform::Gcp, &req).unwrap();
1833        // Should pick n2-standard-8 (8 vCPU, 32 GiB) — NOT c3-standard-44
1834        assert_eq!(sel.instance_type, "n2-standard-8");
1835        assert!(sel.max_machines >= 2);
1836    }
1837
1838    /// When `nested_virt` is set on the workload, the selector must
1839    /// restrict to nested-virt-capable families. On AWS that means an m8i
1840    /// (or other 8th-gen Intel) entry, never a Graviton (`*7g`, `t4g`) or
1841    /// burstable. Without this filter the launch template gets created
1842    /// with `CpuOptions.NestedVirtualization=enabled` paired with an
1843    /// instance type AWS rejects at RunInstances.
1844    #[test]
1845    fn test_select_aws_picks_m8i_when_nested_virt_required() {
1846        let req = WorkloadRequirements {
1847            total_cpu_at_desired: 4.0,
1848            total_memory_bytes_at_desired: 8 * GI,
1849            total_cpu_at_max: 4.0,
1850            total_memory_bytes_at_max: 8 * GI,
1851            max_cpu_per_container: 4.0,
1852            max_memory_per_container: 8 * GI,
1853            max_ephemeral_storage_bytes: 10 * GI,
1854            gpu: None,
1855            architecture: Some(Architecture::X86_64),
1856            nested_virt: true,
1857        };
1858        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1859        assert!(
1860            sel.instance_type.starts_with("m8i.")
1861                || sel.instance_type.starts_with("c8i.")
1862                || sel.instance_type.starts_with("r8i."),
1863            "expected an m8i/c8i/r8i instance, got {}",
1864            sel.instance_type
1865        );
1866        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1867        assert!(spec.is_nested_virt_capable());
1868    }
1869
1870    #[test]
1871    fn test_select_gcp_picks_n2_when_nested_virt_required() {
1872        let req = WorkloadRequirements {
1873            total_cpu_at_desired: 4.0,
1874            total_memory_bytes_at_desired: 8 * GI,
1875            total_cpu_at_max: 4.0,
1876            total_memory_bytes_at_max: 8 * GI,
1877            max_cpu_per_container: 4.0,
1878            max_memory_per_container: 8 * GI,
1879            max_ephemeral_storage_bytes: 0,
1880            architecture: Some(Architecture::X86_64),
1881            gpu: None,
1882            nested_virt: true,
1883        };
1884
1885        let selection = select_instance_type(Platform::Gcp, &req).unwrap();
1886        assert_eq!(selection.instance_type, "n2-standard-8");
1887        assert!(find_instance_type(Platform::Gcp, selection.instance_type)
1888            .unwrap()
1889            .is_nested_virt_capable());
1890    }
1891
1892    #[test]
1893    fn test_select_azure_picks_dsv5_when_nested_virt_required() {
1894        let req = WorkloadRequirements {
1895            total_cpu_at_desired: 4.0,
1896            total_memory_bytes_at_desired: 8 * GI,
1897            total_cpu_at_max: 4.0,
1898            total_memory_bytes_at_max: 8 * GI,
1899            max_cpu_per_container: 4.0,
1900            max_memory_per_container: 8 * GI,
1901            max_ephemeral_storage_bytes: 0,
1902            architecture: Some(Architecture::X86_64),
1903            gpu: None,
1904            nested_virt: true,
1905        };
1906
1907        let selection = select_instance_type(Platform::Azure, &req).unwrap();
1908        assert_eq!(selection.instance_type, "Standard_D8s_v5");
1909        assert!(find_instance_type(Platform::Azure, selection.instance_type)
1910            .unwrap()
1911            .is_nested_virt_capable());
1912    }
1913
1914    #[test]
1915    fn test_select_aws_defaults_to_image_target_architecture() {
1916        let req = WorkloadRequirements {
1917            total_cpu_at_desired: 4.0,
1918            total_memory_bytes_at_desired: 8 * GI,
1919            total_cpu_at_max: 4.0,
1920            total_memory_bytes_at_max: 8 * GI,
1921            max_cpu_per_container: 4.0,
1922            max_memory_per_container: 8 * GI,
1923            max_ephemeral_storage_bytes: 10 * GI,
1924            gpu: None,
1925            architecture: None,
1926            nested_virt: false,
1927        };
1928        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1929        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1930        assert_eq!(spec.architecture, Architecture::Arm64);
1931    }
1932
1933    #[test]
1934    fn test_cloud_defaults_match_image_target_architectures() {
1935        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
1936            let target = BinaryTarget::defaults_for_platform(platform)
1937                .into_iter()
1938                .next()
1939                .expect("managed cloud should have a default image target");
1940            let image_architecture = match target.oci_arch() {
1941                "arm64" => Architecture::Arm64,
1942                "amd64" => Architecture::X86_64,
1943                architecture => {
1944                    panic!("unsupported managed-cloud image architecture {architecture}")
1945                }
1946            };
1947
1948            assert_eq!(default_architecture(platform), Some(image_architecture));
1949        }
1950    }
1951
1952    /// ARM remains available when the workload or capacity profile declares it.
1953    #[test]
1954    fn test_select_aws_uses_graviton_for_explicit_arm64() {
1955        let req = WorkloadRequirements {
1956            total_cpu_at_desired: 4.0,
1957            total_memory_bytes_at_desired: 8 * GI,
1958            total_cpu_at_max: 4.0,
1959            total_memory_bytes_at_max: 8 * GI,
1960            max_cpu_per_container: 4.0,
1961            max_memory_per_container: 8 * GI,
1962            max_ephemeral_storage_bytes: 10 * GI,
1963            gpu: None,
1964            architecture: Some(Architecture::Arm64),
1965            nested_virt: false,
1966        };
1967        let sel = select_instance_type(Platform::Aws, &req).unwrap();
1968        let spec = find_instance_type(Platform::Aws, sel.instance_type).unwrap();
1969        assert_eq!(spec.architecture, Architecture::Arm64);
1970    }
1971
1972    #[test]
1973    fn test_select_rejects_explicit_architecture_missing_from_cloud_catalog() {
1974        let req = WorkloadRequirements {
1975            total_cpu_at_desired: 1.0,
1976            total_memory_bytes_at_desired: 2 * GI,
1977            total_cpu_at_max: 1.0,
1978            total_memory_bytes_at_max: 2 * GI,
1979            max_cpu_per_container: 1.0,
1980            max_memory_per_container: 2 * GI,
1981            max_ephemeral_storage_bytes: 10 * GI,
1982            gpu: None,
1983            architecture: Some(Architecture::Arm64),
1984            nested_virt: false,
1985        };
1986
1987        let error = select_instance_type(Platform::Gcp, &req)
1988            .expect_err("GCP catalog has no ARM64 machine");
1989
1990        assert!(error.contains("architecture Arm64 is unavailable"));
1991    }
1992
1993    #[test]
1994    fn test_profile_has_required_fields() {
1995        let req = WorkloadRequirements {
1996            total_cpu_at_desired: 4.0,
1997            total_memory_bytes_at_desired: 16 * GI,
1998            total_cpu_at_max: 4.0,
1999            total_memory_bytes_at_max: 16 * GI,
2000            max_cpu_per_container: 1.0,
2001            max_memory_per_container: 4 * GI,
2002            max_ephemeral_storage_bytes: 10 * GI,
2003            gpu: None,
2004            architecture: None,
2005            nested_virt: false,
2006        };
2007        let sel = select_instance_type(Platform::Aws, &req).unwrap();
2008        assert!(!sel.profile.cpu.is_empty());
2009        assert!(sel.profile.memory_bytes > 0);
2010        assert!(sel.profile.ephemeral_storage_bytes > 0);
2011    }
2012
2013    #[test]
2014    fn test_error_for_unsupported_gpu_type() {
2015        let req = WorkloadRequirements {
2016            total_cpu_at_desired: 8.0,
2017            total_memory_bytes_at_desired: 32 * GI,
2018            total_cpu_at_max: 8.0,
2019            total_memory_bytes_at_max: 32 * GI,
2020            max_cpu_per_container: 4.0,
2021            max_memory_per_container: 16 * GI,
2022            max_ephemeral_storage_bytes: 10 * GI,
2023            gpu: Some(GpuSpec {
2024                gpu_type: "amd-mi300".to_string(),
2025                count: 1,
2026            }),
2027            architecture: None,
2028            nested_virt: false,
2029        };
2030        let result = select_instance_type(Platform::Aws, &req);
2031        assert!(result.is_err());
2032    }
2033
2034    #[test]
2035    fn test_catalog_instance_types_sorted_by_vcpu_within_family() {
2036        // Verify that within each (platform, family) group, vcpu is non-decreasing.
2037        // This ensures our "min_by_key(vcpu)" logic works correctly.
2038        for platform in [Platform::Aws, Platform::Gcp, Platform::Azure] {
2039            let entries = catalog_for_platform(platform);
2040            let mut by_family: std::collections::HashMap<_, Vec<_>> =
2041                std::collections::HashMap::new();
2042            for entry in entries {
2043                by_family
2044                    .entry(format!("{:?}", entry.family))
2045                    .or_default()
2046                    .push(entry);
2047            }
2048            for (family, instances) in &by_family {
2049                for window in instances.windows(2) {
2050                    assert!(
2051                        window[0].vcpu <= window[1].vcpu,
2052                        "catalog not sorted by vcpu for {platform}/{family}: {} ({}) > {} ({})",
2053                        window[0].name,
2054                        window[0].vcpu,
2055                        window[1].name,
2056                        window[1].vcpu
2057                    );
2058                }
2059            }
2060        }
2061    }
2062}