Skip to main content

alien_core/resources/
compute_cluster.rs

1//! ComputeCluster resource for long-running container workloads.
2//!
3//! A ComputeCluster represents the setup-owned compute boundary for containers.
4//! Setup provisions:
5//! - Auto Scaling Groups (AWS), Managed Instance Groups (GCP), or VM Scale Sets (Azure)
6//! - IAM roles/service accounts for machine authentication
7//! - Security groups/firewall rules
8//! - Launch templates/instance configurations
9
10use crate::error::{ErrorData, Result};
11use crate::instance_catalog::Architecture;
12use crate::resource::{ResourceDefinition, ResourceOutputsDefinition, ResourceRef};
13use crate::ResourceType;
14use alien_error::AlienError;
15use bon::Builder;
16use serde::{Deserialize, Serialize};
17use std::any::Any;
18use std::collections::BTreeMap;
19use std::fmt::Debug;
20
21/// GPU specification for a capacity group.
22#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
23#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
24#[serde(rename_all = "camelCase")]
25pub struct GpuSpec {
26    /// GPU type identifier (e.g., "nvidia-a100", "nvidia-t4")
27    #[serde(rename = "type")]
28    pub gpu_type: String,
29    /// Number of GPUs per machine
30    pub count: u32,
31}
32
33/// Machine resource profile for a capacity group.
34///
35/// Represents the hardware specifications for machines in a capacity group.
36/// These are hardware totals (what the instance type advertises), not allocatable
37/// capacity. The managed container scheduler internally subtracts system reserves for planning.
38#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
39#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
40#[serde(rename_all = "camelCase")]
41pub struct MachineProfile {
42    /// CPU cores per machine (hardware total) - stored as string to preserve precision
43    /// (e.g., "8.0", "4.5")
44    pub cpu: String,
45    /// Memory in bytes (hardware total)
46    pub memory_bytes: u64,
47    /// Ephemeral storage in bytes (hardware total)
48    pub ephemeral_storage_bytes: u64,
49    /// CPU architecture required or provided by this machine profile.
50    #[serde(skip_serializing_if = "Option::is_none")]
51    pub architecture: Option<Architecture>,
52    /// GPU specification (optional)
53    #[serde(skip_serializing_if = "Option::is_none")]
54    pub gpu: Option<GpuSpec>,
55}
56
57/// Allowed range and default for a count selected by the installer.
58#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
59#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
60#[serde(rename_all = "camelCase")]
61pub struct ComputeChoiceRange {
62    /// Lowest allowed value.
63    pub min: u32,
64    /// Highest allowed value.
65    pub max: u32,
66    /// Default value recommended when no installer override is supplied.
67    pub default: u32,
68}
69
70impl ComputeChoiceRange {
71    /// Returns whether a selected value is inside the allowed range.
72    pub fn contains(&self, value: u32) -> bool {
73        self.min <= value && value <= self.max
74    }
75}
76
77/// Source-declared scale policy for a capacity group.
78#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
79#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
80#[serde(rename_all = "camelCase", tag = "type")]
81pub enum CapacityGroupScalePolicy {
82    /// A fixed-size pool where the installer can choose the fixed machine count.
83    Fixed {
84        /// Allowed fixed machine count range.
85        machines: ComputeChoiceRange,
86    },
87    /// An autoscaling pool with separately bounded min and max counts.
88    Autoscale {
89        /// Allowed minimum machine count range.
90        min: ComputeChoiceRange,
91        /// Allowed maximum machine count range.
92        max: ComputeChoiceRange,
93    },
94}
95
96impl CapacityGroupScalePolicy {
97    /// Derive the legacy policy represented by selected min/max values.
98    pub fn from_selected_bounds(min_size: u32, max_size: u32) -> Self {
99        if min_size == max_size {
100            Self::Fixed {
101                machines: ComputeChoiceRange {
102                    min: min_size,
103                    max: max_size,
104                    default: min_size,
105                },
106            }
107        } else {
108            Self::Autoscale {
109                min: ComputeChoiceRange {
110                    min: min_size,
111                    max: min_size,
112                    default: min_size,
113                },
114                max: ComputeChoiceRange {
115                    min: max_size,
116                    max: max_size,
117                    default: max_size,
118                },
119            }
120        }
121    }
122
123    /// Default selected min bound.
124    pub fn default_min_size(&self) -> u32 {
125        match self {
126            Self::Fixed { machines } => machines.default,
127            Self::Autoscale { min, .. } => min.default,
128        }
129    }
130
131    /// Default selected max bound.
132    pub fn default_max_size(&self) -> u32 {
133        match self {
134            Self::Fixed { machines } => machines.default,
135            Self::Autoscale { max, .. } => max.default,
136        }
137    }
138}
139
140/// Capacity group definition.
141///
142/// A capacity group represents machines with identical hardware profiles.
143/// Each group becomes a separate Auto Scaling Group (AWS), Managed Instance Group (GCP),
144/// or VM Scale Set (Azure).
145#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
146#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
147#[serde(rename_all = "camelCase")]
148pub struct CapacityGroup {
149    /// Unique identifier for this capacity group (must be lowercase alphanumeric with hyphens)
150    pub group_id: String,
151    /// Provider machine selected at deployment time.
152    /// `alien.ts` should declare portable requirements; preflight materialization
153    /// fills this field from `StackSettings.compute`.
154    #[serde(skip_serializing_if = "Option::is_none")]
155    pub instance_type: Option<String>,
156    /// Machine resource profile (auto-derived from instance_type if not specified)
157    #[serde(skip_serializing_if = "Option::is_none")]
158    pub profile: Option<MachineProfile>,
159    /// Minimum number of machines (can be 0 for scale-to-zero)
160    pub min_size: u32,
161    /// Maximum number of machines (must be ≤ 10)
162    pub max_size: u32,
163    /// Source-declared scale policy and installer-editable bounds.
164    ///
165    /// Older stacks only have `minSize` and `maxSize`; planners derive an exact
166    /// fixed/autoscale policy from those selected bounds when this field is absent.
167    #[serde(skip_serializing_if = "Option::is_none")]
168    pub scale_policy: Option<CapacityGroupScalePolicy>,
169    /// Require instance types that expose nested virtualization (VT-x/EPT)
170    /// to guest VMs. This is needed by workloads that boot nested VMs inside
171    /// containers.
172    /// When true, the controller's instance-type selector is constrained
173    /// to a vetted nested-virt-capable allowlist.
174    #[serde(skip_serializing_if = "Option::is_none")]
175    pub nested_virtualization: Option<bool>,
176}
177
178/// ComputeCluster resource for running long-running container workloads.
179///
180/// A ComputeCluster provides the setup-owned machine boundary for containers.
181/// Alien may manage the worker fleet inside that boundary when setup grants
182/// `compute-cluster/management`.
183///
184/// ## Architecture
185///
186/// - **Setup** creates cloud resources: ASGs/MIGs/VMSSs, IAM roles, security groups
187/// - **Alien** manages allowed fleet operations: machine count and runtime
188///   machine image rollout
189/// - A node agent runs on each machine from the selected runtime image channel
190///
191/// ## Example
192///
193/// ```rust
194/// use alien_core::{CapacityGroup, ComputeCluster, MachineProfile};
195///
196/// let cluster = ComputeCluster::new("compute".to_string())
197///     .capacity_group(CapacityGroup {
198///         group_id: "general".to_string(),
199///         instance_type: None,
200///         profile: Some(MachineProfile {
201///             cpu: "4.0".to_string(),
202///             memory_bytes: 16 * 1024 * 1024 * 1024,
203///             ephemeral_storage_bytes: 20 * 1024 * 1024 * 1024,
204///             architecture: None,
205///             gpu: None,
206///         }),
207///         min_size: 1,
208///         max_size: 5,
209///         scale_policy: None,
210///         nested_virtualization: None,
211///     })
212///     .build();
213/// ```
214#[derive(Debug, Clone, PartialEq, Serialize, Deserialize, Builder)]
215#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
216#[serde(rename_all = "camelCase", deny_unknown_fields)]
217#[builder(start_fn = new)]
218pub struct ComputeCluster {
219    /// Unique identifier for the container cluster.
220    /// Must contain only alphanumeric characters, hyphens, and underscores.
221    #[builder(start_fn)]
222    pub id: String,
223
224    /// Capacity groups defining the machine pools for this cluster.
225    /// Each group becomes a separate ASG/MIG/VMSS.
226    #[builder(field)]
227    pub capacity_groups: Vec<CapacityGroup>,
228
229    /// Pool reserved for containers created after a deployment is installed.
230    /// If absent, the runtime uses the `general` pool when it exists.
231    #[serde(default, skip_serializing_if = "Option::is_none")]
232    pub dynamic_container_pool: Option<String>,
233
234    /// Concrete provider failure domains selected during setup, keyed by capacity group.
235    /// Empty preserves the existing aggregate layout when no spread policy is configured.
236    #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
237    #[builder(default)]
238    pub selected_failure_domains: BTreeMap<String, Vec<String>>,
239
240    /// Requested failure-domain spread keyed by capacity group.
241    /// Empty preserves the existing aggregate layout.
242    #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
243    #[builder(default)]
244    pub failure_domain_spread: BTreeMap<String, u8>,
245
246    /// Container CIDR block for internal container networking.
247    /// Auto-generated as "10.244.0.0/16" if not specified.
248    /// Each machine gets a /24 subnet from this range.
249    #[serde(skip_serializing_if = "Option::is_none")]
250    pub container_cidr: Option<String>,
251}
252
253impl ComputeCluster {
254    /// The resource type identifier for ComputeCluster
255    pub const RESOURCE_TYPE: ResourceType = ResourceType::from_static("compute-cluster");
256
257    /// Returns the cluster's unique identifier.
258    pub fn id(&self) -> &str {
259        &self.id
260    }
261
262    /// Returns the container CIDR, defaulting to "10.244.0.0/16" if not specified.
263    pub fn container_cidr(&self) -> &str {
264        self.container_cidr.as_deref().unwrap_or("10.244.0.0/16")
265    }
266}
267
268impl<S: compute_cluster_builder::State> ComputeClusterBuilder<S> {
269    /// Adds a capacity group to the cluster.
270    pub fn capacity_group(mut self, group: CapacityGroup) -> Self {
271        self.capacity_groups.push(group);
272        self
273    }
274}
275
276/// Status of a single capacity group within a ComputeCluster.
277#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
278#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
279#[serde(rename_all = "camelCase")]
280pub struct CapacityGroupStatus {
281    /// Capacity group ID
282    pub group_id: String,
283    /// Current number of machines
284    pub current_machines: u32,
285    /// Desired number of machines (from the managed container capacity plan)
286    pub desired_machines: u32,
287    /// Instance type being used
288    pub instance_type: String,
289}
290
291/// Outputs generated by a successfully provisioned ComputeCluster.
292#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
293#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
294#[serde(rename_all = "camelCase")]
295pub struct ComputeClusterOutputs {
296    /// Managed container cluster ID (workspace/project/deployment/resourceid format)
297    pub cluster_id: String,
298    /// Whether the managed container cluster is ready
299    pub horizon_ready: bool,
300    /// Status of each capacity group
301    pub capacity_group_statuses: Vec<CapacityGroupStatus>,
302    /// Total number of machines across all capacity groups
303    pub total_machines: u32,
304}
305
306impl ResourceOutputsDefinition for ComputeClusterOutputs {
307    fn get_resource_type(&self) -> ResourceType {
308        ComputeCluster::RESOURCE_TYPE.clone()
309    }
310
311    fn as_any(&self) -> &dyn Any {
312        self
313    }
314
315    fn box_clone(&self) -> Box<dyn ResourceOutputsDefinition> {
316        Box::new(self.clone())
317    }
318
319    fn outputs_eq(&self, other: &dyn ResourceOutputsDefinition) -> bool {
320        other.as_any().downcast_ref::<ComputeClusterOutputs>() == Some(self)
321    }
322
323    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
324        serde_json::to_value(self)
325    }
326}
327
328impl ResourceDefinition for ComputeCluster {
329    fn get_resource_type(&self) -> ResourceType {
330        Self::RESOURCE_TYPE
331    }
332
333    fn id(&self) -> &str {
334        &self.id
335    }
336
337    fn get_dependencies(&self) -> Vec<ResourceRef> {
338        // ComputeCluster has no static dependencies.
339        // Network dependency is platform-specific:
340        // - AWS/GCP/Azure: Added by ComputeClusterMutation
341        // - Local/Kubernetes: Not needed (Docker/K8s handles networking)
342        // Platform controllers use require_dependency() at runtime to access Network state.
343        Vec::new()
344    }
345
346    fn validate_update(&self, new_config: &dyn ResourceDefinition) -> Result<()> {
347        let new_cluster = new_config
348            .as_any()
349            .downcast_ref::<ComputeCluster>()
350            .ok_or_else(|| {
351                AlienError::new(ErrorData::UnexpectedResourceType {
352                    resource_id: self.id.clone(),
353                    expected: Self::RESOURCE_TYPE,
354                    actual: new_config.get_resource_type(),
355                })
356            })?;
357
358        if self.id != new_cluster.id {
359            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
360                resource_id: self.id.clone(),
361                reason: "the 'id' field is immutable".to_string(),
362            }));
363        }
364
365        // Container CIDR is immutable once set
366        if self.container_cidr.is_some()
367            && new_cluster.container_cidr.is_some()
368            && self.container_cidr != new_cluster.container_cidr
369        {
370            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
371                resource_id: self.id.clone(),
372                reason: "the 'containerCidr' field is immutable once set".to_string(),
373            }));
374        }
375
376        // Validate capacity groups
377        for new_group in &new_cluster.capacity_groups {
378            if let Some(existing_group) = self
379                .capacity_groups
380                .iter()
381                .find(|g| g.group_id == new_group.group_id)
382            {
383                // Instance type is immutable for existing groups
384                if existing_group.instance_type.is_some()
385                    && new_group.instance_type.is_some()
386                    && existing_group.instance_type != new_group.instance_type
387                {
388                    return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
389                        resource_id: self.id.clone(),
390                        reason: format!(
391                            "instance type for capacity group '{}' is immutable",
392                            new_group.group_id
393                        ),
394                    }));
395                }
396            }
397        }
398
399        Ok(())
400    }
401
402    fn as_any(&self) -> &dyn Any {
403        self
404    }
405
406    fn as_any_mut(&mut self) -> &mut dyn Any {
407        self
408    }
409
410    fn box_clone(&self) -> Box<dyn ResourceDefinition> {
411        Box::new(self.clone())
412    }
413
414    fn resource_eq(&self, other: &dyn ResourceDefinition) -> bool {
415        other.as_any().downcast_ref::<ComputeCluster>() == Some(self)
416    }
417
418    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
419        serde_json::to_value(self)
420    }
421}
422
423#[cfg(test)]
424mod tests {
425    use super::*;
426
427    #[test]
428    fn test_compute_cluster_creation() {
429        let cluster = ComputeCluster::new("compute".to_string())
430            .capacity_group(CapacityGroup {
431                group_id: "general".to_string(),
432                instance_type: Some("m7g.xlarge".to_string()),
433                profile: None,
434                min_size: 1,
435                max_size: 5,
436                scale_policy: None,
437                nested_virtualization: None,
438            })
439            .build();
440
441        assert_eq!(cluster.id(), "compute");
442        assert_eq!(cluster.capacity_groups.len(), 1);
443        assert_eq!(cluster.capacity_groups[0].group_id, "general");
444        assert_eq!(cluster.container_cidr(), "10.244.0.0/16");
445    }
446
447    #[test]
448    fn test_compute_cluster_multiple_capacity_groups() {
449        let cluster = ComputeCluster::new("multi-pool".to_string())
450            .capacity_group(CapacityGroup {
451                group_id: "general".to_string(),
452                instance_type: Some("m7g.xlarge".to_string()),
453                profile: None,
454                min_size: 1,
455                max_size: 3,
456                scale_policy: None,
457                nested_virtualization: None,
458            })
459            .capacity_group(CapacityGroup {
460                group_id: "gpu".to_string(),
461                instance_type: Some("g5.xlarge".to_string()),
462                profile: Some(MachineProfile {
463                    cpu: "4.0".to_string(),
464                    memory_bytes: 17179869184,             // 16 GiB
465                    ephemeral_storage_bytes: 214748364800, // 200 GiB
466                    architecture: None,
467                    gpu: Some(GpuSpec {
468                        gpu_type: "nvidia-a10g".to_string(),
469                        count: 1,
470                    }),
471                }),
472                min_size: 0,
473                max_size: 2,
474                scale_policy: None,
475                nested_virtualization: None,
476            })
477            .build();
478
479        assert_eq!(cluster.capacity_groups.len(), 2);
480        assert_eq!(cluster.capacity_groups[0].group_id, "general");
481        assert_eq!(cluster.capacity_groups[1].group_id, "gpu");
482        assert!(cluster.capacity_groups[1]
483            .profile
484            .as_ref()
485            .unwrap()
486            .gpu
487            .is_some());
488    }
489
490    #[test]
491    fn test_compute_cluster_custom_cidr() {
492        let cluster = ComputeCluster::new("custom-net".to_string())
493            .container_cidr("172.30.0.0/16".to_string())
494            .capacity_group(CapacityGroup {
495                group_id: "general".to_string(),
496                instance_type: None,
497                profile: None,
498                min_size: 1,
499                max_size: 5,
500                scale_policy: None,
501                nested_virtualization: None,
502            })
503            .build();
504
505        assert_eq!(cluster.container_cidr(), "172.30.0.0/16");
506    }
507
508    #[test]
509    fn test_compute_cluster_validate_update_immutable_id() {
510        let cluster1 = ComputeCluster::new("cluster-1".to_string())
511            .capacity_group(CapacityGroup {
512                group_id: "general".to_string(),
513                instance_type: None,
514                profile: None,
515                min_size: 1,
516                max_size: 5,
517                scale_policy: None,
518                nested_virtualization: None,
519            })
520            .build();
521
522        let cluster2 = ComputeCluster::new("cluster-2".to_string())
523            .capacity_group(CapacityGroup {
524                group_id: "general".to_string(),
525                instance_type: None,
526                profile: None,
527                min_size: 1,
528                max_size: 5,
529                scale_policy: None,
530                nested_virtualization: None,
531            })
532            .build();
533
534        let result = cluster1.validate_update(&cluster2);
535        assert!(result.is_err());
536    }
537
538    #[test]
539    fn test_compute_cluster_validate_update_scale_change() {
540        let cluster1 = ComputeCluster::new("compute".to_string())
541            .capacity_group(CapacityGroup {
542                group_id: "general".to_string(),
543                instance_type: Some("m7g.xlarge".to_string()),
544                profile: None,
545                min_size: 1,
546                max_size: 5,
547                scale_policy: None,
548                nested_virtualization: None,
549            })
550            .build();
551
552        let cluster2 = ComputeCluster::new("compute".to_string())
553            .capacity_group(CapacityGroup {
554                group_id: "general".to_string(),
555                instance_type: Some("m7g.xlarge".to_string()),
556                profile: None,
557                min_size: 2,
558                max_size: 10,
559                scale_policy: None,
560                nested_virtualization: None,
561            })
562            .build();
563
564        // Scale changes should be allowed
565        let result = cluster1.validate_update(&cluster2);
566        assert!(result.is_ok());
567    }
568
569    #[test]
570    fn test_compute_cluster_serialization() {
571        let cluster = ComputeCluster::new("test-cluster".to_string())
572            .capacity_group(CapacityGroup {
573                group_id: "general".to_string(),
574                instance_type: Some("m7g.xlarge".to_string()),
575                profile: None,
576                min_size: 1,
577                max_size: 5,
578                scale_policy: None,
579                nested_virtualization: None,
580            })
581            .build();
582
583        let json = serde_json::to_string(&cluster).unwrap();
584        let deserialized: ComputeCluster = serde_json::from_str(&json).unwrap();
585        assert_eq!(cluster, deserialized);
586    }
587}