Skip to main content

alien_core/resources/
sandbox.rs

1//! Sandbox resource for running untrusted code in an isolated environment.
2//!
3//! The declaration provisions a durable parent, and the application creates and destroys
4//! individual sandboxes through its binding at runtime.
5//!
6//! The capability set differs per platform and is published rather than assumed. Calling an
7//! unsupported capability is a typed error naming both the platform and the capability, so a
8//! portable application can branch on `SandboxCapabilities` before it calls.
9
10use crate::error::{ErrorData, Result};
11use crate::resource::{ResourceDefinition, ResourceOutputsDefinition, ResourceRef, ResourceType};
12use crate::resources::ToolchainConfig;
13use crate::Platform;
14use alien_error::AlienError;
15use bon::Builder;
16use serde::{Deserialize, Serialize};
17use std::any::Any;
18use std::fmt::Debug;
19
20/// Specifies where the sandbox's root filesystem comes from.
21#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
22#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
23#[serde(rename_all = "camelCase", tag = "type")]
24pub enum SandboxCode {
25    /// A prebuilt container image used as the sandbox root filesystem.
26    #[serde(rename_all = "camelCase")]
27    Image {
28        /// Image reference (e.g. `ubuntu:24.04`, `ghcr.io/myorg/sandbox:latest`).
29        ///
30        /// Two backends narrow it in opposite directions: AWS wants an `s3://` bundle, Azure a
31        /// bare catalog name such as `ubuntu`. Each refuses the other's shape while planning.
32        image: String,
33    },
34    /// Source built into a sandbox image at deploy time.
35    #[serde(rename_all = "camelCase")]
36    Source {
37        /// The source directory to build from
38        src: String,
39        /// Toolchain configuration with type-safe options
40        toolchain: ToolchainConfig,
41    },
42}
43
44/// Hard ceilings enforced on a sandbox.
45///
46/// These are limits, not scheduling requests. Untrusted code does not respect a hint, so every
47/// field is enforced by the platform and a platform that cannot enforce one is rejected at plan
48/// time rather than silently ignoring it.
49#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
50#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
51#[serde(rename_all = "camelCase", deny_unknown_fields)]
52pub struct SandboxLimits {
53    /// CPU ceiling in cores or millicores (e.g. `"1"`, `"500m"`)
54    pub cpu: String,
55    /// Memory ceiling (e.g. `"2Gi"`, `"512Mi"`)
56    pub memory: String,
57    /// Disk ceiling (e.g. `"20Gi"`)
58    pub disk: String,
59    /// Maximum number of processes, which bounds fork bombs.
60    ///
61    /// Optional because only a container runtime has the primitive: Kubernetes sets a pid ceiling
62    /// per node, not per pod, and neither AWS MicroVMs nor Azure sandboxes expose one. Declaring
63    /// it on a platform that cannot apply it is refused at plan time.
64    #[serde(default, skip_serializing_if = "Option::is_none")]
65    pub max_processes: Option<u32>,
66}
67
68/// One of the five sizes a Lambda MicroVM can be built at.
69///
70/// AWS has no ceiling knob: `minimumMemoryInMiB` sets a *baseline* and a running MicroVM bursts
71/// vertically to four times it with no way to opt out. A declared ceiling is therefore honoured by
72/// picking the tier whose **peak** stays inside it, not the tier whose baseline matches it.
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74pub struct MicrovmTier {
75    /// What `minimumMemoryInMiB` is set to.
76    pub baseline_memory_mib: i64,
77    /// The most memory the MicroVM can reach, in MiB.
78    pub peak_memory_mib: i64,
79    /// The most vCPU the MicroVM can reach.
80    pub peak_vcpu: u32,
81    /// The most disk the MicroVM can use, in MiB.
82    pub max_disk_mib: i64,
83}
84
85/// The published sizes, smallest first. Baseline memory to vCPU is 2 GB per vCPU, peak is four
86/// times baseline, and disk is fixed per tier rather than independently selectable.
87/// Longest life AWS will run a MicroVM for, from `RunMicrovm`'s `maximumDurationInSeconds`.
88const AWS_MAX_LIFETIME_SECONDS: u32 = 28_800;
89
90/// Azure's sandbox sizing rule, quoted from the data plane's own refusal of an oversized request:
91/// *CPU must be n×250m for n=1..64 (0.25–16 cores); Memory ≤ cores × 2Gi; Disk ≤ cores × 20Gi*.
92const AZURE_CPU_STEP_MILLICORES: i64 = 250;
93const AZURE_MAX_CPU_MILLICORES: i64 = 16_000;
94const AZURE_MEMORY_MIB_PER_CORE: i64 = 2 * 1024;
95const AZURE_DISK_MIB_PER_CORE: i64 = 20 * 1024;
96
97const MICROVM_TIERS: &[MicrovmTier] = &[
98    MicrovmTier {
99        baseline_memory_mib: 512,
100        peak_memory_mib: 2048,
101        peak_vcpu: 1,
102        max_disk_mib: 8192,
103    },
104    MicrovmTier {
105        baseline_memory_mib: 1024,
106        peak_memory_mib: 4096,
107        peak_vcpu: 2,
108        max_disk_mib: 8192,
109    },
110    MicrovmTier {
111        baseline_memory_mib: 2048,
112        peak_memory_mib: 8192,
113        peak_vcpu: 4,
114        max_disk_mib: 8192,
115    },
116    MicrovmTier {
117        baseline_memory_mib: 4096,
118        peak_memory_mib: 16384,
119        peak_vcpu: 8,
120        max_disk_mib: 16384,
121    },
122    MicrovmTier {
123        baseline_memory_mib: 8192,
124        peak_memory_mib: 32768,
125        peak_vcpu: 16,
126        max_disk_mib: 32768,
127    },
128];
129
130/// Outbound network policy for a sandbox.
131#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
132#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
133#[serde(rename_all = "camelCase", tag = "mode")]
134pub enum SandboxEgress {
135    /// No outbound network access.
136    ///
137    /// Routed traffic only. Link-local is not outbound and no backend's egress control reaches
138    /// it, so this is not a boundary against instance metadata.
139    Deny,
140    /// Unrestricted outbound access to the public internet, and none to private ranges or the
141    /// deployment's own network.
142    ///
143    /// Link-local carries the same exception as `Deny`. AWS and Kubernetes deliver both halves.
144    /// Azure and GCP deliver the first only: one matches host patterns and the other is a single
145    /// switch, so neither can name an address range to exclude.
146    Allow,
147    /// Outbound access only to the listed hostnames.
148    ///
149    /// Azure alone expresses it: its egress proxy matches on host pattern. The others filter by
150    /// CIDR or carry a single switch, and both would approximate the list rather than keep it.
151    #[serde(rename_all = "camelCase")]
152    AllowDomains {
153        /// Hostnames the sandbox may reach
154        domains: Vec<String>,
155    },
156}
157
158impl SandboxEgress {
159    /// The single outbound switch for a backend that has no host matcher, or `None` for a mode a
160    /// boolean cannot carry.
161    ///
162    /// `AllowDomains` needs a host list, so it maps to nothing and each caller refuses it in its
163    /// own error naming the sandbox. One source for what a mode means, so a template and a sandbox
164    /// cannot disagree on it.
165    pub fn internet_access_switch(&self) -> Option<bool> {
166        match self {
167            SandboxEgress::Allow => Some(true),
168            SandboxEgress::Deny => Some(false),
169            SandboxEgress::AllowDomains { .. } => None,
170        }
171    }
172}
173
174/// How long a sandbox may live and when it is paused.
175///
176/// Declaration-time ceilings, not per-request values: every sandbox created through this
177/// declaration's binding is held to them, whatever a caller asks for at runtime.
178#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
179#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
180#[serde(rename_all = "camelCase", deny_unknown_fields)]
181pub struct SandboxLifecyclePolicy {
182    /// Wall-clock ceiling on a single sandbox, after which the platform terminates it.
183    ///
184    /// Optional because not every backend has the primitive: Kubernetes has
185    /// `activeDeadlineSeconds` and AWS `maximumDurationInSeconds`, while neither Azure nor Local
186    /// expose one, so declaring a ceiling there is refused at plan time rather than accepted and
187    /// never applied. AWS caps it at 8 hours.
188    #[serde(default, skip_serializing_if = "Option::is_none")]
189    pub max_lifetime_seconds: Option<u32>,
190    /// Idle period after which the sandbox is paused, where the platform supports it
191    #[serde(skip_serializing_if = "Option::is_none")]
192    pub idle_pause_seconds: Option<u32>,
193}
194
195/// What a platform's sandbox backend can actually do.
196///
197/// Published so portable code can branch before calling rather than discovering a gap through
198/// an error. Every field here corresponds to a capability that at least one platform lacks;
199/// create, exec and terminate are the guaranteed floor and are therefore not listed.
200#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
201#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
202#[serde(rename_all = "camelCase", deny_unknown_fields)]
203pub struct SandboxCapabilities {
204    /// Files can be moved in and out of a sandbox
205    pub files: bool,
206    /// A later call can reach a sandbox created by an earlier one
207    pub reconnect: bool,
208    /// A command can be started, polled and cancelled across separate calls, so it outlives the
209    /// one that started it. False where nothing inside the sandbox owns the process in between.
210    pub jobs: bool,
211    /// An authenticated, port-scoped capability to reach a service inside the sandbox
212    pub preview: bool,
213    /// Sandbox state can be paused and resumed
214    pub pause_resume: bool,
215    /// A sandbox's full state can be captured and used to create another
216    pub snapshot: bool,
217    /// Egress can be restricted to a hostname allowlist
218    pub domain_egress_rules: bool,
219    /// Whether a declared `deny` is actually enforced, rather than accepted and dropped
220    pub egress_deny: bool,
221    /// The platform enforces the declared cpu, memory and disk ceilings
222    pub enforced_limits: bool,
223    /// The platform can cap how many processes a sandbox runs
224    pub process_limit: bool,
225    /// The platform terminates a sandbox at a declared wall-clock deadline
226    pub sandbox_lifetime: bool,
227    /// A command runs in its own PID namespace and cannot see or signal the agent's processes.
228    ///
229    /// Only where an agent runs as root. Creating the namespace needs `CAP_SYS_ADMIN`, and the
230    /// Kubernetes sandbox pod drops every capability — which is also what denies `ptrace` by
231    /// construction, so granting it there would remove a lock to add one.
232    pub supervisor_pid_namespace: bool,
233    /// The process supervising a command is a different identity from the command.
234    ///
235    /// False where a command runs as the agent's own user: it can then read the supervisor's
236    /// environment and signal it. Separate from `supervisorPidNamespace`, which is about
237    /// visibility rather than identity — a backend can have one without the other.
238    pub supervisor_isolation: bool,
239}
240
241impl SandboxCapabilities {
242    /// Returns what the given platform's sandbox backend supports.
243    ///
244    /// Errors for platforms with no sandbox backend, rather than returning an all-false set —
245    /// "every capability is missing" and "this platform has no sandboxes" are different
246    /// conditions and an application should not have to tell them apart by inspection.
247    pub fn for_platform(platform: Platform) -> Result<Self> {
248        match platform {
249            Platform::Aws => Ok(Self {
250                files: true,
251                reconnect: true,
252                jobs: true,
253                preview: true,
254                pause_resume: true,
255                snapshot: false,
256                domain_egress_rules: false,
257                egress_deny: true,
258                enforced_limits: true,
259                // Nothing in the API bounds process count.
260                process_limit: false,
261                // `maximumDurationInSeconds` on `RunMicrovm`, which Lambda enforces by
262                // terminating the MicroVM. Capped at 8 hours by the service.
263                sandbox_lifetime: true,
264                // Measured, not assumed: the agent inside a Lambda MicroVM runs as uid 0 with
265                // `CapEff: 00000000a80425fb`, the standard container default set, which excludes
266                // `CAP_SYS_ADMIN`. It can drop privilege (`CAP_SETUID`/`CAP_SETGID` are held) and
267                // it cannot create a namespace. No backend offers this today.
268                supervisor_pid_namespace: false,
269                // The agent runs as uid 0 and `setuid`s the command to uid 60000, so the command
270                // runs under a different identity than the process supervising it.
271                supervisor_isolation: true,
272            }),
273            Platform::Azure => Ok(Self::azure()),
274            Platform::Gcp => Ok(Self::gcp_agent_platform()),
275            // Preview needs a gateway that validates a sandbox-and-port capability, which this
276            // backend has none of.
277            Platform::Kubernetes => Ok(Self {
278                files: true,
279                reconnect: true,
280                jobs: true,
281                preview: false,
282                pause_resume: false,
283                snapshot: false,
284                domain_egress_rules: false,
285                egress_deny: true,
286                enforced_limits: true,
287                // A pid ceiling is a kubelet setting per node, not a pod field.
288                process_limit: false,
289                // `activeDeadlineSeconds` on the pod, which the kubelet enforces.
290                sandbox_lifetime: true,
291                // The pod drops every capability, including the `CAP_SYS_ADMIN` the agent would
292                // need to unshare. That is also what denies `ptrace`, so this stays false rather
293                // than the pod being weakened to make it true.
294                supervisor_pid_namespace: false,
295                // The pod pins one uid (`run_as_user: 65534` on both pod and container) with
296                // `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no
297                // process can setuid to split the command off from a supervisor. No uid split is
298                // possible, so none exists.
299                supervisor_isolation: false,
300            }),
301            Platform::Local => Ok(Self {
302                files: true,
303                reconnect: true,
304                // Nothing runs inside the sandbox: the manager drives Docker from outside it.
305                jobs: false,
306                preview: true,
307                pause_resume: false,
308                snapshot: false,
309                domain_egress_rules: false,
310                egress_deny: true,
311                enforced_limits: true,
312                // Docker's `--pids-limit`.
313                process_limit: true,
314                sandbox_lifetime: false,
315                // Local has no in-sandbox agent: the manager drives Docker from outside, so
316                // there is no supervisor inside the sandbox to isolate from.
317                supervisor_pid_namespace: false,
318                // The supervisor is the manager on the host, outside the container entirely, and
319                // `docker exec` runs the command as the workload uid — a different identity by
320                // construction.
321                supervisor_isolation: true,
322            }),
323            Platform::Machines | Platform::Test => {
324                Err(AlienError::new(ErrorData::SandboxPlatformUnsupported {
325                    platform: platform.to_string(),
326                }))
327            }
328        }
329    }
330
331    /// What the Azure sandbox backend supports; the body of the `Platform::Azure` arm.
332    pub fn azure() -> Self {
333        Self {
334            files: true,
335            reconnect: true,
336            // No Alien process runs inside the sandbox to own a command between two calls.
337            jobs: false,
338            // A sandbox port carries a URL and an auth config, and the auth config offers two
339            // things: anonymous, or Entra ID with an allowlist of human email addresses.
340            // Neither is a credential scoped to a port for a fixed time, which is what a
341            // preview capability is. Returning the anonymous URL would publish the port.
342            preview: false,
343            pause_resume: true,
344            // False for a client reason, not a cloud one: this client has no snapshot call, and
345            // `CreateSandboxRequest` has no field to consume the id it would return. Also
346            // unclaimed: Microsoft does not garbage-collect snapshots, so an id is a bill that grows.
347            snapshot: false,
348            domain_egress_rules: true,
349            egress_deny: true,
350            // Enforced inside the sandbox, not at create: an over-allocation raises `MemoryError`
351            // while the sandbox keeps running. `azure_sandbox_limits` checks the continuous
352            // sizing rule at plan time instead of matching a tier.
353            enforced_limits: true,
354            process_limit: false,
355            // Auto-suspend and auto-delete exist; a wall-clock ceiling does not. Accepting
356            // `maxLifetimeSeconds` here would be the silent no-op the capability set exists
357            // to prevent, so this is a decision rather than a gap.
358            sandbox_lifetime: false,
359            // No Alien process inside an Azure sandbox, so there is no supervisor to isolate.
360            supervisor_pid_namespace: false,
361            // No Alien process runs the command at all — the platform's own data plane does,
362            // so there is no separate supervisor identity to speak of.
363            supervisor_isolation: false,
364        }
365    }
366
367    /// What the GCP Agent Platform sandbox backend supports; the body of the `Platform::Gcp` arm.
368    pub fn gcp_agent_platform() -> Self {
369        Self {
370            // Agent file operations move over the sandbox envelope.
371            files: true,
372            // Reaching a sandbox across processes is safe because `generation` is derived from the
373            // container boot id read through the agent's health op, so a caller detects a container
374            // replaced under a stable sandbox name rather than reconnecting to a blank one.
375            reconnect: true,
376            jobs: true,
377            // No method mints a port-scoped ingress capability; the only ingress is `:execute`.
378            preview: false,
379            // `:pause` and `:resume` preserve the running container.
380            pause_resume: true,
381            // The create path never sends `sandbox_environment_snapshot`, so no sandbox state is
382            // reachable through the trait; declared false until the client carries it.
383            snapshot: false,
384            // Egress is shaped by VPC and DNS peering, which is not a hostname allowlist.
385            domain_egress_rules: false,
386            // A declared `deny` blocks both routed egress and DNS.
387            egress_deny: true,
388            // The declared ceilings are enforced, but by terminating the sandbox on breach rather
389            // than by refusing the allocation — a caller reading `true` should expect the sandbox
390            // to die, not a clean error at the point of the request.
391            enforced_limits: true,
392            // No ceiling on process count is observed.
393            process_limit: false,
394            // `ttl` maps to a sandbox `expireTime` the platform terminates at.
395            sandbox_lifetime: true,
396            // No PID-namespace isolation between the command and anything supervising it.
397            supervisor_pid_namespace: false,
398            // No separate supervisor identity: the command is not run under a different identity
399            // than the process supervising it.
400            supervisor_isolation: false,
401        }
402    }
403
404    /// Returns a typed error if the named capability is absent on this platform.
405    pub fn require(&self, capability: SandboxCapability, platform: Platform) -> Result<()> {
406        let available = match capability {
407            SandboxCapability::Files => self.files,
408            SandboxCapability::Reconnect => self.reconnect,
409            SandboxCapability::Jobs => self.jobs,
410            SandboxCapability::Preview => self.preview,
411            SandboxCapability::PauseResume => self.pause_resume,
412            SandboxCapability::Snapshot => self.snapshot,
413            SandboxCapability::DomainEgressRules => self.domain_egress_rules,
414            SandboxCapability::EgressDeny => self.egress_deny,
415            SandboxCapability::EnforcedLimits => self.enforced_limits,
416            SandboxCapability::ProcessLimit => self.process_limit,
417            SandboxCapability::SandboxLifetime => self.sandbox_lifetime,
418            SandboxCapability::SupervisorPidNamespace => self.supervisor_pid_namespace,
419            SandboxCapability::SupervisorIsolation => self.supervisor_isolation,
420        };
421
422        if available {
423            return Ok(());
424        }
425
426        Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
427            capability: capability.as_str().to_string(),
428            platform: platform.to_string(),
429        }))
430    }
431}
432
433/// Names a single sandbox capability, so an unsupported call can report which one it needed.
434#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
435#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
436#[serde(rename_all = "camelCase")]
437pub enum SandboxCapability {
438    /// Moving files in and out of a sandbox
439    Files,
440    /// Reaching a sandbox created by an earlier call
441    Reconnect,
442    /// Starting, polling and cancelling a command across separate calls
443    Jobs,
444    /// An authenticated, port-scoped ingress capability
445    Preview,
446    /// Pausing and resuming sandbox state
447    PauseResume,
448    /// Capturing full sandbox state
449    Snapshot,
450    /// Restricting egress to a hostname allowlist
451    DomainEgressRules,
452    /// Refusing outbound access when a sandbox declares none
453    EgressDeny,
454    /// Platform-enforced resource ceilings
455    EnforcedLimits,
456    /// A ceiling on the number of processes a sandbox may run
457    ProcessLimit,
458    /// A wall-clock ceiling on a sandbox, applied by the platform rather than by a caller
459    SandboxLifetime,
460    /// A command runs in its own PID namespace, isolated from the agent supervising it
461    SupervisorPidNamespace,
462    /// A command runs under a different identity than the process supervising it
463    SupervisorIsolation,
464}
465
466impl SandboxCapability {
467    /// Returns the stable identifier used in errors and capability queries.
468    pub fn as_str(&self) -> &'static str {
469        match self {
470            Self::Files => "files",
471            Self::Reconnect => "reconnect",
472            Self::Jobs => "jobs",
473            Self::Preview => "preview",
474            Self::PauseResume => "pauseResume",
475            Self::Snapshot => "snapshot",
476            Self::DomainEgressRules => "domainEgressRules",
477            Self::EgressDeny => "egressDeny",
478            Self::EnforcedLimits => "enforcedLimits",
479            Self::ProcessLimit => "processLimit",
480            Self::SandboxLifetime => "sandboxLifetime",
481            Self::SupervisorPidNamespace => "supervisorPidNamespace",
482            Self::SupervisorIsolation => "supervisorIsolation",
483        }
484    }
485}
486
487/// An isolated environment for running untrusted code, created at runtime.
488#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Builder)]
489#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
490#[serde(rename_all = "camelCase", deny_unknown_fields)]
491#[builder(start_fn = new)]
492pub struct Sandbox {
493    /// Identifier for the sandbox. Must contain only alphanumeric characters, hyphens, and
494    /// underscores ([A-Za-z0-9-_]). Maximum 64 characters.
495    #[builder(start_fn)]
496    pub id: String,
497    /// Where the sandbox's root filesystem comes from
498    pub code: SandboxCode,
499    /// Private ECR base image an AWS build pulls; `code.image` names only the S3 bundle, so this is
500    /// what the cross-account grant opens. Live only: the grant needs the customer account,
501    /// which registration reports. Absent means the base image is pulled anonymously.
502    #[serde(skip_serializing_if = "Option::is_none")]
503    pub private_base_image: Option<String>,
504    /// Enforced resource ceilings.
505    ///
506    /// Optional because not every platform can enforce them, and a declaration that names none
507    /// takes the platform's own defaults. Naming them on a platform that cannot enforce them is
508    /// rejected at plan time rather than silently ignored.
509    #[serde(skip_serializing_if = "Option::is_none")]
510    pub limits: Option<SandboxLimits>,
511    /// Outbound network policy
512    pub egress: SandboxEgress,
513    /// Sandbox lifetime ceiling and idle behaviour
514    pub lifecycle: SandboxLifecyclePolicy,
515    /// Ports eligible for a preview capability. An application reaches its sandbox through the
516    /// provider, so it cannot widen its own ingress at runtime; a holder of a remote binding's
517    /// credentials is bounded by no port condition, which is why a remote sandbox declares none.
518    #[builder(default)]
519    #[serde(default, skip_serializing_if = "Vec::is_empty")]
520    pub preview_ports: Vec<u16>,
521}
522
523/// Whether the artifact being rendered restricts which network modes it accepts.
524///
525/// Cloud setup needs explicit subnets for restricted sandbox connectors. Kubernetes targets do
526/// not emit these cloud backends.
527pub fn restricts_network_mode(stack: &crate::Stack, targets_kubernetes: bool) -> bool {
528    !targets_kubernetes && stack_needs_named_subnets_at_setup(stack)
529}
530
531/// Whether any setup-owned resource forces setup to name subnets.
532///
533/// Restricted sandbox connectors require subnet IDs. Callers rendering an artifact want
534/// [`restricts_network_mode`] instead:
535/// this one answers for the declaration, which on a Kubernetes target is not what gets emitted.
536pub fn stack_needs_named_subnets_at_setup(stack: &crate::Stack) -> bool {
537    stack.resources().any(|(_resource_id, resource)| {
538        resource
539            .config
540            .downcast_ref::<Sandbox>()
541            .is_some_and(|sandbox| !matches!(sandbox.egress, SandboxEgress::Allow))
542    })
543}
544
545impl Sandbox {
546    /// The resource type identifier for Sandbox
547    pub const RESOURCE_TYPE: ResourceType = ResourceType::from_static("sandbox");
548
549    /// Returns the sandbox's unique identifier.
550    pub fn id(&self) -> &str {
551        &self.id
552    }
553
554    /// The declared ceilings, or the defaults a platform applies when none were named.
555    ///
556    /// Backends want a concrete set: a sandbox with no declared ceilings still runs inside
557    /// whatever the platform gives it, and a backend that had to branch on `None` would end up
558    /// inventing its own default anyway.
559    pub fn resolved_limits(&self) -> SandboxLimits {
560        self.limits.clone().unwrap_or_else(default_limits)
561    }
562
563    /// Validates the declaration against what the target platform can enforce.
564    ///
565    /// Runs at plan time so an unenforceable limit or an unsupported egress mode fails before
566    /// anything is provisioned, rather than at the first exec.
567    pub fn validate_for_platform(&self, platform: Platform) -> Result<()> {
568        let capabilities = SandboxCapabilities::for_platform(platform)?;
569
570        // No backend builds a sandbox image from source: an empty image string schedules a pod
571        // that can never run, the silent no-op the capability contract forbids — the failure
572        // has to land here instead.
573        if let SandboxCode::Source { .. } = &self.code {
574            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
575                resource_id: self.id.clone(),
576                field: "code".to_string(),
577                value: "source".to_string(),
578                reason: "no sandbox backend builds an image from source yet; give code.image a \
579                         prebuilt reference"
580                    .to_string(),
581            }));
582        }
583
584        // Elsewhere `code.image` is pulled directly, so a second reference would be a grant
585        // nothing reads.
586        if self.private_base_image.is_some() && platform != Platform::Aws {
587            return Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
588                capability: "privateBaseImage".to_string(),
589                platform: platform.to_string(),
590            }));
591        }
592
593        // Read before the limits, because the image is declared whether or not any are.
594        if platform == Platform::Azure {
595            self.azure_catalog_image()?;
596        }
597
598        let Some(limits) = self.limits.as_ref() else {
599            // Nothing declared, so nothing to enforce and nothing to reject.
600            return self.validate_capabilities(&capabilities, platform);
601        };
602
603        validate_quantity(&self.id, "cpu", &limits.cpu)?;
604        validate_quantity(&self.id, "memory", &limits.memory)?;
605        validate_quantity(&self.id, "disk", &limits.disk)?;
606
607        if let Some(max_processes) = limits.max_processes {
608            if max_processes == 0 {
609                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
610                    resource_id: self.id.clone(),
611                    field: "maxProcesses".to_string(),
612                    value: "0".to_string(),
613                    reason: "a sandbox that may run no processes cannot run code".to_string(),
614                }));
615            }
616            capabilities.require(SandboxCapability::ProcessLimit, platform)?;
617        }
618
619        // Declaring limits a platform ignores is worse than not declaring them: the stack reads
620        // as bounded while the sandbox is not.
621        capabilities.require(SandboxCapability::EnforcedLimits, platform)?;
622
623        if platform == Platform::Azure {
624            self.azure_sandbox_limits()?;
625        }
626
627        if platform == Platform::Aws {
628            // Refused here rather than at emit so a customer sees it while planning, and so both
629            // package formats inherit the same answer.
630            self.microvm_tier()?;
631
632            // The ceiling is Lambda's, and it rejects the run rather than clamping — so a value
633            // outside it would pass planning, render into the package, and fail at the first
634            // sandbox. Kubernetes takes the same field with no such bound, which is why this
635            // sits under the AWS gate rather than on the type.
636            if let Some(seconds) = self.lifecycle.max_lifetime_seconds {
637                if !(1..=AWS_MAX_LIFETIME_SECONDS).contains(&seconds) {
638                    return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
639                        resource_id: self.id.clone(),
640                        field: "maxLifetimeSeconds".to_string(),
641                        value: seconds.to_string(),
642                        reason: format!(
643                            "AWS runs a MicroVM for between 1 and \
644                             {AWS_MAX_LIFETIME_SECONDS} seconds"
645                        ),
646                    }));
647                }
648            }
649        }
650
651        self.validate_capabilities(&capabilities, platform)
652    }
653
654    /// The catalog disk image Azure creates a sandbox from.
655    ///
656    /// Azure names a public catalog entry rather than pulling a reference, so a registry path,
657    /// tag or digest has nowhere to go. An allowlist, because the answer to "what else could be
658    /// in there" is a name the data plane rejects at the first sandbox, long after the apply.
659    pub fn azure_catalog_image(&self) -> Result<&str> {
660        let refused = |value: &str, reason: &str| {
661            AlienError::new(ErrorData::SandboxLimitInvalid {
662                resource_id: self.id.clone(),
663                field: "code.image".to_string(),
664                value: value.to_string(),
665                reason: reason.to_string(),
666            })
667        };
668
669        let SandboxCode::Image { image } = &self.code else {
670            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
671                resource_id: self.id.clone(),
672                field: "code".to_string(),
673                value: "source".to_string(),
674                reason: "no sandbox backend builds an image from source yet".to_string(),
675            }));
676        };
677
678        let image = image.trim();
679        if image.is_empty() {
680            return Err(refused(image, "a sandbox has to name an image"));
681        }
682        if !image
683            .chars()
684            .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '-'))
685        {
686            return Err(refused(
687                image,
688                "Azure creates a sandbox from a public catalog disk image, so code.image must be \
689                 a bare catalog name such as 'ubuntu'",
690            ));
691        }
692        Ok(image)
693    }
694
695    /// Checks the declared ceilings against Azure's sizing rule (the `AZURE_*` constants above).
696    /// Refused at plan time, like [`Self::microvm_tier`], so a bad value is a declaration to fix
697    /// rather than a runtime fault at create.
698    pub fn azure_sandbox_limits(&self) -> Result<()> {
699        let Some(limits) = self.limits.as_ref() else {
700            // Nothing declared means the binding substitutes Alien's own default sizing, which is
701            // inside the rule — asserted where those constants live, since this cannot see them.
702            return Ok(());
703        };
704
705        let refused = |field: &str, value: &str, reason: &str| {
706            AlienError::new(ErrorData::SandboxLimitInvalid {
707                resource_id: self.id.clone(),
708                field: field.to_string(),
709                value: value.to_string(),
710                reason: reason.to_string(),
711            })
712        };
713
714        let cpu_millicores = millicores(&limits.cpu)
715            .ok_or_else(|| refused("cpu", &limits.cpu, "expected cores or millicores"))?;
716
717        // The multiple is checked, not just the range: `333m` sits inside 0.25–16 cores and is
718        // still refused on the wire, so a bounds-only check would pass a declaration that fails
719        // at create.
720        if cpu_millicores % AZURE_CPU_STEP_MILLICORES != 0
721            || !(AZURE_CPU_STEP_MILLICORES..=AZURE_MAX_CPU_MILLICORES).contains(&cpu_millicores)
722        {
723            return Err(refused(
724                "cpu",
725                &limits.cpu,
726                "Azure allocates cpu in steps of 250m from 250m to 16000m",
727            ));
728        }
729
730        // Both ceilings are derived from the cpu, so they cannot be checked before it is known.
731        let memory_ceiling_mib = cpu_millicores * AZURE_MEMORY_MIB_PER_CORE / 1000;
732        let disk_ceiling_mib = cpu_millicores * AZURE_DISK_MIB_PER_CORE / 1000;
733
734        let memory_mib = quantity_mib(&limits.memory)
735            .ok_or_else(|| refused("memory", &limits.memory, "Azure sizes memory in whole MiB"))?;
736        if memory_mib > memory_ceiling_mib {
737            return Err(refused(
738                "memory",
739                &limits.memory,
740                &format!(
741                    "Azure allows at most 2Gi of memory per core, or {memory_ceiling_mib}Mi \
742                          at the declared cpu"
743                ),
744            ));
745        }
746
747        let disk_mib = quantity_mib(&limits.disk)
748            .ok_or_else(|| refused("disk", &limits.disk, "Azure sizes disk in whole MiB"))?;
749        if disk_mib > disk_ceiling_mib {
750            return Err(refused(
751                "disk",
752                &limits.disk,
753                &format!(
754                    "Azure allows at most 20Gi of disk per core, or {disk_ceiling_mib}Mi at \
755                          the declared cpu"
756                ),
757            ));
758        }
759
760        Ok(())
761    }
762
763    /// The MicroVM size that keeps every declared ceiling, or why none does.
764    ///
765    /// AWS sizes are discrete and a running MicroVM bursts to four times its baseline, so the
766    /// only tier that honours a ceiling is one whose peak fits inside it. A declaration no tier
767    /// satisfies is refused: shipping the nearest size would give the customer a sandbox that
768    /// exceeds the bound they wrote down.
769    pub fn microvm_tier(&self) -> Result<MicrovmTier> {
770        let Some(limits) = self.limits.as_ref() else {
771            // Nothing declared: AWS's own default baseline, which is also `default_limits`.
772            return Ok(MICROVM_TIERS[2]);
773        };
774
775        let memory_mib = quantity_mib(&limits.memory).ok_or_else(|| {
776            AlienError::new(ErrorData::SandboxLimitInvalid {
777                resource_id: self.id.clone(),
778                field: "memory".to_string(),
779                value: limits.memory.clone(),
780                reason: "AWS sizes a MicroVM in whole MiB".to_string(),
781            })
782        })?;
783        let disk_mib = quantity_mib(&limits.disk).ok_or_else(|| {
784            AlienError::new(ErrorData::SandboxLimitInvalid {
785                resource_id: self.id.clone(),
786                field: "disk".to_string(),
787                value: limits.disk.clone(),
788                reason: "AWS sizes a MicroVM's disk in whole MiB".to_string(),
789            })
790        })?;
791        let cpu_millicores = millicores(&limits.cpu).ok_or_else(|| {
792            AlienError::new(ErrorData::SandboxLimitInvalid {
793                resource_id: self.id.clone(),
794                field: "cpu".to_string(),
795                value: limits.cpu.clone(),
796                reason: "expected cores or millicores".to_string(),
797            })
798        })?;
799
800        // Memory and disk choose the size; cpu is then checked rather than used to choose.
801        // AWS couples cpu to memory at 2 GB per vCPU, so letting a low cpu ceiling select the
802        // size too would quietly hand back a machine four times smaller than the memory ceiling
803        // asked for, with nothing to indicate it.
804        let sized = |tier: &&MicrovmTier| {
805            tier.peak_memory_mib <= memory_mib && tier.max_disk_mib <= disk_mib
806        };
807
808        let tier = MICROVM_TIERS
809            .iter()
810            .rev()
811            .find(sized)
812            .copied()
813            .ok_or_else(|| {
814                AlienError::new(ErrorData::SandboxLimitInvalid {
815                    resource_id: self.id.clone(),
816                    field: "memory".to_string(),
817                    value: limits.memory.clone(),
818                    reason: format!(
819                        "a Lambda MicroVM bursts to four times its baseline, so the smallest \
820                         ceiling AWS can hold is 2Gi memory with 8Gi disk; '{}' memory and '{}' \
821                         disk fit no size",
822                        limits.memory, limits.disk
823                    ),
824                })
825            })?;
826
827        let required_millicores = i64::from(tier.peak_vcpu) * 1000;
828        if cpu_millicores < required_millicores {
829            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
830                resource_id: self.id.clone(),
831                field: "cpu".to_string(),
832                value: limits.cpu.clone(),
833                reason: format!(
834                    "AWS allocates one vCPU per 2GB, so a MicroVM sized to a '{}' memory ceiling \
835                     reaches {} vCPU; declare cpu '{}' or lower the memory ceiling",
836                    limits.memory, tier.peak_vcpu, tier.peak_vcpu
837                ),
838            }));
839        }
840
841        Ok(tier)
842    }
843
844    /// The capability checks that do not depend on declared limits.
845    fn validate_capabilities(
846        &self,
847        capabilities: &SandboxCapabilities,
848        platform: Platform,
849    ) -> Result<()> {
850        if matches!(self.egress, SandboxEgress::AllowDomains { .. }) {
851            capabilities.require(SandboxCapability::DomainEgressRules, platform)?;
852        }
853
854        // `allow` asks for no restriction, so a backend that ignores it fails loudly on the first
855        // blocked connection. `deny` asks for one, and a backend that ignores it puts untrusted
856        // code on the internet with nothing to notice — so only this direction is gated.
857        // An empty list is not a restriction anyone wrote down: it renders as a deny-all wearing
858        // an allowlist's label, which reads at a glance as the opposite of what it does.
859        if let SandboxEgress::AllowDomains { domains } = &self.egress {
860            if domains.is_empty() {
861                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
862                    resource_id: self.id.clone(),
863                    field: "egress.domains".to_string(),
864                    value: "[]".to_string(),
865                    reason: "an allowlist naming no domain denies everything; declare \
866                             egress: deny if that is what was meant"
867                        .to_string(),
868                }));
869            }
870        }
871
872        if matches!(self.egress, SandboxEgress::Deny) {
873            capabilities.require(SandboxCapability::EgressDeny, platform)?;
874        }
875
876        if !self.preview_ports.is_empty() {
877            capabilities.require(SandboxCapability::Preview, platform)?;
878        }
879
880        if self.lifecycle.idle_pause_seconds.is_some() {
881            capabilities.require(SandboxCapability::PauseResume, platform)?;
882        }
883
884        if self.lifecycle.max_lifetime_seconds.is_some() {
885            capabilities.require(SandboxCapability::SandboxLifetime, platform)?;
886        }
887
888        Ok(())
889    }
890}
891
892/// Ceilings applied when a declaration names none.
893///
894/// Modest on purpose: an undeclared sandbox is one whose author did not think about sizing, and
895/// the safe reading of that is a small box rather than a generous one.
896fn default_limits() -> SandboxLimits {
897    SandboxLimits {
898        cpu: "1".to_string(),
899        memory: "2Gi".to_string(),
900        disk: "8Gi".to_string(),
901        max_processes: None,
902    }
903}
904
905/// Validates a Kubernetes-style resource quantity such as `500m`, `2Gi` or `1`.
906fn validate_quantity(resource_id: &str, field: &str, value: &str) -> Result<()> {
907    let invalid = |reason: &str| {
908        AlienError::new(ErrorData::SandboxLimitInvalid {
909            resource_id: resource_id.to_string(),
910            field: field.to_string(),
911            value: value.to_string(),
912            reason: reason.to_string(),
913        })
914    };
915
916    let digits_end = value
917        .find(|c: char| !c.is_ascii_digit() && c != '.')
918        .unwrap_or(value.len());
919    let (number, suffix) = value.split_at(digits_end);
920
921    let parsed: f64 = number
922        .parse()
923        .map_err(|_| invalid("expected a number, optionally followed by a unit suffix"))?;
924
925    if parsed <= 0.0 {
926        return Err(invalid("must be greater than zero"));
927    }
928
929    const SUFFIXES: &[&str] = &["", "m", "k", "M", "G", "T", "Ki", "Mi", "Gi", "Ti"];
930    if !SUFFIXES.contains(&suffix) {
931        return Err(invalid(
932            "unit must be one of m, k, M, G, T, Ki, Mi, Gi, Ti, or absent",
933        ));
934    }
935
936    Ok(())
937}
938
939/// Splits a quantity into its number and unit suffix.
940fn split_quantity(value: &str) -> Option<(f64, &str)> {
941    let trimmed = value.trim();
942    let digits_end = trimmed
943        .find(|c: char| !c.is_ascii_digit() && c != '.')
944        .unwrap_or(trimmed.len());
945    let (number, suffix) = trimmed.split_at(digits_end);
946    number.parse().ok().map(|number| (number, suffix))
947}
948
949/// A memory or disk quantity in whole MiB, rounded down.
950///
951/// Every suffix `validate_quantity` accepts is handled here. Reading only `Gi` and `Mi` and
952/// falling back for the rest would turn a declared `4G` into a different size than the customer
953/// asked for, which for a ceiling means a sandbox larger than its bound.
954pub fn quantity_mib(value: &str) -> Option<i64> {
955    let (number, suffix) = split_quantity(value)?;
956    let bytes = match suffix {
957        "" => number,
958        "k" => number * 1e3,
959        "M" => number * 1e6,
960        "G" => number * 1e9,
961        "T" => number * 1e12,
962        "Ki" => number * 1024.0,
963        "Mi" => number * 1024.0 * 1024.0,
964        "Gi" => number * 1024.0 * 1024.0 * 1024.0,
965        "Ti" => number * 1024.0 * 1024.0 * 1024.0 * 1024.0,
966        // `m` is a millicore suffix; memory has no use for it.
967        _ => return None,
968    };
969    Some((bytes / (1024.0 * 1024.0)) as i64)
970}
971
972/// A CPU quantity in millicores.
973pub fn millicores(value: &str) -> Option<i64> {
974    let (number, suffix) = split_quantity(value)?;
975    match suffix {
976        "" => Some((number * 1000.0) as i64),
977        "m" => Some(number as i64),
978        _ => None,
979    }
980}
981
982/// Outputs generated by a successfully provisioned Sandbox parent.
983#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
984#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
985#[serde(rename_all = "camelCase")]
986pub struct SandboxOutputs {
987    /// Name of the durable parent that sandboxes are created inside
988    pub parent_name: String,
989    /// Platform-specific identifier for the parent (image ARN, sandbox group id, namespace)
990    #[serde(skip_serializing_if = "Option::is_none")]
991    pub identifier: Option<String>,
992    /// Data-plane endpoint sandboxes are created through, where the platform has one
993    #[serde(skip_serializing_if = "Option::is_none")]
994    pub endpoint: Option<String>,
995}
996
997impl ResourceOutputsDefinition for SandboxOutputs {
998    fn get_resource_type(&self) -> ResourceType {
999        Sandbox::RESOURCE_TYPE
1000    }
1001
1002    fn as_any(&self) -> &dyn Any {
1003        self
1004    }
1005
1006    fn box_clone(&self) -> Box<dyn ResourceOutputsDefinition> {
1007        Box::new(self.clone())
1008    }
1009
1010    fn outputs_eq(&self, other: &dyn ResourceOutputsDefinition) -> bool {
1011        other.as_any().downcast_ref::<SandboxOutputs>() == Some(self)
1012    }
1013
1014    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1015        serde_json::to_value(self)
1016    }
1017}
1018
1019impl ResourceDefinition for Sandbox {
1020    fn get_resource_type(&self) -> ResourceType {
1021        Self::RESOURCE_TYPE
1022    }
1023
1024    fn id(&self) -> &str {
1025        &self.id
1026    }
1027
1028    fn get_dependencies(&self) -> Vec<ResourceRef> {
1029        Vec::new()
1030    }
1031
1032    fn validate_update(&self, new_config: &dyn ResourceDefinition) -> Result<()> {
1033        let new_sandbox = new_config
1034            .as_any()
1035            .downcast_ref::<Sandbox>()
1036            .ok_or_else(|| {
1037                AlienError::new(ErrorData::UnexpectedResourceType {
1038                    resource_id: self.id.clone(),
1039                    expected: Self::RESOURCE_TYPE,
1040                    actual: new_config.get_resource_type(),
1041                })
1042            })?;
1043
1044        if self.id != new_sandbox.id {
1045            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
1046                resource_id: self.id.clone(),
1047                reason: "the 'id' field is immutable".to_string(),
1048            }));
1049        }
1050
1051        Ok(())
1052    }
1053
1054    fn as_any(&self) -> &dyn Any {
1055        self
1056    }
1057
1058    fn as_any_mut(&mut self) -> &mut dyn Any {
1059        self
1060    }
1061
1062    fn box_clone(&self) -> Box<dyn ResourceDefinition> {
1063        Box::new(self.clone())
1064    }
1065
1066    fn resource_eq(&self, other: &dyn ResourceDefinition) -> bool {
1067        other.as_any().downcast_ref::<Sandbox>() == Some(self)
1068    }
1069
1070    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1071        serde_json::to_value(self)
1072    }
1073}
1074
1075/// The one token a sandbox bundle URI may carry, replaced with the deploying region.
1076///
1077/// AWS builds a MicroVM image only from a bucket in the image's own region, so a vendor
1078/// publishing to every supported region needs one stored URI that resolves per region.
1079pub const BUNDLE_REGION_TOKEN: &str = "{region}";
1080
1081/// A bundle URI split around its region token, or carried whole when it has none.
1082#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1083pub enum BundleUri<'a> {
1084    /// No token: emitted exactly as it is today.
1085    Literal(&'a str),
1086    /// The text either side of the token, for an emitter to rejoin around its own region
1087    /// expression.
1088    Regional { before: &'a str, after: &'a str },
1089}
1090
1091/// The prefix a rebuild's new key still sits under: everything above the file name and the
1092/// version segment beneath it. `None` when nothing sits there, meaning no prefix can be granted
1093/// without also granting objects a rebuild never reads. Shared so both emitters agree on it.
1094pub fn stable_bundle_key_prefix(key: &str) -> Option<&str> {
1095    let (above_file, _) = key.rsplit_once('/')?;
1096    let (above_version, _) = above_file.rsplit_once('/')?;
1097    Some(above_version)
1098}
1099
1100/// Reads a sandbox bundle URI, refusing anything an image build would only reject later.
1101///
1102/// The token is accepted in the bucket alone. A key-position token would name an object that does
1103/// not exist, and any other brace is a typo that would otherwise reach S3 verbatim and fail ~160s
1104/// into the build — which is the failure this whole check exists to move to plan time.
1105pub fn parse_bundle_uri(uri: &str) -> std::result::Result<BundleUri<'_>, String> {
1106    let path = uri
1107        .strip_prefix("s3://")
1108        .ok_or_else(|| format!("'{uri}' is not an s3:// URI"))?;
1109    let (bucket, key) = path
1110        .split_once('/')
1111        .ok_or_else(|| format!("'{uri}' names a bucket with no object key"))?;
1112
1113    // Both emitters interpolate this path into the build role's resource ARN, where `*` and `?`
1114    // are IAM wildcards rather than literal characters. S3 accepts them in a key, so a bundle
1115    // published under one would silently widen the grant past the bundle it names.
1116    if path.contains('*') || path.contains('?') {
1117        return Err(format!(
1118            "'{uri}' carries an IAM wildcard; the bundle's path is interpolated into the build \
1119             role's grant, so '*' and '?' would widen it past the bundle"
1120        ));
1121    }
1122
1123    if key.contains('{') || key.contains('}') {
1124        return Err(format!(
1125            "'{uri}' places a token in the object key; {BUNDLE_REGION_TOKEN} is accepted in the \
1126             bucket name alone"
1127        ));
1128    }
1129
1130    let Some((before, after)) = bucket.split_once(BUNDLE_REGION_TOKEN) else {
1131        if bucket.contains('{') || bucket.contains('}') {
1132            return Err(format!(
1133                "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the \
1134                 only one"
1135            ));
1136        }
1137        return Ok(BundleUri::Literal(uri));
1138    };
1139
1140    if after.contains(BUNDLE_REGION_TOKEN) {
1141        return Err(format!("'{uri}' repeats {BUNDLE_REGION_TOKEN}"));
1142    }
1143    if before.contains('{') || before.contains('}') || after.contains('{') || after.contains('}') {
1144        return Err(format!(
1145            "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the only one"
1146        ));
1147    }
1148
1149    Ok(BundleUri::Regional {
1150        before: &uri[.."s3://".len() + before.len()],
1151        after: &uri["s3://".len() + before.len() + BUNDLE_REGION_TOKEN.len()..],
1152    })
1153}
1154
1155#[cfg(test)]
1156mod tests {
1157    use super::*;
1158
1159    #[test]
1160    fn private_database_setup_accepts_the_default_network() {
1161        for lifecycle in [
1162            crate::ResourceLifecycle::Frozen,
1163            crate::ResourceLifecycle::Live,
1164        ] {
1165            let stack = crate::Stack::new("database".to_string())
1166                .add(
1167                    crate::Postgres::new("metadata".to_string()).build(),
1168                    lifecycle,
1169                )
1170                .build();
1171            assert!(!restricts_network_mode(&stack, false));
1172            assert!(!restricts_network_mode(&stack, true));
1173        }
1174        assert!(!restricts_network_mode(
1175            &crate::Stack::new("empty".to_string()).build(),
1176            false,
1177        ));
1178    }
1179
1180    /// A wildcard reaching the grant would widen it past the bundle, and it widens the Frozen
1181    /// object grant as readily as the Live prefix — both interpolate the path into the ARN.
1182    #[test]
1183    fn a_uri_carrying_an_iam_wildcard_is_refused() {
1184        for uri in [
1185            "s3://acme/team-*/v1/bundle.zip",
1186            "s3://acme/sandbox-bundle/f00d/bundle?.zip",
1187            "s3://acme-*/sandbox-bundle/f00d/bundle.zip",
1188        ] {
1189            let error = parse_bundle_uri(uri).expect_err("a wildcard must be refused");
1190            assert!(error.contains("IAM wildcard"), "for {uri}: {error}");
1191        }
1192
1193        parse_bundle_uri("s3://acme/sandbox-bundle/f00d/bundle.zip")
1194            .expect("an ordinary key still parses");
1195    }
1196
1197    /// The rule both package formats grant by, pinned here rather than in either. The
1198    /// near-misses the cases separate: the key's first segment grants objects a rebuild never
1199    /// reads, and the object's own directory pins the version segment that moves.
1200    #[test]
1201    fn a_grantable_prefix_stops_above_the_segment_that_moves() {
1202        assert_eq!(
1203            stable_bundle_key_prefix("sandbox-bundle/f00dcafe/bundle.zip"),
1204            Some("sandbox-bundle")
1205        );
1206        assert_eq!(
1207            stable_bundle_key_prefix("artifacts/team-a/sandbox/f00dcafe/bundle.zip"),
1208            Some("artifacts/team-a/sandbox"),
1209            "a deeper key narrows the prefix, it never widens to the first segment"
1210        );
1211
1212        // Nothing sits above the version segment, so no prefix a moved bundle stays inside
1213        // exists. Emitting the object grant instead installs a role that denies the next rebuild.
1214        assert_eq!(stable_bundle_key_prefix("agents/bundle.zip"), None);
1215        assert_eq!(stable_bundle_key_prefix("bundle.zip"), None);
1216    }
1217
1218    fn sandbox_with(egress: SandboxEgress, preview_ports: Vec<u16>) -> Sandbox {
1219        Sandbox::new("agent-sbx".to_string())
1220            .code(SandboxCode::Image {
1221                image: "ubuntu".to_string(),
1222            })
1223            .limits(SandboxLimits {
1224                cpu: "1".to_string(),
1225                memory: "2Gi".to_string(),
1226                disk: "20Gi".to_string(),
1227                max_processes: None,
1228            })
1229            .egress(egress)
1230            .lifecycle(SandboxLifecyclePolicy {
1231                max_lifetime_seconds: None,
1232                idle_pause_seconds: None,
1233            })
1234            .preview_ports(preview_ports)
1235            .build()
1236    }
1237
1238    /// A URI with no token must come back whole, because every bundle configured today has none
1239    /// and emitting one differently would change every existing customer's template.
1240    #[test]
1241    fn a_uri_without_a_token_is_carried_whole() {
1242        assert_eq!(
1243            parse_bundle_uri("s3://acme-artifacts-us-east-2/agents/bundle.zip"),
1244            Ok(BundleUri::Literal(
1245                "s3://acme-artifacts-us-east-2/agents/bundle.zip"
1246            ))
1247        );
1248    }
1249
1250    /// The split has to rejoin to the original with the region in place, or an emitter builds a
1251    /// URI that is subtly not the one the vendor configured.
1252    #[test]
1253    fn a_regional_uri_splits_either_side_of_the_token() {
1254        let BundleUri::Regional { before, after } =
1255            parse_bundle_uri("s3://acme-artifacts-{region}/agents/bundle.zip")
1256                .expect("the token is accepted in the bucket")
1257        else {
1258            panic!("a bucket-position token must split");
1259        };
1260
1261        assert_eq!(before, "s3://acme-artifacts-");
1262        assert_eq!(after, "/agents/bundle.zip");
1263        assert_eq!(
1264            format!("{before}us-east-2{after}"),
1265            "s3://acme-artifacts-us-east-2/agents/bundle.zip",
1266            "the halves must rejoin to the URI the vendor meant"
1267        );
1268    }
1269
1270    /// Each of these reaches S3 verbatim and dies ~160s into an image build if it is not refused
1271    /// here, which is the whole reason this runs at plan time.
1272    #[test]
1273    fn a_token_this_build_cannot_resolve_is_refused() {
1274        for uri in [
1275            "s3://acme-artifacts-{regio}/bundle.zip",
1276            "s3://acme-artifacts/{region}/bundle.zip",
1277            "s3://acme-artifacts-{region}-{region}/bundle.zip",
1278            "s3://acme-artifacts/bundle-{version}.zip",
1279            "s3://acme}-artifacts-{region}/bundle.zip",
1280            "s3://acme{-artifacts-{region}/bundle.zip",
1281        ] {
1282            assert!(
1283                parse_bundle_uri(uri).is_err(),
1284                "'{uri}' must be refused before it can reach an image build"
1285            );
1286        }
1287    }
1288
1289    #[test]
1290    fn resource_type_is_stable() {
1291        assert_eq!(Sandbox::RESOURCE_TYPE.as_ref(), "sandbox");
1292    }
1293
1294    #[test]
1295    fn capability_sets_are_per_platform() {
1296        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1297        assert!(
1298            gcp.reconnect,
1299            "generation from the container boot id makes a sandbox reachable across processes"
1300        );
1301        assert!(!gcp.preview);
1302        assert!(gcp.enforced_limits);
1303
1304        let azure = SandboxCapabilities::for_platform(Platform::Azure).expect("azure is supported");
1305        assert!(azure.files, "every backend moves files");
1306        assert!(gcp.files);
1307        // Azure is the only backend whose egress policy matches on host pattern, and the only
1308        // one where `deny` and a hostname list are the same object.
1309        assert!(azure.domain_egress_rules);
1310        assert!(azure.egress_deny);
1311        // The data plane honours a continuous cpu/memory/disk surface and refuses anything
1312        // outside `n×250m`, with the rule in the message.
1313        assert!(azure.enforced_limits);
1314        assert!(azure.pause_resume);
1315        // Both stay false for reasons that are not "unbuilt": a snapshot id has nothing to
1316        // consume it on any backend, and an Azure port's auth is anonymous or a human allowlist,
1317        // neither of which is a port-scoped credential.
1318        assert!(!azure.snapshot);
1319        assert!(!azure.preview);
1320
1321        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1322        assert!(!aws.snapshot, "AWS has no user-callable sandbox snapshot");
1323        assert!(aws.pause_resume);
1324
1325        let k8s =
1326            SandboxCapabilities::for_platform(Platform::Kubernetes).expect("k8s is supported");
1327        assert!(
1328            !k8s.preview,
1329            "the sandbox-scoped ingress gateway does not exist yet"
1330        );
1331    }
1332
1333    /// Whether the process supervising a command is a separate identity from the command.
1334    ///
1335    /// Values are measured, not inferred. AWS: the agent runs as uid 0 with
1336    /// `CapEff: 00000000a80425fb` and `setuid`s the command to uid 60000, so the two differ.
1337    /// Kubernetes: the sandbox pod pins `run_as_user: 65534` on both pod and container with
1338    /// `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no uid split is
1339    /// possible (`kubernetes_spec.rs`). Local: `docker exec` runs as the workload uid while the
1340    /// manager supervises from the host. Azure and Agent Platform have no in-sandbox supervisor.
1341    #[test]
1342    fn supervisor_isolation_is_per_platform() {
1343        let value = |platform| {
1344            SandboxCapabilities::for_platform(platform)
1345                .expect("supported")
1346                .supervisor_isolation
1347        };
1348
1349        assert!(
1350            value(Platform::Aws),
1351            "root agent setuids the command to 60000"
1352        );
1353        assert!(
1354            value(Platform::Local),
1355            "the supervisor is on the host, outside the container"
1356        );
1357        assert!(
1358            !value(Platform::Kubernetes),
1359            "a single pinned uid cannot be split"
1360        );
1361        assert!(!value(Platform::Azure), "no Alien process runs the command");
1362        assert!(
1363            !value(Platform::Gcp),
1364            "no separate supervisor identity runs the command"
1365        );
1366    }
1367
1368    /// The point of the field: AWS and GCP report the *same* `supervisor_pid_namespace` (neither
1369    /// has `CAP_SYS_ADMIN`), so that axis alone reads them as equivalent. They are not — AWS
1370    /// separates the command's identity from the supervisor's and Agent Platform does not.
1371    #[test]
1372    fn supervisor_isolation_separates_aws_from_a_subprocess_backend() {
1373        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1374        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1375
1376        assert_eq!(
1377            aws.supervisor_pid_namespace, gcp.supervisor_pid_namespace,
1378            "the older axis cannot tell them apart"
1379        );
1380        assert!(
1381            aws.supervisor_isolation,
1382            "AWS setuids the command off the supervisor"
1383        );
1384        assert!(
1385            !gcp.supervisor_isolation,
1386            "the command runs under no separate supervisor identity"
1387        );
1388    }
1389
1390    /// The Agent Platform row, each value against the behaviour it was measured from. `reconnect`
1391    /// is the tripwire: it is `true` only because `generation` is derived from the container boot
1392    /// id read through the agent's health op, so a caller detects a replaced container instead of
1393    /// reconnecting to a blank one. It is also the body of the `Platform::Gcp` arm, asserted below.
1394    #[test]
1395    fn gcp_agent_platform_row_matches_measured_backend() {
1396        let row = SandboxCapabilities::gcp_agent_platform();
1397
1398        assert!(row.files, "agent file ops move over the sandbox envelope");
1399        assert!(
1400            row.reconnect,
1401            "generation is derived from the container boot id, so a sandbox is reachable across \
1402             processes"
1403        );
1404        assert!(
1405            !row.preview,
1406            "the only ingress is :execute; no port-scoped capability"
1407        );
1408        assert!(
1409            row.pause_resume,
1410            ":pause and :resume preserve the container"
1411        );
1412        assert!(
1413            !row.snapshot,
1414            "the create path never sends a snapshot, so none is reachable through the trait"
1415        );
1416        assert!(
1417            !row.domain_egress_rules,
1418            "VPC and DNS peering is not a hostname allowlist"
1419        );
1420        assert!(
1421            row.egress_deny,
1422            "a declared deny blocks both egress and DNS"
1423        );
1424        assert!(
1425            row.enforced_limits,
1426            "ceilings are enforced, by terminating the sandbox on breach"
1427        );
1428        assert!(!row.process_limit, "no process-count ceiling is observed");
1429        assert!(row.sandbox_lifetime, "ttl maps to a sandbox expireTime");
1430        assert!(!row.supervisor_pid_namespace, "no PID-namespace isolation");
1431        assert!(
1432            !row.supervisor_isolation,
1433            "the command is not run under a separate supervisor identity"
1434        );
1435
1436        // Agent Platform is the registered GCP backend, so the arm returns exactly this row.
1437        let live = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1438        assert_eq!(
1439            live, row,
1440            "the Platform::Gcp arm is the Agent Platform capability row"
1441        );
1442    }
1443
1444    #[test]
1445    fn platforms_without_a_backend_are_an_error_not_an_empty_set() {
1446        let error = SandboxCapabilities::for_platform(Platform::Machines)
1447            .expect_err("Machines has no sandbox backend");
1448        assert_eq!(error.code, "SANDBOX_PLATFORM_UNSUPPORTED");
1449    }
1450
1451    #[test]
1452    fn unsupported_capability_names_platform_and_capability() {
1453        let capabilities = SandboxCapabilities::for_platform(Platform::Gcp).expect("supported");
1454        let error = capabilities
1455            .require(SandboxCapability::Preview, Platform::Gcp)
1456            .expect_err("GCP has no preview");
1457
1458        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1459        let rendered = error.to_string();
1460        assert!(
1461            rendered.contains("preview"),
1462            "names the capability: {rendered}"
1463        );
1464        assert!(rendered.contains("gcp"), "names the platform: {rendered}");
1465    }
1466
1467    /// Azure matches on hostname; AWS and Kubernetes match CIDRs, and Local and GCP have a
1468    /// switch rather than a filter. Accepting a hostname list on those four would leave a stack
1469    /// reading as restricted while the sandbox reaches the whole internet.
1470    #[test]
1471    fn a_hostname_allowlist_is_refused_everywhere_it_would_be_approximated() {
1472        let sandbox = sandbox_with(
1473            SandboxEgress::AllowDomains {
1474                domains: vec!["example.com".to_string()],
1475            },
1476            vec![],
1477        );
1478
1479        for platform in [
1480            Platform::Aws,
1481            Platform::Gcp,
1482            Platform::Kubernetes,
1483            Platform::Local,
1484        ] {
1485            let error = sandbox
1486                .validate_for_platform(platform)
1487                .expect_err("only Azure expresses a hostname allowlist");
1488            assert_eq!(
1489                error.code, "SANDBOX_CAPABILITY_UNSUPPORTED",
1490                "on {platform:?}"
1491            );
1492        }
1493
1494        assert!(
1495            SandboxCapabilities::for_platform(Platform::Azure)
1496                .expect("supported")
1497                .domain_egress_rules,
1498            "Azure's egress policy matches on host pattern"
1499        );
1500    }
1501
1502    /// `deny` is the declaration that carries a security promise, so a backend that cannot keep
1503    /// it has to refuse rather than accept it and run the code with open egress.
1504    #[test]
1505    fn a_denied_egress_is_refused_where_it_would_not_be_enforced() {
1506        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1507
1508        // GCP is asserted at the capability rather than through validation: this sandbox declares
1509        // ceilings GCP cannot enforce, so it is refused for a reason unrelated to egress.
1510        assert!(
1511            SandboxCapabilities::for_platform(Platform::Gcp)
1512                .expect("supported")
1513                .egress_deny
1514        );
1515
1516        for platform in [Platform::Aws, Platform::Kubernetes, Platform::Local] {
1517            sandbox
1518                .validate_for_platform(platform)
1519                .expect("deny is enforced here");
1520        }
1521
1522        // Declares no ceilings, which Azure refuses for its own reason, so this isolates egress.
1523        let egress_only = Sandbox::new("sbx".to_string())
1524            .code(SandboxCode::Image {
1525                image: "alpine".to_string(),
1526            })
1527            .egress(SandboxEgress::Deny)
1528            .lifecycle(SandboxLifecyclePolicy {
1529                max_lifetime_seconds: None,
1530                idle_pause_seconds: None,
1531            })
1532            .build();
1533
1534        egress_only
1535            .validate_for_platform(Platform::Azure)
1536            .expect("Azure creates the sandbox under a Deny policy with full inspection");
1537    }
1538
1539    /// A sandbox naming no ceilings is valid on every platform and still resolves to a concrete
1540    /// set. The rule Azure applies to ceilings that *are* declared is pinned separately, by
1541    /// `azure_sizes_follow_the_rule_the_data_plane_states`.
1542    #[test]
1543    fn a_sandbox_declaring_no_ceilings_takes_the_platforms_own() {
1544        let undeclared = Sandbox::new("sbx".to_string())
1545            .code(SandboxCode::Image {
1546                image: "alpine".to_string(),
1547            })
1548            .egress(SandboxEgress::Deny)
1549            .lifecycle(SandboxLifecyclePolicy {
1550                max_lifetime_seconds: None,
1551                idle_pause_seconds: None,
1552            })
1553            .build();
1554
1555        undeclared
1556            .validate_for_platform(Platform::Azure)
1557            .expect("a sandbox naming no ceilings takes the platform's own");
1558
1559        // A backend still gets a concrete set, so nothing downstream has to invent one.
1560        assert_eq!(undeclared.resolved_limits().cpu, "1");
1561    }
1562
1563    /// Pins Azure's own sizing rule: `250m`, `1500m` and `4000m` are valid, `32000m` and `333m`
1564    /// are refused. Checked at plan time so a bad value is a declaration to fix, not a package
1565    /// that renders without error and dies at the first sandbox.
1566    #[test]
1567    fn azure_sizes_follow_the_rule_the_data_plane_states() {
1568        let sized = |cpu: &str, memory: &str, disk: &str| {
1569            let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1570            let limits = sandbox
1571                .limits
1572                .as_mut()
1573                .expect("the fixture declares limits");
1574            limits.cpu = cpu.to_string();
1575            limits.memory = memory.to_string();
1576            limits.disk = disk.to_string();
1577            sandbox.validate_for_platform(Platform::Azure)
1578        };
1579
1580        sized("250m", "512Mi", "5120Mi").expect("the smallest step the data plane accepts");
1581        sized("4000m", "8192Mi", "40960Mi").expect("cpu, memory and disk are all honoured");
1582        sized("16000m", "32Gi", "320Gi").expect("the top of the range");
1583
1584        // Same off-step case `azure_sandbox_limits` checks the multiple for, not just the range.
1585        let off_step = sized("333m", "512Mi", "5120Mi").expect_err("333m is not a step of 250m");
1586        assert_eq!(off_step.code, "SANDBOX_LIMIT_INVALID", "{off_step}");
1587        assert!(off_step.to_string().contains("cpu"), "{off_step}");
1588
1589        let too_big = sized("32000m", "64Gi", "640Gi").expect_err("32 cores is over the ceiling");
1590        assert_eq!(too_big.code, "SANDBOX_LIMIT_INVALID", "{too_big}");
1591
1592        // Both ceilings are derived from the cpu, so the same memory passes at one size and fails
1593        // at another - which is what makes them worth checking rather than bounding absolutely.
1594        sized("1000m", "2Gi", "20Gi").expect("2Gi is exactly one core's worth");
1595        let over_memory = sized("250m", "2Gi", "5120Mi").expect_err("2Gi needs a full core");
1596        assert_eq!(over_memory.code, "SANDBOX_LIMIT_INVALID", "{over_memory}");
1597        assert!(over_memory.to_string().contains("memory"), "{over_memory}");
1598
1599        let over_disk = sized("250m", "512Mi", "20Gi").expect_err("20Gi needs a full core");
1600        assert!(over_disk.to_string().contains("disk"), "{over_disk}");
1601    }
1602
1603    #[test]
1604    fn preview_ports_require_the_preview_capability() {
1605        let sandbox = sandbox_with(SandboxEgress::Deny, vec![8080]);
1606
1607        sandbox
1608            .validate_for_platform(Platform::Aws)
1609            .expect("AWS mints a port-scoped JWE");
1610
1611        let error = sandbox
1612            .validate_for_platform(Platform::Kubernetes)
1613            .expect_err("Kubernetes preview is deferred");
1614        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1615    }
1616
1617    /// A grant nothing reads is the silent no-op the capability contract exists to prevent, and
1618    /// here it is worse than useless: the reader would take it for a base image that needs
1619    /// authenticating while the platform pulls `code.image` itself.
1620    #[test]
1621    fn a_private_base_image_is_refused_off_aws() {
1622        let mut sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1623        sandbox.code = SandboxCode::Image {
1624            image: "s3://acme-artifacts/agents/bundle.zip".to_string(),
1625        };
1626        sandbox.private_base_image =
1627            Some("123456789012.dkr.ecr.{region}.amazonaws.com/acme:tag".to_string());
1628
1629        sandbox
1630            .validate_for_platform(Platform::Aws)
1631            .expect("AWS builds its image from a bundle, so a base image sits behind code.image");
1632
1633        for platform in [
1634            Platform::Gcp,
1635            Platform::Azure,
1636            Platform::Kubernetes,
1637            Platform::Local,
1638        ] {
1639            let error = sandbox
1640                .validate_for_platform(platform)
1641                .expect_err("a backend that builds no image must refuse a base image for one");
1642            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1643            assert!(
1644                error.to_string().contains("privateBaseImage"),
1645                "the refusal must name the field the user declared: {error}"
1646            );
1647        }
1648    }
1649
1650    #[test]
1651    fn gcp_accepts_a_sandbox_declaring_enforced_limits() {
1652        let sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1653        sandbox
1654            .validate_for_platform(Platform::Gcp)
1655            .expect("Agent Platform enforces declared ceilings, by terminating on breach");
1656    }
1657
1658    #[test]
1659    fn invalid_quantities_are_rejected_with_the_offending_field() {
1660        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1661        sandbox
1662            .limits
1663            .as_mut()
1664            .expect("the fixture declares limits")
1665            .memory = "2Gb".to_string();
1666
1667        let error = sandbox
1668            .validate_for_platform(Platform::Aws)
1669            .expect_err("Gb is not a valid suffix");
1670        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1671        assert!(error.to_string().contains("memory"));
1672
1673        sandbox
1674            .limits
1675            .as_mut()
1676            .expect("the fixture declares limits")
1677            .memory = "2Gi".to_string();
1678        sandbox
1679            .limits
1680            .as_mut()
1681            .expect("the fixture declares limits")
1682            .cpu = "0".to_string();
1683        let error = sandbox
1684            .validate_for_platform(Platform::Aws)
1685            .expect_err("zero cpu is not a ceiling");
1686        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1687    }
1688
1689    #[test]
1690    fn zero_max_processes_is_rejected() {
1691        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1692        sandbox
1693            .limits
1694            .as_mut()
1695            .expect("the fixture declares limits")
1696            .max_processes = Some(0);
1697
1698        let error = sandbox
1699            .validate_for_platform(Platform::Local)
1700            .expect_err("a sandbox must be able to run at least one process");
1701        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1702        assert!(error.to_string().contains("maxProcesses"));
1703    }
1704
1705    /// A process ceiling needs a container runtime. Kubernetes sets one per node rather than per
1706    /// pod, and neither MicroVMs nor Azure sandboxes expose one, so accepting the declaration
1707    /// anywhere else would mean carrying a bound nothing applies.
1708    #[test]
1709    fn a_process_ceiling_is_accepted_only_where_a_runtime_can_apply_it() {
1710        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1711        sandbox
1712            .limits
1713            .as_mut()
1714            .expect("the fixture declares limits")
1715            .max_processes = Some(256);
1716
1717        sandbox
1718            .validate_for_platform(Platform::Local)
1719            .expect("Docker takes a pids limit");
1720
1721        for platform in [Platform::Aws, Platform::Azure, Platform::Kubernetes] {
1722            let error = sandbox
1723                .validate_for_platform(platform)
1724                .expect_err("a process ceiling nothing applies must be refused");
1725            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1726        }
1727    }
1728
1729    /// Lambda rejects a run outside 1–28,800 rather than clamping it, so a value beyond that
1730    /// would pass planning, render into the package, and fail at the first sandbox. Kubernetes
1731    /// takes the same field with no such bound, so the check is AWS's alone.
1732    #[test]
1733    fn a_lifetime_aws_would_reject_is_refused_while_planning() {
1734        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1735
1736        for seconds in [0, 28_801, 100_000] {
1737            sandbox.lifecycle.max_lifetime_seconds = Some(seconds);
1738            let error = sandbox
1739                .validate_for_platform(Platform::Aws)
1740                .expect_err("a lifetime outside what AWS runs is refused");
1741            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "{seconds}s");
1742
1743            // Kubernetes has no such ceiling, so the same declaration is fine there.
1744            sandbox
1745                .validate_for_platform(Platform::Kubernetes)
1746                .expect("the kubelet takes any activeDeadlineSeconds");
1747        }
1748
1749        sandbox.lifecycle.max_lifetime_seconds = Some(28_800);
1750        sandbox
1751            .validate_for_platform(Platform::Aws)
1752            .expect("the ceiling itself is allowed");
1753    }
1754
1755    /// An image reference Azure cannot honour is refused while planning, not at the first sandbox.
1756    ///
1757    /// `code.image`'s own documentation gives a tag and a registry path as examples — exactly
1758    /// what Azure cannot take, so this is the shape a customer is most likely to declare.
1759    #[test]
1760    fn an_image_azure_cannot_pull_is_refused_while_planning() {
1761        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1762        // Azure enforces no declared ceiling, so a sandbox carrying limits is refused before the
1763        // image is ever read.
1764        sandbox.limits = None;
1765
1766        for image in [
1767            "ubuntu:24.04",
1768            "ghcr.io/myorg/sandbox:latest",
1769            "ubuntu@sha256:abc",
1770            "",
1771            "   ",
1772            "ubuntu latest",
1773            "ubuntu?x",
1774        ] {
1775            sandbox.code = SandboxCode::Image {
1776                image: image.to_string(),
1777            };
1778            let error = sandbox
1779                .validate_for_platform(Platform::Azure)
1780                .expect_err("an image Azure has nowhere to put is refused");
1781            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "image '{image}'");
1782
1783            // The same declaration is ordinary everywhere that pulls a reference.
1784            sandbox
1785                .validate_for_platform(Platform::Kubernetes)
1786                .expect("a registry reference is what every other backend takes");
1787        }
1788
1789        for image in ["ubuntu", "ubuntu-22.04", "debian_slim"] {
1790            sandbox.code = SandboxCode::Image {
1791                image: image.to_string(),
1792            };
1793            sandbox
1794                .validate_for_platform(Platform::Azure)
1795                .unwrap_or_else(|error| panic!("'{image}' is a catalog name: {error}"));
1796        }
1797
1798        // Surrounding space is trimmed rather than carried into the create body.
1799        sandbox.code = SandboxCode::Image {
1800            image: " ubuntu ".to_string(),
1801        };
1802        assert_eq!(
1803            sandbox
1804                .azure_catalog_image()
1805                .expect("a padded name is still a name"),
1806            "ubuntu"
1807        );
1808    }
1809
1810    /// A deadline is accepted only where the platform itself terminates on it — the kubelet's
1811    /// `activeDeadlineSeconds` and Lambda's `maximumDurationInSeconds`. Everywhere else it would
1812    /// need a reaper that does not exist, so it is refused rather than accepted and dropped.
1813    #[test]
1814    fn a_sandbox_deadline_is_accepted_only_where_the_platform_applies_it() {
1815        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1816        sandbox.lifecycle.max_lifetime_seconds = Some(3600);
1817
1818        sandbox
1819            .validate_for_platform(Platform::Kubernetes)
1820            .expect("the kubelet enforces activeDeadlineSeconds");
1821        sandbox
1822            .validate_for_platform(Platform::Aws)
1823            .expect("Lambda terminates the MicroVM at maximumDurationInSeconds");
1824
1825        for platform in [Platform::Azure, Platform::Local] {
1826            let error = sandbox
1827                .validate_for_platform(platform)
1828                .expect_err("a deadline nothing applies must be refused");
1829            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1830        }
1831    }
1832
1833    /// A MicroVM bursts to four times its baseline with no way to opt out, so a ceiling is kept
1834    /// by choosing the size whose *peak* fits inside it. Sizing by baseline would hand back a
1835    /// sandbox that can reach four times what the customer declared.
1836    #[test]
1837    fn an_aws_size_is_chosen_so_its_peak_stays_inside_the_declared_ceiling() {
1838        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1839        let tier = sandbox
1840            .microvm_tier()
1841            .expect("2Gi/1cpu/20Gi is satisfiable");
1842
1843        assert_eq!(
1844            tier.peak_memory_mib, 2048,
1845            "the peak is the declared ceiling"
1846        );
1847        assert_eq!(
1848            tier.baseline_memory_mib, 512,
1849            "which is a quarter of it as the baseline"
1850        );
1851        assert!(tier.max_disk_mib <= 20 * 1024);
1852    }
1853
1854    /// AWS allocates one vCPU per 2GB, so a cpu ceiling below what the memory ceiling implies
1855    /// cannot be honoured together with it. Letting cpu choose the size instead would hand back a
1856    /// machine four times smaller than the memory asked for, with nothing to indicate it.
1857    #[test]
1858    fn a_cpu_ceiling_below_what_the_memory_implies_is_refused_not_quietly_downsized() {
1859        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1860        {
1861            let limits = sandbox
1862                .limits
1863                .as_mut()
1864                .expect("the fixture declares limits");
1865            limits.cpu = "1".to_string();
1866            limits.memory = "8Gi".to_string();
1867        }
1868
1869        let error = sandbox
1870            .microvm_tier()
1871            .expect_err("1 cpu and 8Gi cannot both be ceilings on AWS");
1872        assert!(
1873            error.to_string().contains("4 vCPU"),
1874            "the refusal must say what the memory ceiling implies: {error}"
1875        );
1876
1877        sandbox
1878            .limits
1879            .as_mut()
1880            .expect("the fixture declares limits")
1881            .cpu = "4".to_string();
1882        let tier = sandbox.microvm_tier().expect("4 cpu matches 8Gi");
1883        assert_eq!(tier.peak_memory_mib, 8192);
1884    }
1885
1886    /// Below AWS's smallest peak there is no size that holds the ceiling, and rounding up to the
1887    /// nearest one would silently exceed it.
1888    #[test]
1889    fn an_aws_ceiling_smaller_than_any_size_is_refused_rather_than_rounded() {
1890        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1891        sandbox
1892            .limits
1893            .as_mut()
1894            .expect("the fixture declares limits")
1895            .memory = "1Gi".to_string();
1896
1897        let error = sandbox
1898            .validate_for_platform(Platform::Aws)
1899            .expect_err("no MicroVM size peaks at or below 1Gi");
1900        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1901        assert!(
1902            error.to_string().contains("2Gi"),
1903            "the refusal must say what the smallest holdable ceiling is: {error}"
1904        );
1905    }
1906
1907    /// `Source` is a public part of the type that no backend builds: an empty image string
1908    /// schedules a pod that can never run, so the refusal has to happen at plan time and on
1909    /// every platform, not in one emitter.
1910    #[test]
1911    fn source_code_is_refused_everywhere_rather_than_producing_a_broken_manifest() {
1912        let sandbox = Sandbox::new("agent".to_string())
1913            .code(SandboxCode::Source {
1914                src: "./sandbox".to_string(),
1915                toolchain: ToolchainConfig::Docker {
1916                    dockerfile: None,
1917                    build_args: None,
1918                    target: None,
1919                },
1920            })
1921            .egress(SandboxEgress::Deny)
1922            .lifecycle(SandboxLifecyclePolicy {
1923                max_lifetime_seconds: None,
1924                idle_pause_seconds: None,
1925            })
1926            .build();
1927
1928        for platform in [
1929            Platform::Aws,
1930            Platform::Azure,
1931            Platform::Gcp,
1932            Platform::Kubernetes,
1933            Platform::Local,
1934        ] {
1935            let error = sandbox
1936                .validate_for_platform(platform)
1937                .expect_err("no backend builds a sandbox image from source");
1938            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1939            assert!(
1940                error.to_string().contains("code.image"),
1941                "the refusal must say what to write instead: {error}"
1942            );
1943        }
1944    }
1945
1946    /// `validate_quantity` accepts nine suffixes. Reading only `Gi` and `Mi` would size a
1947    /// declared `4G` as though it were `4Gi`, which for a ceiling means exceeding it.
1948    #[test]
1949    fn every_accepted_unit_converts_rather_than_falling_back() {
1950        assert_eq!(quantity_mib("2Gi"), Some(2048));
1951        assert_eq!(quantity_mib("512Mi"), Some(512));
1952        assert_eq!(quantity_mib("4G"), Some(3814));
1953        assert_eq!(quantity_mib("1Ti"), Some(1024 * 1024));
1954        assert_eq!(millicores("1"), Some(1000));
1955        assert_eq!(millicores("500m"), Some(500));
1956    }
1957
1958    #[test]
1959    fn unknown_fields_are_rejected() {
1960        let json = r#"{
1961            "id": "sbx",
1962            "code": {"type": "image", "image": "ubuntu:24.04"},
1963            "limits": {"cpu": "1", "memory": "2Gi", "disk": "20Gi"},
1964            "egress": {"mode": "deny"},
1965            "lifecycle": {},
1966            "unexpected": true
1967        }"#;
1968
1969        serde_json::from_str::<Sandbox>(json).expect_err("deny_unknown_fields must reject");
1970    }
1971
1972    #[test]
1973    fn serialization_roundtrips() {
1974        let sandbox = sandbox_with(
1975            SandboxEgress::AllowDomains {
1976                domains: vec!["example.com".to_string()],
1977            },
1978            vec![8080, 9090],
1979        );
1980
1981        let json = serde_json::to_string(&sandbox).expect("serializes");
1982        let restored: Sandbox = serde_json::from_str(&json).expect("deserializes");
1983        assert_eq!(sandbox, restored);
1984    }
1985
1986    #[test]
1987    fn id_is_immutable_across_updates() {
1988        let original = sandbox_with(SandboxEgress::Deny, vec![]);
1989        let renamed = Sandbox::new("other".to_string())
1990            .code(SandboxCode::Image {
1991                image: "ubuntu".to_string(),
1992            })
1993            .limits(
1994                original
1995                    .limits
1996                    .clone()
1997                    .expect("the fixture declares limits"),
1998            )
1999            .egress(SandboxEgress::Deny)
2000            .lifecycle(SandboxLifecyclePolicy {
2001                max_lifetime_seconds: None,
2002                idle_pause_seconds: None,
2003            })
2004            .build();
2005
2006        original
2007            .validate_update(&original.clone())
2008            .expect("an unchanged config is a valid update");
2009        original
2010            .validate_update(&renamed)
2011            .expect_err("renaming a sandbox is not an update");
2012    }
2013
2014    /// Azure declares an idle-pause policy but not a wall-clock ceiling.
2015    ///
2016    /// The two travel together in `SandboxLifecyclePolicy` and are gated separately on purpose:
2017    /// Azure pauses on idle and has no maximum lifetime, so accepting one and refusing the
2018    /// other is the honest split rather than an inconsistency.
2019    #[test]
2020    fn azure_takes_an_idle_policy_and_still_refuses_a_lifetime_ceiling() {
2021        let with_policy = |lifecycle: SandboxLifecyclePolicy| {
2022            Sandbox::new("sbx".to_string())
2023                .code(SandboxCode::Image {
2024                    image: "ubuntu".to_string(),
2025                })
2026                .egress(SandboxEgress::Allow)
2027                .lifecycle(lifecycle)
2028                .build()
2029                .validate_for_platform(Platform::Azure)
2030        };
2031
2032        with_policy(SandboxLifecyclePolicy {
2033            max_lifetime_seconds: None,
2034            idle_pause_seconds: Some(900),
2035        })
2036        .expect("Azure pauses a sandbox on idle");
2037
2038        let error = with_policy(SandboxLifecyclePolicy {
2039            max_lifetime_seconds: Some(3600),
2040            idle_pause_seconds: None,
2041        })
2042        .expect_err("Azure has no wall-clock ceiling to enforce one with");
2043        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2044        assert!(
2045            error.message.contains("sandboxLifetime"),
2046            "names the capability: {}",
2047            error.message
2048        );
2049    }
2050
2051    /// An allowlist naming nothing is a deny-all wearing an allowlist's label.
2052    ///
2053    /// It renders as a `Deny` default with no rules — the shape the Azure provider adds a
2054    /// catch-all to avoid — and a reader scanning the declaration sees "allowDomains" and reads
2055    /// the opposite of what it does.
2056    #[test]
2057    fn an_allowlist_with_no_domains_is_refused() {
2058        let declared = |domains: Vec<String>| {
2059            Sandbox::new("sbx".to_string())
2060                .code(SandboxCode::Image {
2061                    image: "ubuntu".to_string(),
2062                })
2063                .egress(SandboxEgress::AllowDomains { domains })
2064                .lifecycle(SandboxLifecyclePolicy {
2065                    max_lifetime_seconds: None,
2066                    idle_pause_seconds: None,
2067                })
2068                .build()
2069                .validate_for_platform(Platform::Azure)
2070        };
2071
2072        let error = declared(vec![]).expect_err("an empty allowlist must be refused");
2073        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2074
2075        declared(vec!["api.example.com".to_string()])
2076            .expect("a named domain is what an allowlist is for");
2077    }
2078
2079    /// The two expressible modes map to the boolean; a host list maps to nothing so the caller has
2080    /// to refuse rather than silently pick a side.
2081    #[test]
2082    fn internet_access_switch_maps_only_the_two_expressible_modes() {
2083        assert_eq!(SandboxEgress::Allow.internet_access_switch(), Some(true));
2084        assert_eq!(SandboxEgress::Deny.internet_access_switch(), Some(false));
2085        assert_eq!(
2086            SandboxEgress::AllowDomains {
2087                domains: vec!["api.example.com".to_string()]
2088            }
2089            .internet_access_switch(),
2090            None,
2091            "a host list has no boolean and must not be approximated"
2092        );
2093    }
2094}