Skip to main content

alien_core/resources/
sandbox.rs

1//! Sandbox resource for running untrusted code in an isolated environment.
2//!
3//! The declaration provisions a durable parent, and the application creates and destroys
4//! individual sandboxes through its binding at runtime.
5//!
6//! The capability set differs per platform and is published rather than assumed. Calling an
7//! unsupported capability is a typed error naming both the platform and the capability, so a
8//! portable application can branch on `SandboxCapabilities` before it calls.
9
10use crate::error::{ErrorData, Result};
11use crate::resource::{ResourceDefinition, ResourceOutputsDefinition, ResourceRef, ResourceType};
12use crate::resources::ToolchainConfig;
13use crate::Platform;
14use alien_error::AlienError;
15use bon::Builder;
16use serde::{Deserialize, Serialize};
17use std::any::Any;
18use std::fmt::Debug;
19
20/// Specifies where the sandbox's root filesystem comes from.
21#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
22#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
23#[serde(rename_all = "camelCase", tag = "type")]
24pub enum SandboxCode {
25    /// A prebuilt container image used as the sandbox root filesystem.
26    #[serde(rename_all = "camelCase")]
27    Image {
28        /// Image reference (e.g. `ubuntu:24.04`, `ghcr.io/myorg/sandbox:latest`).
29        ///
30        /// Two backends narrow it in opposite directions: AWS wants an `s3://` bundle, Azure a
31        /// bare catalog name such as `ubuntu`. Each refuses the other's shape while planning.
32        image: String,
33    },
34    /// A Dockerfile `alien build` builds into the sandbox's base image.
35    ///
36    /// AWS only, and docker only: the base image is a root filesystem, not a binary laid on one.
37    /// `alien release` pushes it and the bundle layers the sandbox agent on afterwards.
38    #[serde(rename_all = "camelCase")]
39    Source {
40        /// The source directory to build from
41        src: String,
42        /// Toolchain configuration with type-safe options
43        toolchain: ToolchainConfig,
44    },
45}
46
47/// Hard ceilings enforced on a sandbox.
48///
49/// These are limits, not scheduling requests. Untrusted code does not respect a hint, so every
50/// field is enforced by the platform and a platform that cannot enforce one is rejected at plan
51/// time rather than silently ignoring it.
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
54#[serde(rename_all = "camelCase", deny_unknown_fields)]
55pub struct SandboxLimits {
56    /// CPU ceiling in cores or millicores (e.g. `"1"`, `"500m"`)
57    pub cpu: String,
58    /// Memory ceiling (e.g. `"2Gi"`, `"512Mi"`)
59    pub memory: String,
60    /// Disk ceiling (e.g. `"20Gi"`)
61    pub disk: String,
62    /// Maximum number of processes, which bounds fork bombs.
63    ///
64    /// Optional because only a container runtime has the primitive: Kubernetes sets a pid ceiling
65    /// per node, not per pod, and neither AWS MicroVMs nor Azure sandboxes expose one. Declaring
66    /// it on a platform that cannot apply it is refused at plan time.
67    #[serde(default, skip_serializing_if = "Option::is_none")]
68    pub max_processes: Option<u32>,
69}
70
71/// One of the five sizes a Lambda MicroVM can be built at.
72///
73/// AWS has no ceiling knob: `minimumMemoryInMiB` sets a *baseline* and a running MicroVM bursts
74/// vertically to four times it with no way to opt out. A declared ceiling is therefore honoured by
75/// picking the tier whose **peak** stays inside it, not the tier whose baseline matches it.
76#[derive(Debug, Clone, Copy, PartialEq, Eq)]
77pub struct MicrovmTier {
78    /// What `minimumMemoryInMiB` is set to.
79    pub baseline_memory_mib: i64,
80    /// The most memory the MicroVM can reach, in MiB.
81    pub peak_memory_mib: i64,
82    /// The most vCPU the MicroVM can reach.
83    pub peak_vcpu: u32,
84    /// The most disk the MicroVM can use, in MiB.
85    pub max_disk_mib: i64,
86}
87
88/// The published sizes, smallest first. Baseline memory to vCPU is 2 GB per vCPU, peak is four
89/// times baseline, and disk is fixed per tier rather than independently selectable.
90/// Longest life AWS will run a MicroVM for, from `RunMicrovm`'s `maximumDurationInSeconds`.
91const AWS_MAX_LIFETIME_SECONDS: u32 = 28_800;
92
93/// Azure's sandbox sizing rule, quoted from the data plane's own refusal of an oversized request:
94/// *CPU must be n×250m for n=1..64 (0.25–16 cores); Memory ≤ cores × 2Gi; Disk ≤ cores × 20Gi*.
95const AZURE_CPU_STEP_MILLICORES: i64 = 250;
96const AZURE_MAX_CPU_MILLICORES: i64 = 16_000;
97const AZURE_MEMORY_MIB_PER_CORE: i64 = 2 * 1024;
98const AZURE_DISK_MIB_PER_CORE: i64 = 20 * 1024;
99
100const MICROVM_TIERS: &[MicrovmTier] = &[
101    MicrovmTier {
102        baseline_memory_mib: 512,
103        peak_memory_mib: 2048,
104        peak_vcpu: 1,
105        max_disk_mib: 8192,
106    },
107    MicrovmTier {
108        baseline_memory_mib: 1024,
109        peak_memory_mib: 4096,
110        peak_vcpu: 2,
111        max_disk_mib: 8192,
112    },
113    MicrovmTier {
114        baseline_memory_mib: 2048,
115        peak_memory_mib: 8192,
116        peak_vcpu: 4,
117        max_disk_mib: 8192,
118    },
119    MicrovmTier {
120        baseline_memory_mib: 4096,
121        peak_memory_mib: 16384,
122        peak_vcpu: 8,
123        max_disk_mib: 16384,
124    },
125    MicrovmTier {
126        baseline_memory_mib: 8192,
127        peak_memory_mib: 32768,
128        peak_vcpu: 16,
129        max_disk_mib: 32768,
130    },
131];
132
133/// Outbound network policy for a sandbox.
134#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
135#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
136#[serde(rename_all = "camelCase", tag = "mode")]
137pub enum SandboxEgress {
138    /// No outbound network access.
139    ///
140    /// Routed traffic only. Link-local is not outbound and no backend's egress control reaches
141    /// it, so this is not a boundary against instance metadata.
142    Deny,
143    /// Unrestricted outbound access to the public internet, and none to private ranges or the
144    /// deployment's own network.
145    ///
146    /// Link-local carries the same exception as `Deny`. AWS and Kubernetes deliver both halves.
147    /// Azure and GCP deliver the first only: one matches host patterns and the other is a single
148    /// switch, so neither can name an address range to exclude.
149    Allow,
150    /// Outbound access only to the listed hostnames.
151    ///
152    /// Azure alone expresses it: its egress proxy matches on host pattern. The others filter by
153    /// CIDR or carry a single switch, and both would approximate the list rather than keep it.
154    #[serde(rename_all = "camelCase")]
155    AllowDomains {
156        /// Hostnames the sandbox may reach
157        domains: Vec<String>,
158    },
159}
160
161impl SandboxEgress {
162    /// The single outbound switch for a backend that has no host matcher, or `None` for a mode a
163    /// boolean cannot carry.
164    ///
165    /// `AllowDomains` needs a host list, so it maps to nothing and each caller refuses it in its
166    /// own error naming the sandbox. One source for what a mode means, so a template and a sandbox
167    /// cannot disagree on it.
168    pub fn internet_access_switch(&self) -> Option<bool> {
169        match self {
170            SandboxEgress::Allow => Some(true),
171            SandboxEgress::Deny => Some(false),
172            SandboxEgress::AllowDomains { .. } => None,
173        }
174    }
175}
176
177/// How long a sandbox may live and when it is paused.
178///
179/// Declaration-time ceilings, not per-request values: every sandbox created through this
180/// declaration's binding is held to them, whatever a caller asks for at runtime.
181#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
182#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
183#[serde(rename_all = "camelCase", deny_unknown_fields)]
184pub struct SandboxLifecyclePolicy {
185    /// Wall-clock ceiling on a single sandbox, after which the platform terminates it.
186    ///
187    /// Optional because not every backend has the primitive: Kubernetes has
188    /// `activeDeadlineSeconds` and AWS `maximumDurationInSeconds`, while neither Azure nor Local
189    /// expose one, so declaring a ceiling there is refused at plan time rather than accepted and
190    /// never applied. AWS caps it at 8 hours.
191    #[serde(default, skip_serializing_if = "Option::is_none")]
192    pub max_lifetime_seconds: Option<u32>,
193    /// Idle period after which the sandbox is paused, where the platform supports it
194    #[serde(skip_serializing_if = "Option::is_none")]
195    pub idle_pause_seconds: Option<u32>,
196}
197
198/// What a platform's sandbox backend can actually do.
199///
200/// Published so portable code can branch before calling rather than discovering a gap through
201/// an error. Every field here corresponds to a capability that at least one platform lacks;
202/// create, exec and terminate are the guaranteed floor and are therefore not listed.
203#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
204#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
205#[serde(rename_all = "camelCase", deny_unknown_fields)]
206pub struct SandboxCapabilities {
207    /// Files can be moved in and out of a sandbox
208    pub files: bool,
209    /// A later call can reach a sandbox created by an earlier one
210    pub reconnect: bool,
211    /// A command can be started, polled and cancelled across separate calls, so it outlives the
212    /// one that started it. False where nothing inside the sandbox owns the process in between.
213    pub jobs: bool,
214    /// An authenticated, port-scoped capability to reach a service inside the sandbox
215    pub preview: bool,
216    /// Sandbox state can be paused and resumed
217    pub pause_resume: bool,
218    /// A sandbox's full state can be captured and used to create another
219    pub snapshot: bool,
220    /// Egress can be restricted to a hostname allowlist
221    pub domain_egress_rules: bool,
222    /// Whether a declared `deny` is actually enforced, rather than accepted and dropped
223    pub egress_deny: bool,
224    /// The platform enforces the declared cpu, memory and disk ceilings
225    pub enforced_limits: bool,
226    /// The platform can cap how many processes a sandbox runs
227    pub process_limit: bool,
228    /// The platform terminates a sandbox at a declared wall-clock deadline
229    pub sandbox_lifetime: bool,
230    /// A command runs in its own PID namespace and cannot see or signal the agent's processes.
231    ///
232    /// Only where an agent runs as root. Creating the namespace needs `CAP_SYS_ADMIN`, and the
233    /// Kubernetes sandbox pod drops every capability — which is also what denies `ptrace` by
234    /// construction, so granting it there would remove a lock to add one.
235    pub supervisor_pid_namespace: bool,
236    /// The process supervising a command is a different identity from the command.
237    ///
238    /// False where a command runs as the agent's own user: it can then read the supervisor's
239    /// environment and signal it. Separate from `supervisorPidNamespace`, which is about
240    /// visibility rather than identity — a backend can have one without the other.
241    pub supervisor_isolation: bool,
242}
243
244impl SandboxCapabilities {
245    /// Returns what the given platform's sandbox backend supports.
246    ///
247    /// Errors for platforms with no sandbox backend, rather than returning an all-false set —
248    /// "every capability is missing" and "this platform has no sandboxes" are different
249    /// conditions and an application should not have to tell them apart by inspection.
250    pub fn for_platform(platform: Platform) -> Result<Self> {
251        match platform {
252            Platform::Aws => Ok(Self {
253                files: true,
254                reconnect: true,
255                jobs: true,
256                preview: true,
257                pause_resume: true,
258                snapshot: false,
259                domain_egress_rules: false,
260                egress_deny: true,
261                enforced_limits: true,
262                // Nothing in the API bounds process count.
263                process_limit: false,
264                // `maximumDurationInSeconds` on `RunMicrovm`, which Lambda enforces by
265                // terminating the MicroVM. Capped at 8 hours by the service.
266                sandbox_lifetime: true,
267                // Measured, not assumed: the agent inside a Lambda MicroVM runs as uid 0 with
268                // `CapEff: 00000000a80425fb`, the standard container default set, which excludes
269                // `CAP_SYS_ADMIN`. It can drop privilege (`CAP_SETUID`/`CAP_SETGID` are held) and
270                // it cannot create a namespace. No backend offers this today.
271                supervisor_pid_namespace: false,
272                // The agent runs as uid 0 and `setuid`s the command to uid 60000, so the command
273                // runs under a different identity than the process supervising it.
274                supervisor_isolation: true,
275            }),
276            Platform::Azure => Ok(Self::azure()),
277            Platform::Gcp => Ok(Self::gcp_agent_platform()),
278            // Preview needs a gateway that validates a sandbox-and-port capability, which this
279            // backend has none of.
280            Platform::Kubernetes => Ok(Self {
281                files: true,
282                reconnect: true,
283                jobs: true,
284                preview: false,
285                pause_resume: false,
286                snapshot: false,
287                domain_egress_rules: false,
288                egress_deny: true,
289                enforced_limits: true,
290                // A pid ceiling is a kubelet setting per node, not a pod field.
291                process_limit: false,
292                // `activeDeadlineSeconds` on the pod, which the kubelet enforces.
293                sandbox_lifetime: true,
294                // The pod drops every capability, including the `CAP_SYS_ADMIN` the agent would
295                // need to unshare. That is also what denies `ptrace`, so this stays false rather
296                // than the pod being weakened to make it true.
297                supervisor_pid_namespace: false,
298                // The pod pins one uid (`run_as_user: 65534` on both pod and container) with
299                // `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no
300                // process can setuid to split the command off from a supervisor. No uid split is
301                // possible, so none exists.
302                supervisor_isolation: false,
303            }),
304            Platform::Local => Ok(Self {
305                files: true,
306                reconnect: true,
307                // Nothing runs inside the sandbox: the manager drives Docker from outside it.
308                jobs: false,
309                preview: true,
310                pause_resume: false,
311                snapshot: false,
312                domain_egress_rules: false,
313                egress_deny: true,
314                enforced_limits: true,
315                // Docker's `--pids-limit`.
316                process_limit: true,
317                sandbox_lifetime: false,
318                // Local has no in-sandbox agent: the manager drives Docker from outside, so
319                // there is no supervisor inside the sandbox to isolate from.
320                supervisor_pid_namespace: false,
321                // The supervisor is the manager on the host, outside the container entirely, and
322                // `docker exec` runs the command as the workload uid — a different identity by
323                // construction.
324                supervisor_isolation: true,
325            }),
326            Platform::Machines | Platform::Test => {
327                Err(AlienError::new(ErrorData::SandboxPlatformUnsupported {
328                    platform: platform.to_string(),
329                }))
330            }
331        }
332    }
333
334    /// What the Azure sandbox backend supports; the body of the `Platform::Azure` arm.
335    pub fn azure() -> Self {
336        Self {
337            files: true,
338            reconnect: true,
339            // No Alien process runs inside the sandbox to own a command between two calls.
340            jobs: false,
341            // A sandbox port carries a URL and an auth config, and the auth config offers two
342            // things: anonymous, or Entra ID with an allowlist of human email addresses.
343            // Neither is a credential scoped to a port for a fixed time, which is what a
344            // preview capability is. Returning the anonymous URL would publish the port.
345            preview: false,
346            pause_resume: true,
347            // False for a client reason, not a cloud one: this client has no snapshot call, and
348            // `CreateSandboxRequest` has no field to consume the id it would return. Also
349            // unclaimed: Microsoft does not garbage-collect snapshots, so an id is a bill that grows.
350            snapshot: false,
351            domain_egress_rules: true,
352            egress_deny: true,
353            // Enforced inside the sandbox, not at create: an over-allocation raises `MemoryError`
354            // while the sandbox keeps running. `azure_sandbox_limits` checks the continuous
355            // sizing rule at plan time instead of matching a tier.
356            enforced_limits: true,
357            process_limit: false,
358            // Auto-suspend and auto-delete exist; a wall-clock ceiling does not. Accepting
359            // `maxLifetimeSeconds` here would be the silent no-op the capability set exists
360            // to prevent, so this is a decision rather than a gap.
361            sandbox_lifetime: false,
362            // No Alien process inside an Azure sandbox, so there is no supervisor to isolate.
363            supervisor_pid_namespace: false,
364            // No Alien process runs the command at all — the platform's own data plane does,
365            // so there is no separate supervisor identity to speak of.
366            supervisor_isolation: false,
367        }
368    }
369
370    /// What the GCP Agent Platform sandbox backend supports; the body of the `Platform::Gcp` arm.
371    pub fn gcp_agent_platform() -> Self {
372        Self {
373            // Agent file operations move over the sandbox envelope.
374            files: true,
375            // Reaching a sandbox across processes is safe because `generation` is derived from the
376            // container boot id read through the agent's health op, so a caller detects a container
377            // replaced under a stable sandbox name rather than reconnecting to a blank one.
378            reconnect: true,
379            jobs: true,
380            // No method mints a port-scoped ingress capability; the only ingress is `:execute`.
381            preview: false,
382            // `:pause` and `:resume` preserve the running container.
383            pause_resume: true,
384            // The create path never sends `sandbox_environment_snapshot`, so no sandbox state is
385            // reachable through the trait; declared false until the client carries it.
386            snapshot: false,
387            // Egress is shaped by VPC and DNS peering, which is not a hostname allowlist.
388            domain_egress_rules: false,
389            // A declared `deny` blocks both routed egress and DNS.
390            egress_deny: true,
391            // The declared ceilings are enforced, but by terminating the sandbox on breach rather
392            // than by refusing the allocation — a caller reading `true` should expect the sandbox
393            // to die, not a clean error at the point of the request.
394            enforced_limits: true,
395            // No ceiling on process count is observed.
396            process_limit: false,
397            // `ttl` maps to a sandbox `expireTime` the platform terminates at.
398            sandbox_lifetime: true,
399            // No PID-namespace isolation between the command and anything supervising it.
400            supervisor_pid_namespace: false,
401            // No separate supervisor identity: the command is not run under a different identity
402            // than the process supervising it.
403            supervisor_isolation: false,
404        }
405    }
406
407    /// Returns a typed error if the named capability is absent on this platform.
408    pub fn require(&self, capability: SandboxCapability, platform: Platform) -> Result<()> {
409        let available = match capability {
410            SandboxCapability::Files => self.files,
411            SandboxCapability::Reconnect => self.reconnect,
412            SandboxCapability::Jobs => self.jobs,
413            SandboxCapability::Preview => self.preview,
414            SandboxCapability::PauseResume => self.pause_resume,
415            SandboxCapability::Snapshot => self.snapshot,
416            SandboxCapability::DomainEgressRules => self.domain_egress_rules,
417            SandboxCapability::EgressDeny => self.egress_deny,
418            SandboxCapability::EnforcedLimits => self.enforced_limits,
419            SandboxCapability::ProcessLimit => self.process_limit,
420            SandboxCapability::SandboxLifetime => self.sandbox_lifetime,
421            SandboxCapability::SupervisorPidNamespace => self.supervisor_pid_namespace,
422            SandboxCapability::SupervisorIsolation => self.supervisor_isolation,
423        };
424
425        if available {
426            return Ok(());
427        }
428
429        Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
430            capability: capability.as_str().to_string(),
431            platform: platform.to_string(),
432        }))
433    }
434}
435
436/// Names a single sandbox capability, so an unsupported call can report which one it needed.
437#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
438#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
439#[serde(rename_all = "camelCase")]
440pub enum SandboxCapability {
441    /// Moving files in and out of a sandbox
442    Files,
443    /// Reaching a sandbox created by an earlier call
444    Reconnect,
445    /// Starting, polling and cancelling a command across separate calls
446    Jobs,
447    /// An authenticated, port-scoped ingress capability
448    Preview,
449    /// Pausing and resuming sandbox state
450    PauseResume,
451    /// Capturing full sandbox state
452    Snapshot,
453    /// Restricting egress to a hostname allowlist
454    DomainEgressRules,
455    /// Refusing outbound access when a sandbox declares none
456    EgressDeny,
457    /// Platform-enforced resource ceilings
458    EnforcedLimits,
459    /// A ceiling on the number of processes a sandbox may run
460    ProcessLimit,
461    /// A wall-clock ceiling on a sandbox, applied by the platform rather than by a caller
462    SandboxLifetime,
463    /// A command runs in its own PID namespace, isolated from the agent supervising it
464    SupervisorPidNamespace,
465    /// A command runs under a different identity than the process supervising it
466    SupervisorIsolation,
467}
468
469impl SandboxCapability {
470    /// Returns the stable identifier used in errors and capability queries.
471    pub fn as_str(&self) -> &'static str {
472        match self {
473            Self::Files => "files",
474            Self::Reconnect => "reconnect",
475            Self::Jobs => "jobs",
476            Self::Preview => "preview",
477            Self::PauseResume => "pauseResume",
478            Self::Snapshot => "snapshot",
479            Self::DomainEgressRules => "domainEgressRules",
480            Self::EgressDeny => "egressDeny",
481            Self::EnforcedLimits => "enforcedLimits",
482            Self::ProcessLimit => "processLimit",
483            Self::SandboxLifetime => "sandboxLifetime",
484            Self::SupervisorPidNamespace => "supervisorPidNamespace",
485            Self::SupervisorIsolation => "supervisorIsolation",
486        }
487    }
488}
489
490/// An isolated environment for running untrusted code, created at runtime.
491#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Builder)]
492#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
493#[serde(rename_all = "camelCase", deny_unknown_fields)]
494#[builder(start_fn = new)]
495pub struct Sandbox {
496    /// Identifier for the sandbox. Must contain only alphanumeric characters, hyphens, and
497    /// underscores ([A-Za-z0-9-_]). Maximum 64 characters.
498    #[builder(start_fn)]
499    pub id: String,
500    /// Where the sandbox's root filesystem comes from
501    pub code: SandboxCode,
502    /// Private ECR base image an AWS build pulls; `code.image` names only the S3 bundle, so this is
503    /// what the cross-account grant opens. Live only: the grant needs the customer account,
504    /// which registration reports. Absent means the base image is pulled anonymously.
505    #[serde(skip_serializing_if = "Option::is_none")]
506    pub private_base_image: Option<String>,
507    /// Enforced resource ceilings.
508    ///
509    /// Optional because not every platform can enforce them, and a declaration that names none
510    /// takes the platform's own defaults. Naming them on a platform that cannot enforce them is
511    /// rejected at plan time rather than silently ignored.
512    #[serde(skip_serializing_if = "Option::is_none")]
513    pub limits: Option<SandboxLimits>,
514    /// Outbound network policy
515    pub egress: SandboxEgress,
516    /// Sandbox lifetime ceiling and idle behaviour
517    pub lifecycle: SandboxLifecyclePolicy,
518    /// Ports eligible for a preview capability. An application reaches its sandbox through the
519    /// provider, so it cannot widen its own ingress at runtime; a holder of a remote binding's
520    /// credentials is bounded by no port condition, which is why a remote sandbox declares none.
521    #[builder(default)]
522    #[serde(default, skip_serializing_if = "Vec::is_empty")]
523    pub preview_ports: Vec<u16>,
524}
525
526/// Whether the artifact being rendered restricts which network modes it accepts.
527///
528/// Cloud setup needs explicit subnets for restricted sandbox connectors. Kubernetes targets do
529/// not emit these cloud backends.
530pub fn restricts_network_mode(stack: &crate::Stack, targets_kubernetes: bool) -> bool {
531    !targets_kubernetes && stack_needs_named_subnets_at_setup(stack)
532}
533
534/// Whether any setup-owned resource forces setup to name subnets.
535///
536/// Restricted sandbox connectors require subnet IDs. Callers rendering an artifact want
537/// [`restricts_network_mode`] instead:
538/// this one answers for the declaration, which on a Kubernetes target is not what gets emitted.
539pub fn stack_needs_named_subnets_at_setup(stack: &crate::Stack) -> bool {
540    stack.resources().any(|(_resource_id, resource)| {
541        resource
542            .config
543            .downcast_ref::<Sandbox>()
544            .is_some_and(|sandbox| !matches!(sandbox.egress, SandboxEgress::Allow))
545    })
546}
547
548impl Sandbox {
549    /// The resource type identifier for Sandbox
550    pub const RESOURCE_TYPE: ResourceType = ResourceType::from_static("sandbox");
551
552    /// Returns the sandbox's unique identifier.
553    pub fn id(&self) -> &str {
554        &self.id
555    }
556
557    /// The declared ceilings, or the defaults a platform applies when none were named.
558    ///
559    /// Backends want a concrete set: a sandbox with no declared ceilings still runs inside
560    /// whatever the platform gives it, and a backend that had to branch on `None` would end up
561    /// inventing its own default anyway.
562    pub fn resolved_limits(&self) -> SandboxLimits {
563        self.limits.clone().unwrap_or_else(default_limits)
564    }
565
566    /// Validates the declaration against what the target platform can enforce.
567    ///
568    /// Runs at plan time so an unenforceable limit or an unsupported egress mode fails before
569    /// anything is provisioned, rather than at the first exec.
570    pub fn validate_for_platform(&self, platform: Platform) -> Result<()> {
571        let capabilities = SandboxCapabilities::for_platform(platform)?;
572
573        // `alien build` builds an AWS sandbox's base image, so source is a declaration there and
574        // the emitters refuse it only if it reaches them unbuilt. Everywhere else the image is
575        // pulled as declared, and an empty image string would schedule a pod that can never run.
576        if matches!(&self.code, SandboxCode::Source { .. }) && platform != Platform::Aws {
577            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
578                resource_id: self.id.clone(),
579                field: "code".to_string(),
580                value: "source".to_string(),
581                reason: format!(
582                    "no sandbox backend builds an image from source on {platform}; give \
583                     code.image a prebuilt reference"
584                ),
585            }));
586        }
587
588        // Elsewhere `code.image` is pulled directly, so a second reference would be a grant
589        // nothing reads.
590        if self.private_base_image.is_some() && platform != Platform::Aws {
591            return Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
592                capability: "privateBaseImage".to_string(),
593                platform: platform.to_string(),
594            }));
595        }
596
597        // Read before the limits, because the image is declared whether or not any are.
598        if platform == Platform::Azure {
599            self.azure_catalog_image()?;
600        }
601
602        let Some(limits) = self.limits.as_ref() else {
603            // Nothing declared, so nothing to enforce and nothing to reject.
604            return self.validate_capabilities(&capabilities, platform);
605        };
606
607        validate_quantity(&self.id, "cpu", &limits.cpu)?;
608        validate_quantity(&self.id, "memory", &limits.memory)?;
609        validate_quantity(&self.id, "disk", &limits.disk)?;
610
611        if let Some(max_processes) = limits.max_processes {
612            if max_processes == 0 {
613                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
614                    resource_id: self.id.clone(),
615                    field: "maxProcesses".to_string(),
616                    value: "0".to_string(),
617                    reason: "a sandbox that may run no processes cannot run code".to_string(),
618                }));
619            }
620            capabilities.require(SandboxCapability::ProcessLimit, platform)?;
621        }
622
623        // Declaring limits a platform ignores is worse than not declaring them: the stack reads
624        // as bounded while the sandbox is not.
625        capabilities.require(SandboxCapability::EnforcedLimits, platform)?;
626
627        if platform == Platform::Azure {
628            self.azure_sandbox_limits()?;
629        }
630
631        if platform == Platform::Aws {
632            // Refused here rather than at emit so a customer sees it while planning, and so both
633            // package formats inherit the same answer.
634            self.microvm_tier()?;
635
636            // The ceiling is Lambda's, and it rejects the run rather than clamping — so a value
637            // outside it would pass planning, render into the package, and fail at the first
638            // sandbox. Kubernetes takes the same field with no such bound, which is why this
639            // sits under the AWS gate rather than on the type.
640            if let Some(seconds) = self.lifecycle.max_lifetime_seconds {
641                if !(1..=AWS_MAX_LIFETIME_SECONDS).contains(&seconds) {
642                    return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
643                        resource_id: self.id.clone(),
644                        field: "maxLifetimeSeconds".to_string(),
645                        value: seconds.to_string(),
646                        reason: format!(
647                            "AWS runs a MicroVM for between 1 and \
648                             {AWS_MAX_LIFETIME_SECONDS} seconds"
649                        ),
650                    }));
651                }
652            }
653        }
654
655        self.validate_capabilities(&capabilities, platform)
656    }
657
658    /// The catalog disk image Azure creates a sandbox from.
659    ///
660    /// Azure names a public catalog entry rather than pulling a reference, so a registry path,
661    /// tag or digest has nowhere to go. An allowlist, because the answer to "what else could be
662    /// in there" is a name the data plane rejects at the first sandbox, long after the apply.
663    pub fn azure_catalog_image(&self) -> Result<&str> {
664        let refused = |value: &str, reason: &str| {
665            AlienError::new(ErrorData::SandboxLimitInvalid {
666                resource_id: self.id.clone(),
667                field: "code.image".to_string(),
668                value: value.to_string(),
669                reason: reason.to_string(),
670            })
671        };
672
673        let SandboxCode::Image { image } = &self.code else {
674            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
675                resource_id: self.id.clone(),
676                field: "code".to_string(),
677                value: "source".to_string(),
678                reason: "no sandbox backend builds an image from source yet".to_string(),
679            }));
680        };
681
682        let image = image.trim();
683        if image.is_empty() {
684            return Err(refused(image, "a sandbox has to name an image"));
685        }
686        if !image
687            .chars()
688            .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '-'))
689        {
690            return Err(refused(
691                image,
692                "Azure creates a sandbox from a public catalog disk image, so code.image must be \
693                 a bare catalog name such as 'ubuntu'",
694            ));
695        }
696        Ok(image)
697    }
698
699    /// Checks the declared ceilings against Azure's sizing rule (the `AZURE_*` constants above).
700    /// Refused at plan time, like [`Self::microvm_tier`], so a bad value is a declaration to fix
701    /// rather than a runtime fault at create.
702    pub fn azure_sandbox_limits(&self) -> Result<()> {
703        let Some(limits) = self.limits.as_ref() else {
704            // Nothing declared means the binding substitutes Alien's own default sizing, which is
705            // inside the rule — asserted where those constants live, since this cannot see them.
706            return Ok(());
707        };
708
709        let refused = |field: &str, value: &str, reason: &str| {
710            AlienError::new(ErrorData::SandboxLimitInvalid {
711                resource_id: self.id.clone(),
712                field: field.to_string(),
713                value: value.to_string(),
714                reason: reason.to_string(),
715            })
716        };
717
718        let cpu_millicores = millicores(&limits.cpu)
719            .ok_or_else(|| refused("cpu", &limits.cpu, "expected cores or millicores"))?;
720
721        // The multiple is checked, not just the range: `333m` sits inside 0.25–16 cores and is
722        // still refused on the wire, so a bounds-only check would pass a declaration that fails
723        // at create.
724        if cpu_millicores % AZURE_CPU_STEP_MILLICORES != 0
725            || !(AZURE_CPU_STEP_MILLICORES..=AZURE_MAX_CPU_MILLICORES).contains(&cpu_millicores)
726        {
727            return Err(refused(
728                "cpu",
729                &limits.cpu,
730                "Azure allocates cpu in steps of 250m from 250m to 16000m",
731            ));
732        }
733
734        // Both ceilings are derived from the cpu, so they cannot be checked before it is known.
735        let memory_ceiling_mib = cpu_millicores * AZURE_MEMORY_MIB_PER_CORE / 1000;
736        let disk_ceiling_mib = cpu_millicores * AZURE_DISK_MIB_PER_CORE / 1000;
737
738        let memory_mib = quantity_mib(&limits.memory)
739            .ok_or_else(|| refused("memory", &limits.memory, "Azure sizes memory in whole MiB"))?;
740        if memory_mib > memory_ceiling_mib {
741            return Err(refused(
742                "memory",
743                &limits.memory,
744                &format!(
745                    "Azure allows at most 2Gi of memory per core, or {memory_ceiling_mib}Mi \
746                          at the declared cpu"
747                ),
748            ));
749        }
750
751        let disk_mib = quantity_mib(&limits.disk)
752            .ok_or_else(|| refused("disk", &limits.disk, "Azure sizes disk in whole MiB"))?;
753        if disk_mib > disk_ceiling_mib {
754            return Err(refused(
755                "disk",
756                &limits.disk,
757                &format!(
758                    "Azure allows at most 20Gi of disk per core, or {disk_ceiling_mib}Mi at \
759                          the declared cpu"
760                ),
761            ));
762        }
763
764        Ok(())
765    }
766
767    /// The MicroVM size that keeps every declared ceiling, or why none does.
768    ///
769    /// AWS sizes are discrete and a running MicroVM bursts to four times its baseline, so the
770    /// only tier that honours a ceiling is one whose peak fits inside it. A declaration no tier
771    /// satisfies is refused: shipping the nearest size would give the customer a sandbox that
772    /// exceeds the bound they wrote down.
773    pub fn microvm_tier(&self) -> Result<MicrovmTier> {
774        let Some(limits) = self.limits.as_ref() else {
775            // Nothing declared: AWS's own default baseline, which is also `default_limits`.
776            return Ok(MICROVM_TIERS[2]);
777        };
778
779        let memory_mib = quantity_mib(&limits.memory).ok_or_else(|| {
780            AlienError::new(ErrorData::SandboxLimitInvalid {
781                resource_id: self.id.clone(),
782                field: "memory".to_string(),
783                value: limits.memory.clone(),
784                reason: "AWS sizes a MicroVM in whole MiB".to_string(),
785            })
786        })?;
787        let disk_mib = quantity_mib(&limits.disk).ok_or_else(|| {
788            AlienError::new(ErrorData::SandboxLimitInvalid {
789                resource_id: self.id.clone(),
790                field: "disk".to_string(),
791                value: limits.disk.clone(),
792                reason: "AWS sizes a MicroVM's disk in whole MiB".to_string(),
793            })
794        })?;
795        let cpu_millicores = millicores(&limits.cpu).ok_or_else(|| {
796            AlienError::new(ErrorData::SandboxLimitInvalid {
797                resource_id: self.id.clone(),
798                field: "cpu".to_string(),
799                value: limits.cpu.clone(),
800                reason: "expected cores or millicores".to_string(),
801            })
802        })?;
803
804        // Memory and disk choose the size; cpu is then checked rather than used to choose.
805        // AWS couples cpu to memory at 2 GB per vCPU, so letting a low cpu ceiling select the
806        // size too would quietly hand back a machine four times smaller than the memory ceiling
807        // asked for, with nothing to indicate it.
808        let sized = |tier: &&MicrovmTier| {
809            tier.peak_memory_mib <= memory_mib && tier.max_disk_mib <= disk_mib
810        };
811
812        let tier = MICROVM_TIERS
813            .iter()
814            .rev()
815            .find(sized)
816            .copied()
817            .ok_or_else(|| {
818                AlienError::new(ErrorData::SandboxLimitInvalid {
819                    resource_id: self.id.clone(),
820                    field: "memory".to_string(),
821                    value: limits.memory.clone(),
822                    reason: format!(
823                        "a Lambda MicroVM bursts to four times its baseline, so the smallest \
824                         ceiling AWS can hold is 2Gi memory with 8Gi disk; '{}' memory and '{}' \
825                         disk fit no size",
826                        limits.memory, limits.disk
827                    ),
828                })
829            })?;
830
831        let required_millicores = i64::from(tier.peak_vcpu) * 1000;
832        if cpu_millicores < required_millicores {
833            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
834                resource_id: self.id.clone(),
835                field: "cpu".to_string(),
836                value: limits.cpu.clone(),
837                reason: format!(
838                    "AWS allocates one vCPU per 2GB, so a MicroVM sized to a '{}' memory ceiling \
839                     reaches {} vCPU; declare cpu '{}' or lower the memory ceiling",
840                    limits.memory, tier.peak_vcpu, tier.peak_vcpu
841                ),
842            }));
843        }
844
845        Ok(tier)
846    }
847
848    /// The capability checks that do not depend on declared limits.
849    fn validate_capabilities(
850        &self,
851        capabilities: &SandboxCapabilities,
852        platform: Platform,
853    ) -> Result<()> {
854        if matches!(self.egress, SandboxEgress::AllowDomains { .. }) {
855            capabilities.require(SandboxCapability::DomainEgressRules, platform)?;
856        }
857
858        // `allow` asks for no restriction, so a backend that ignores it fails loudly on the first
859        // blocked connection. `deny` asks for one, and a backend that ignores it puts untrusted
860        // code on the internet with nothing to notice — so only this direction is gated.
861        // An empty list is not a restriction anyone wrote down: it renders as a deny-all wearing
862        // an allowlist's label, which reads at a glance as the opposite of what it does.
863        if let SandboxEgress::AllowDomains { domains } = &self.egress {
864            if domains.is_empty() {
865                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
866                    resource_id: self.id.clone(),
867                    field: "egress.domains".to_string(),
868                    value: "[]".to_string(),
869                    reason: "an allowlist naming no domain denies everything; declare \
870                             egress: deny if that is what was meant"
871                        .to_string(),
872                }));
873            }
874        }
875
876        if matches!(self.egress, SandboxEgress::Deny) {
877            capabilities.require(SandboxCapability::EgressDeny, platform)?;
878        }
879
880        if !self.preview_ports.is_empty() {
881            capabilities.require(SandboxCapability::Preview, platform)?;
882        }
883
884        if self.lifecycle.idle_pause_seconds.is_some() {
885            capabilities.require(SandboxCapability::PauseResume, platform)?;
886        }
887
888        if self.lifecycle.max_lifetime_seconds.is_some() {
889            capabilities.require(SandboxCapability::SandboxLifetime, platform)?;
890        }
891
892        Ok(())
893    }
894}
895
896/// Ceilings applied when a declaration names none.
897///
898/// Modest on purpose: an undeclared sandbox is one whose author did not think about sizing, and
899/// the safe reading of that is a small box rather than a generous one.
900fn default_limits() -> SandboxLimits {
901    SandboxLimits {
902        cpu: "1".to_string(),
903        memory: "2Gi".to_string(),
904        disk: "8Gi".to_string(),
905        max_processes: None,
906    }
907}
908
909/// Validates a Kubernetes-style resource quantity such as `500m`, `2Gi` or `1`.
910fn validate_quantity(resource_id: &str, field: &str, value: &str) -> Result<()> {
911    let invalid = |reason: &str| {
912        AlienError::new(ErrorData::SandboxLimitInvalid {
913            resource_id: resource_id.to_string(),
914            field: field.to_string(),
915            value: value.to_string(),
916            reason: reason.to_string(),
917        })
918    };
919
920    let digits_end = value
921        .find(|c: char| !c.is_ascii_digit() && c != '.')
922        .unwrap_or(value.len());
923    let (number, suffix) = value.split_at(digits_end);
924
925    let parsed: f64 = number
926        .parse()
927        .map_err(|_| invalid("expected a number, optionally followed by a unit suffix"))?;
928
929    if parsed <= 0.0 {
930        return Err(invalid("must be greater than zero"));
931    }
932
933    const SUFFIXES: &[&str] = &["", "m", "k", "M", "G", "T", "Ki", "Mi", "Gi", "Ti"];
934    if !SUFFIXES.contains(&suffix) {
935        return Err(invalid(
936            "unit must be one of m, k, M, G, T, Ki, Mi, Gi, Ti, or absent",
937        ));
938    }
939
940    Ok(())
941}
942
943/// Splits a quantity into its number and unit suffix.
944fn split_quantity(value: &str) -> Option<(f64, &str)> {
945    let trimmed = value.trim();
946    let digits_end = trimmed
947        .find(|c: char| !c.is_ascii_digit() && c != '.')
948        .unwrap_or(trimmed.len());
949    let (number, suffix) = trimmed.split_at(digits_end);
950    number.parse().ok().map(|number| (number, suffix))
951}
952
953/// A memory or disk quantity in whole MiB, rounded down.
954///
955/// Every suffix `validate_quantity` accepts is handled here. Reading only `Gi` and `Mi` and
956/// falling back for the rest would turn a declared `4G` into a different size than the customer
957/// asked for, which for a ceiling means a sandbox larger than its bound.
958pub fn quantity_mib(value: &str) -> Option<i64> {
959    let (number, suffix) = split_quantity(value)?;
960    let bytes = match suffix {
961        "" => number,
962        "k" => number * 1e3,
963        "M" => number * 1e6,
964        "G" => number * 1e9,
965        "T" => number * 1e12,
966        "Ki" => number * 1024.0,
967        "Mi" => number * 1024.0 * 1024.0,
968        "Gi" => number * 1024.0 * 1024.0 * 1024.0,
969        "Ti" => number * 1024.0 * 1024.0 * 1024.0 * 1024.0,
970        // `m` is a millicore suffix; memory has no use for it.
971        _ => return None,
972    };
973    Some((bytes / (1024.0 * 1024.0)) as i64)
974}
975
976/// A CPU quantity in millicores.
977pub fn millicores(value: &str) -> Option<i64> {
978    let (number, suffix) = split_quantity(value)?;
979    match suffix {
980        "" => Some((number * 1000.0) as i64),
981        "m" => Some(number as i64),
982        _ => None,
983    }
984}
985
986/// Outputs generated by a successfully provisioned Sandbox parent.
987#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
988#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
989#[serde(rename_all = "camelCase")]
990pub struct SandboxOutputs {
991    /// Name of the durable parent that sandboxes are created inside
992    pub parent_name: String,
993    /// Platform-specific identifier for the parent (image ARN, sandbox group id, namespace)
994    #[serde(skip_serializing_if = "Option::is_none")]
995    pub identifier: Option<String>,
996    /// Data-plane endpoint sandboxes are created through, where the platform has one
997    #[serde(skip_serializing_if = "Option::is_none")]
998    pub endpoint: Option<String>,
999}
1000
1001impl ResourceOutputsDefinition for SandboxOutputs {
1002    fn get_resource_type(&self) -> ResourceType {
1003        Sandbox::RESOURCE_TYPE
1004    }
1005
1006    fn as_any(&self) -> &dyn Any {
1007        self
1008    }
1009
1010    fn box_clone(&self) -> Box<dyn ResourceOutputsDefinition> {
1011        Box::new(self.clone())
1012    }
1013
1014    fn outputs_eq(&self, other: &dyn ResourceOutputsDefinition) -> bool {
1015        other.as_any().downcast_ref::<SandboxOutputs>() == Some(self)
1016    }
1017
1018    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1019        serde_json::to_value(self)
1020    }
1021}
1022
1023impl ResourceDefinition for Sandbox {
1024    fn get_resource_type(&self) -> ResourceType {
1025        Self::RESOURCE_TYPE
1026    }
1027
1028    fn id(&self) -> &str {
1029        &self.id
1030    }
1031
1032    fn get_dependencies(&self) -> Vec<ResourceRef> {
1033        Vec::new()
1034    }
1035
1036    fn validate_update(&self, new_config: &dyn ResourceDefinition) -> Result<()> {
1037        let new_sandbox = new_config
1038            .as_any()
1039            .downcast_ref::<Sandbox>()
1040            .ok_or_else(|| {
1041                AlienError::new(ErrorData::UnexpectedResourceType {
1042                    resource_id: self.id.clone(),
1043                    expected: Self::RESOURCE_TYPE,
1044                    actual: new_config.get_resource_type(),
1045                })
1046            })?;
1047
1048        if self.id != new_sandbox.id {
1049            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
1050                resource_id: self.id.clone(),
1051                reason: "the 'id' field is immutable".to_string(),
1052            }));
1053        }
1054
1055        Ok(())
1056    }
1057
1058    fn as_any(&self) -> &dyn Any {
1059        self
1060    }
1061
1062    fn as_any_mut(&mut self) -> &mut dyn Any {
1063        self
1064    }
1065
1066    fn box_clone(&self) -> Box<dyn ResourceDefinition> {
1067        Box::new(self.clone())
1068    }
1069
1070    fn resource_eq(&self, other: &dyn ResourceDefinition) -> bool {
1071        other.as_any().downcast_ref::<Sandbox>() == Some(self)
1072    }
1073
1074    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1075        serde_json::to_value(self)
1076    }
1077}
1078
1079/// The one token a sandbox bundle URI may carry, replaced with the deploying region.
1080///
1081/// AWS builds a MicroVM image only from a bucket in the image's own region, so a vendor
1082/// publishing to every supported region needs one stored URI that resolves per region.
1083pub const BUNDLE_REGION_TOKEN: &str = "{region}";
1084
1085/// A bundle URI split around its region token, or carried whole when it has none.
1086#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1087pub enum BundleUri<'a> {
1088    /// No token: emitted exactly as it is today.
1089    Literal(&'a str),
1090    /// The text either side of the token, for an emitter to rejoin around its own region
1091    /// expression.
1092    Regional { before: &'a str, after: &'a str },
1093}
1094
1095/// The prefix a rebuild's new key still sits under: everything above the file name and the
1096/// version segment beneath it. `None` when nothing sits there, meaning no prefix can be granted
1097/// without also granting objects a rebuild never reads. Shared so both emitters agree on it.
1098pub fn stable_bundle_key_prefix(key: &str) -> Option<&str> {
1099    let (above_file, _) = key.rsplit_once('/')?;
1100    let (above_version, _) = above_file.rsplit_once('/')?;
1101    Some(above_version)
1102}
1103
1104/// Reads a sandbox bundle URI, refusing anything an image build would only reject later.
1105///
1106/// The token is accepted in the bucket alone. A key-position token would name an object that does
1107/// not exist, and any other brace is a typo that would otherwise reach S3 verbatim and fail ~160s
1108/// into the build — which is the failure this whole check exists to move to plan time.
1109pub fn parse_bundle_uri(uri: &str) -> std::result::Result<BundleUri<'_>, String> {
1110    let path = uri
1111        .strip_prefix("s3://")
1112        .ok_or_else(|| format!("'{uri}' is not an s3:// URI"))?;
1113    let (bucket, key) = path
1114        .split_once('/')
1115        .ok_or_else(|| format!("'{uri}' names a bucket with no object key"))?;
1116
1117    // Both emitters interpolate this path into the build role's resource ARN, where `*` and `?`
1118    // are IAM wildcards rather than literal characters. S3 accepts them in a key, so a bundle
1119    // published under one would silently widen the grant past the bundle it names.
1120    if path.contains('*') || path.contains('?') {
1121        return Err(format!(
1122            "'{uri}' carries an IAM wildcard; the bundle's path is interpolated into the build \
1123             role's grant, so '*' and '?' would widen it past the bundle"
1124        ));
1125    }
1126
1127    if key.contains('{') || key.contains('}') {
1128        return Err(format!(
1129            "'{uri}' places a token in the object key; {BUNDLE_REGION_TOKEN} is accepted in the \
1130             bucket name alone"
1131        ));
1132    }
1133
1134    let Some((before, after)) = bucket.split_once(BUNDLE_REGION_TOKEN) else {
1135        if bucket.contains('{') || bucket.contains('}') {
1136            return Err(format!(
1137                "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the \
1138                 only one"
1139            ));
1140        }
1141        return Ok(BundleUri::Literal(uri));
1142    };
1143
1144    if after.contains(BUNDLE_REGION_TOKEN) {
1145        return Err(format!("'{uri}' repeats {BUNDLE_REGION_TOKEN}"));
1146    }
1147    if before.contains('{') || before.contains('}') || after.contains('{') || after.contains('}') {
1148        return Err(format!(
1149            "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the only one"
1150        ));
1151    }
1152
1153    Ok(BundleUri::Regional {
1154        before: &uri[.."s3://".len() + before.len()],
1155        after: &uri["s3://".len() + before.len() + BUNDLE_REGION_TOKEN.len()..],
1156    })
1157}
1158
1159#[cfg(test)]
1160mod tests {
1161    use super::*;
1162
1163    #[test]
1164    fn private_database_setup_accepts_the_default_network() {
1165        for lifecycle in [
1166            crate::ResourceLifecycle::Frozen,
1167            crate::ResourceLifecycle::Live,
1168        ] {
1169            let stack = crate::Stack::new("database".to_string())
1170                .add(
1171                    crate::Postgres::new("metadata".to_string()).build(),
1172                    lifecycle,
1173                )
1174                .build();
1175            assert!(!restricts_network_mode(&stack, false));
1176            assert!(!restricts_network_mode(&stack, true));
1177        }
1178        assert!(!restricts_network_mode(
1179            &crate::Stack::new("empty".to_string()).build(),
1180            false,
1181        ));
1182    }
1183
1184    /// A wildcard reaching the grant would widen it past the bundle, and it widens the Frozen
1185    /// object grant as readily as the Live prefix — both interpolate the path into the ARN.
1186    #[test]
1187    fn a_uri_carrying_an_iam_wildcard_is_refused() {
1188        for uri in [
1189            "s3://acme/team-*/v1/bundle.zip",
1190            "s3://acme/sandbox-bundle/f00d/bundle?.zip",
1191            "s3://acme-*/sandbox-bundle/f00d/bundle.zip",
1192        ] {
1193            let error = parse_bundle_uri(uri).expect_err("a wildcard must be refused");
1194            assert!(error.contains("IAM wildcard"), "for {uri}: {error}");
1195        }
1196
1197        parse_bundle_uri("s3://acme/sandbox-bundle/f00d/bundle.zip")
1198            .expect("an ordinary key still parses");
1199    }
1200
1201    /// The rule both package formats grant by, pinned here rather than in either. The
1202    /// near-misses the cases separate: the key's first segment grants objects a rebuild never
1203    /// reads, and the object's own directory pins the version segment that moves.
1204    #[test]
1205    fn a_grantable_prefix_stops_above_the_segment_that_moves() {
1206        assert_eq!(
1207            stable_bundle_key_prefix("sandbox-bundle/f00dcafe/bundle.zip"),
1208            Some("sandbox-bundle")
1209        );
1210        assert_eq!(
1211            stable_bundle_key_prefix("artifacts/team-a/sandbox/f00dcafe/bundle.zip"),
1212            Some("artifacts/team-a/sandbox"),
1213            "a deeper key narrows the prefix, it never widens to the first segment"
1214        );
1215
1216        // Nothing sits above the version segment, so no prefix a moved bundle stays inside
1217        // exists. Emitting the object grant instead installs a role that denies the next rebuild.
1218        assert_eq!(stable_bundle_key_prefix("agents/bundle.zip"), None);
1219        assert_eq!(stable_bundle_key_prefix("bundle.zip"), None);
1220    }
1221
1222    fn sandbox_with(egress: SandboxEgress, preview_ports: Vec<u16>) -> Sandbox {
1223        Sandbox::new("agent-sbx".to_string())
1224            .code(SandboxCode::Image {
1225                image: "ubuntu".to_string(),
1226            })
1227            .limits(SandboxLimits {
1228                cpu: "1".to_string(),
1229                memory: "2Gi".to_string(),
1230                disk: "20Gi".to_string(),
1231                max_processes: None,
1232            })
1233            .egress(egress)
1234            .lifecycle(SandboxLifecyclePolicy {
1235                max_lifetime_seconds: None,
1236                idle_pause_seconds: None,
1237            })
1238            .preview_ports(preview_ports)
1239            .build()
1240    }
1241
1242    /// A URI with no token must come back whole, because every bundle configured today has none
1243    /// and emitting one differently would change every existing customer's template.
1244    #[test]
1245    fn a_uri_without_a_token_is_carried_whole() {
1246        assert_eq!(
1247            parse_bundle_uri("s3://acme-artifacts-us-east-2/agents/bundle.zip"),
1248            Ok(BundleUri::Literal(
1249                "s3://acme-artifacts-us-east-2/agents/bundle.zip"
1250            ))
1251        );
1252    }
1253
1254    /// The split has to rejoin to the original with the region in place, or an emitter builds a
1255    /// URI that is subtly not the one the vendor configured.
1256    #[test]
1257    fn a_regional_uri_splits_either_side_of_the_token() {
1258        let BundleUri::Regional { before, after } =
1259            parse_bundle_uri("s3://acme-artifacts-{region}/agents/bundle.zip")
1260                .expect("the token is accepted in the bucket")
1261        else {
1262            panic!("a bucket-position token must split");
1263        };
1264
1265        assert_eq!(before, "s3://acme-artifacts-");
1266        assert_eq!(after, "/agents/bundle.zip");
1267        assert_eq!(
1268            format!("{before}us-east-2{after}"),
1269            "s3://acme-artifacts-us-east-2/agents/bundle.zip",
1270            "the halves must rejoin to the URI the vendor meant"
1271        );
1272    }
1273
1274    /// Each of these reaches S3 verbatim and dies ~160s into an image build if it is not refused
1275    /// here, which is the whole reason this runs at plan time.
1276    #[test]
1277    fn a_token_this_build_cannot_resolve_is_refused() {
1278        for uri in [
1279            "s3://acme-artifacts-{regio}/bundle.zip",
1280            "s3://acme-artifacts/{region}/bundle.zip",
1281            "s3://acme-artifacts-{region}-{region}/bundle.zip",
1282            "s3://acme-artifacts/bundle-{version}.zip",
1283            "s3://acme}-artifacts-{region}/bundle.zip",
1284            "s3://acme{-artifacts-{region}/bundle.zip",
1285        ] {
1286            assert!(
1287                parse_bundle_uri(uri).is_err(),
1288                "'{uri}' must be refused before it can reach an image build"
1289            );
1290        }
1291    }
1292
1293    #[test]
1294    fn resource_type_is_stable() {
1295        assert_eq!(Sandbox::RESOURCE_TYPE.as_ref(), "sandbox");
1296    }
1297
1298    #[test]
1299    fn capability_sets_are_per_platform() {
1300        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1301        assert!(
1302            gcp.reconnect,
1303            "generation from the container boot id makes a sandbox reachable across processes"
1304        );
1305        assert!(!gcp.preview);
1306        assert!(gcp.enforced_limits);
1307
1308        let azure = SandboxCapabilities::for_platform(Platform::Azure).expect("azure is supported");
1309        assert!(azure.files, "every backend moves files");
1310        assert!(gcp.files);
1311        // Azure is the only backend whose egress policy matches on host pattern, and the only
1312        // one where `deny` and a hostname list are the same object.
1313        assert!(azure.domain_egress_rules);
1314        assert!(azure.egress_deny);
1315        // The data plane honours a continuous cpu/memory/disk surface and refuses anything
1316        // outside `n×250m`, with the rule in the message.
1317        assert!(azure.enforced_limits);
1318        assert!(azure.pause_resume);
1319        // Both stay false for reasons that are not "unbuilt": a snapshot id has nothing to
1320        // consume it on any backend, and an Azure port's auth is anonymous or a human allowlist,
1321        // neither of which is a port-scoped credential.
1322        assert!(!azure.snapshot);
1323        assert!(!azure.preview);
1324
1325        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1326        assert!(!aws.snapshot, "AWS has no user-callable sandbox snapshot");
1327        assert!(aws.pause_resume);
1328
1329        let k8s =
1330            SandboxCapabilities::for_platform(Platform::Kubernetes).expect("k8s is supported");
1331        assert!(
1332            !k8s.preview,
1333            "the sandbox-scoped ingress gateway does not exist yet"
1334        );
1335    }
1336
1337    /// Whether the process supervising a command is a separate identity from the command.
1338    ///
1339    /// Values are measured, not inferred. AWS: the agent runs as uid 0 with
1340    /// `CapEff: 00000000a80425fb` and `setuid`s the command to uid 60000, so the two differ.
1341    /// Kubernetes: the sandbox pod pins `run_as_user: 65534` on both pod and container with
1342    /// `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no uid split is
1343    /// possible (`kubernetes_spec.rs`). Local: `docker exec` runs as the workload uid while the
1344    /// manager supervises from the host. Azure and Agent Platform have no in-sandbox supervisor.
1345    #[test]
1346    fn supervisor_isolation_is_per_platform() {
1347        let value = |platform| {
1348            SandboxCapabilities::for_platform(platform)
1349                .expect("supported")
1350                .supervisor_isolation
1351        };
1352
1353        assert!(
1354            value(Platform::Aws),
1355            "root agent setuids the command to 60000"
1356        );
1357        assert!(
1358            value(Platform::Local),
1359            "the supervisor is on the host, outside the container"
1360        );
1361        assert!(
1362            !value(Platform::Kubernetes),
1363            "a single pinned uid cannot be split"
1364        );
1365        assert!(!value(Platform::Azure), "no Alien process runs the command");
1366        assert!(
1367            !value(Platform::Gcp),
1368            "no separate supervisor identity runs the command"
1369        );
1370    }
1371
1372    /// The point of the field: AWS and GCP report the *same* `supervisor_pid_namespace` (neither
1373    /// has `CAP_SYS_ADMIN`), so that axis alone reads them as equivalent. They are not — AWS
1374    /// separates the command's identity from the supervisor's and Agent Platform does not.
1375    #[test]
1376    fn supervisor_isolation_separates_aws_from_a_subprocess_backend() {
1377        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1378        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1379
1380        assert_eq!(
1381            aws.supervisor_pid_namespace, gcp.supervisor_pid_namespace,
1382            "the older axis cannot tell them apart"
1383        );
1384        assert!(
1385            aws.supervisor_isolation,
1386            "AWS setuids the command off the supervisor"
1387        );
1388        assert!(
1389            !gcp.supervisor_isolation,
1390            "the command runs under no separate supervisor identity"
1391        );
1392    }
1393
1394    /// The Agent Platform row, each value against the behaviour it was measured from. `reconnect`
1395    /// is the tripwire: it is `true` only because `generation` is derived from the container boot
1396    /// id read through the agent's health op, so a caller detects a replaced container instead of
1397    /// reconnecting to a blank one. It is also the body of the `Platform::Gcp` arm, asserted below.
1398    #[test]
1399    fn gcp_agent_platform_row_matches_measured_backend() {
1400        let row = SandboxCapabilities::gcp_agent_platform();
1401
1402        assert!(row.files, "agent file ops move over the sandbox envelope");
1403        assert!(
1404            row.reconnect,
1405            "generation is derived from the container boot id, so a sandbox is reachable across \
1406             processes"
1407        );
1408        assert!(
1409            !row.preview,
1410            "the only ingress is :execute; no port-scoped capability"
1411        );
1412        assert!(
1413            row.pause_resume,
1414            ":pause and :resume preserve the container"
1415        );
1416        assert!(
1417            !row.snapshot,
1418            "the create path never sends a snapshot, so none is reachable through the trait"
1419        );
1420        assert!(
1421            !row.domain_egress_rules,
1422            "VPC and DNS peering is not a hostname allowlist"
1423        );
1424        assert!(
1425            row.egress_deny,
1426            "a declared deny blocks both egress and DNS"
1427        );
1428        assert!(
1429            row.enforced_limits,
1430            "ceilings are enforced, by terminating the sandbox on breach"
1431        );
1432        assert!(!row.process_limit, "no process-count ceiling is observed");
1433        assert!(row.sandbox_lifetime, "ttl maps to a sandbox expireTime");
1434        assert!(!row.supervisor_pid_namespace, "no PID-namespace isolation");
1435        assert!(
1436            !row.supervisor_isolation,
1437            "the command is not run under a separate supervisor identity"
1438        );
1439
1440        // Agent Platform is the registered GCP backend, so the arm returns exactly this row.
1441        let live = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1442        assert_eq!(
1443            live, row,
1444            "the Platform::Gcp arm is the Agent Platform capability row"
1445        );
1446    }
1447
1448    #[test]
1449    fn platforms_without_a_backend_are_an_error_not_an_empty_set() {
1450        let error = SandboxCapabilities::for_platform(Platform::Machines)
1451            .expect_err("Machines has no sandbox backend");
1452        assert_eq!(error.code, "SANDBOX_PLATFORM_UNSUPPORTED");
1453    }
1454
1455    #[test]
1456    fn unsupported_capability_names_platform_and_capability() {
1457        let capabilities = SandboxCapabilities::for_platform(Platform::Gcp).expect("supported");
1458        let error = capabilities
1459            .require(SandboxCapability::Preview, Platform::Gcp)
1460            .expect_err("GCP has no preview");
1461
1462        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1463        let rendered = error.to_string();
1464        assert!(
1465            rendered.contains("preview"),
1466            "names the capability: {rendered}"
1467        );
1468        assert!(rendered.contains("gcp"), "names the platform: {rendered}");
1469    }
1470
1471    /// Azure matches on hostname; AWS and Kubernetes match CIDRs, and Local and GCP have a
1472    /// switch rather than a filter. Accepting a hostname list on those four would leave a stack
1473    /// reading as restricted while the sandbox reaches the whole internet.
1474    #[test]
1475    fn a_hostname_allowlist_is_refused_everywhere_it_would_be_approximated() {
1476        let sandbox = sandbox_with(
1477            SandboxEgress::AllowDomains {
1478                domains: vec!["example.com".to_string()],
1479            },
1480            vec![],
1481        );
1482
1483        for platform in [
1484            Platform::Aws,
1485            Platform::Gcp,
1486            Platform::Kubernetes,
1487            Platform::Local,
1488        ] {
1489            let error = sandbox
1490                .validate_for_platform(platform)
1491                .expect_err("only Azure expresses a hostname allowlist");
1492            assert_eq!(
1493                error.code, "SANDBOX_CAPABILITY_UNSUPPORTED",
1494                "on {platform:?}"
1495            );
1496        }
1497
1498        assert!(
1499            SandboxCapabilities::for_platform(Platform::Azure)
1500                .expect("supported")
1501                .domain_egress_rules,
1502            "Azure's egress policy matches on host pattern"
1503        );
1504    }
1505
1506    /// `deny` is the declaration that carries a security promise, so a backend that cannot keep
1507    /// it has to refuse rather than accept it and run the code with open egress.
1508    #[test]
1509    fn a_denied_egress_is_refused_where_it_would_not_be_enforced() {
1510        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1511
1512        // GCP is asserted at the capability rather than through validation: this sandbox declares
1513        // ceilings GCP cannot enforce, so it is refused for a reason unrelated to egress.
1514        assert!(
1515            SandboxCapabilities::for_platform(Platform::Gcp)
1516                .expect("supported")
1517                .egress_deny
1518        );
1519
1520        for platform in [Platform::Aws, Platform::Kubernetes, Platform::Local] {
1521            sandbox
1522                .validate_for_platform(platform)
1523                .expect("deny is enforced here");
1524        }
1525
1526        // Declares no ceilings, which Azure refuses for its own reason, so this isolates egress.
1527        let egress_only = Sandbox::new("sbx".to_string())
1528            .code(SandboxCode::Image {
1529                image: "alpine".to_string(),
1530            })
1531            .egress(SandboxEgress::Deny)
1532            .lifecycle(SandboxLifecyclePolicy {
1533                max_lifetime_seconds: None,
1534                idle_pause_seconds: None,
1535            })
1536            .build();
1537
1538        egress_only
1539            .validate_for_platform(Platform::Azure)
1540            .expect("Azure creates the sandbox under a Deny policy with full inspection");
1541    }
1542
1543    /// A sandbox naming no ceilings is valid on every platform and still resolves to a concrete
1544    /// set. The rule Azure applies to ceilings that *are* declared is pinned separately, by
1545    /// `azure_sizes_follow_the_rule_the_data_plane_states`.
1546    #[test]
1547    fn a_sandbox_declaring_no_ceilings_takes_the_platforms_own() {
1548        let undeclared = Sandbox::new("sbx".to_string())
1549            .code(SandboxCode::Image {
1550                image: "alpine".to_string(),
1551            })
1552            .egress(SandboxEgress::Deny)
1553            .lifecycle(SandboxLifecyclePolicy {
1554                max_lifetime_seconds: None,
1555                idle_pause_seconds: None,
1556            })
1557            .build();
1558
1559        undeclared
1560            .validate_for_platform(Platform::Azure)
1561            .expect("a sandbox naming no ceilings takes the platform's own");
1562
1563        // A backend still gets a concrete set, so nothing downstream has to invent one.
1564        assert_eq!(undeclared.resolved_limits().cpu, "1");
1565    }
1566
1567    /// Pins Azure's own sizing rule: `250m`, `1500m` and `4000m` are valid, `32000m` and `333m`
1568    /// are refused. Checked at plan time so a bad value is a declaration to fix, not a package
1569    /// that renders without error and dies at the first sandbox.
1570    #[test]
1571    fn azure_sizes_follow_the_rule_the_data_plane_states() {
1572        let sized = |cpu: &str, memory: &str, disk: &str| {
1573            let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1574            let limits = sandbox
1575                .limits
1576                .as_mut()
1577                .expect("the fixture declares limits");
1578            limits.cpu = cpu.to_string();
1579            limits.memory = memory.to_string();
1580            limits.disk = disk.to_string();
1581            sandbox.validate_for_platform(Platform::Azure)
1582        };
1583
1584        sized("250m", "512Mi", "5120Mi").expect("the smallest step the data plane accepts");
1585        sized("4000m", "8192Mi", "40960Mi").expect("cpu, memory and disk are all honoured");
1586        sized("16000m", "32Gi", "320Gi").expect("the top of the range");
1587
1588        // Same off-step case `azure_sandbox_limits` checks the multiple for, not just the range.
1589        let off_step = sized("333m", "512Mi", "5120Mi").expect_err("333m is not a step of 250m");
1590        assert_eq!(off_step.code, "SANDBOX_LIMIT_INVALID", "{off_step}");
1591        assert!(off_step.to_string().contains("cpu"), "{off_step}");
1592
1593        let too_big = sized("32000m", "64Gi", "640Gi").expect_err("32 cores is over the ceiling");
1594        assert_eq!(too_big.code, "SANDBOX_LIMIT_INVALID", "{too_big}");
1595
1596        // Both ceilings are derived from the cpu, so the same memory passes at one size and fails
1597        // at another - which is what makes them worth checking rather than bounding absolutely.
1598        sized("1000m", "2Gi", "20Gi").expect("2Gi is exactly one core's worth");
1599        let over_memory = sized("250m", "2Gi", "5120Mi").expect_err("2Gi needs a full core");
1600        assert_eq!(over_memory.code, "SANDBOX_LIMIT_INVALID", "{over_memory}");
1601        assert!(over_memory.to_string().contains("memory"), "{over_memory}");
1602
1603        let over_disk = sized("250m", "512Mi", "20Gi").expect_err("20Gi needs a full core");
1604        assert!(over_disk.to_string().contains("disk"), "{over_disk}");
1605    }
1606
1607    #[test]
1608    fn preview_ports_require_the_preview_capability() {
1609        let sandbox = sandbox_with(SandboxEgress::Deny, vec![8080]);
1610
1611        sandbox
1612            .validate_for_platform(Platform::Aws)
1613            .expect("AWS mints a port-scoped JWE");
1614
1615        let error = sandbox
1616            .validate_for_platform(Platform::Kubernetes)
1617            .expect_err("Kubernetes preview is deferred");
1618        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1619    }
1620
1621    /// A grant nothing reads is the silent no-op the capability contract exists to prevent, and
1622    /// here it is worse than useless: the reader would take it for a base image that needs
1623    /// authenticating while the platform pulls `code.image` itself.
1624    #[test]
1625    fn a_private_base_image_is_refused_off_aws() {
1626        let mut sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1627        sandbox.code = SandboxCode::Image {
1628            image: "s3://acme-artifacts/agents/bundle.zip".to_string(),
1629        };
1630        sandbox.private_base_image =
1631            Some("123456789012.dkr.ecr.{region}.amazonaws.com/acme:tag".to_string());
1632
1633        sandbox
1634            .validate_for_platform(Platform::Aws)
1635            .expect("AWS builds its image from a bundle, so a base image sits behind code.image");
1636
1637        for platform in [
1638            Platform::Gcp,
1639            Platform::Azure,
1640            Platform::Kubernetes,
1641            Platform::Local,
1642        ] {
1643            let error = sandbox
1644                .validate_for_platform(platform)
1645                .expect_err("a backend that builds no image must refuse a base image for one");
1646            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1647            assert!(
1648                error.to_string().contains("privateBaseImage"),
1649                "the refusal must name the field the user declared: {error}"
1650            );
1651        }
1652    }
1653
1654    #[test]
1655    fn gcp_accepts_a_sandbox_declaring_enforced_limits() {
1656        let sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1657        sandbox
1658            .validate_for_platform(Platform::Gcp)
1659            .expect("Agent Platform enforces declared ceilings, by terminating on breach");
1660    }
1661
1662    #[test]
1663    fn invalid_quantities_are_rejected_with_the_offending_field() {
1664        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1665        sandbox
1666            .limits
1667            .as_mut()
1668            .expect("the fixture declares limits")
1669            .memory = "2Gb".to_string();
1670
1671        let error = sandbox
1672            .validate_for_platform(Platform::Aws)
1673            .expect_err("Gb is not a valid suffix");
1674        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1675        assert!(error.to_string().contains("memory"));
1676
1677        sandbox
1678            .limits
1679            .as_mut()
1680            .expect("the fixture declares limits")
1681            .memory = "2Gi".to_string();
1682        sandbox
1683            .limits
1684            .as_mut()
1685            .expect("the fixture declares limits")
1686            .cpu = "0".to_string();
1687        let error = sandbox
1688            .validate_for_platform(Platform::Aws)
1689            .expect_err("zero cpu is not a ceiling");
1690        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1691    }
1692
1693    #[test]
1694    fn zero_max_processes_is_rejected() {
1695        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1696        sandbox
1697            .limits
1698            .as_mut()
1699            .expect("the fixture declares limits")
1700            .max_processes = Some(0);
1701
1702        let error = sandbox
1703            .validate_for_platform(Platform::Local)
1704            .expect_err("a sandbox must be able to run at least one process");
1705        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1706        assert!(error.to_string().contains("maxProcesses"));
1707    }
1708
1709    /// A process ceiling needs a container runtime. Kubernetes sets one per node rather than per
1710    /// pod, and neither MicroVMs nor Azure sandboxes expose one, so accepting the declaration
1711    /// anywhere else would mean carrying a bound nothing applies.
1712    #[test]
1713    fn a_process_ceiling_is_accepted_only_where_a_runtime_can_apply_it() {
1714        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1715        sandbox
1716            .limits
1717            .as_mut()
1718            .expect("the fixture declares limits")
1719            .max_processes = Some(256);
1720
1721        sandbox
1722            .validate_for_platform(Platform::Local)
1723            .expect("Docker takes a pids limit");
1724
1725        for platform in [Platform::Aws, Platform::Azure, Platform::Kubernetes] {
1726            let error = sandbox
1727                .validate_for_platform(platform)
1728                .expect_err("a process ceiling nothing applies must be refused");
1729            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1730        }
1731    }
1732
1733    /// Lambda rejects a run outside 1–28,800 rather than clamping it, so a value beyond that
1734    /// would pass planning, render into the package, and fail at the first sandbox. Kubernetes
1735    /// takes the same field with no such bound, so the check is AWS's alone.
1736    #[test]
1737    fn a_lifetime_aws_would_reject_is_refused_while_planning() {
1738        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1739
1740        for seconds in [0, 28_801, 100_000] {
1741            sandbox.lifecycle.max_lifetime_seconds = Some(seconds);
1742            let error = sandbox
1743                .validate_for_platform(Platform::Aws)
1744                .expect_err("a lifetime outside what AWS runs is refused");
1745            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "{seconds}s");
1746
1747            // Kubernetes has no such ceiling, so the same declaration is fine there.
1748            sandbox
1749                .validate_for_platform(Platform::Kubernetes)
1750                .expect("the kubelet takes any activeDeadlineSeconds");
1751        }
1752
1753        sandbox.lifecycle.max_lifetime_seconds = Some(28_800);
1754        sandbox
1755            .validate_for_platform(Platform::Aws)
1756            .expect("the ceiling itself is allowed");
1757    }
1758
1759    /// An image reference Azure cannot honour is refused while planning, not at the first sandbox.
1760    ///
1761    /// `code.image`'s own documentation gives a tag and a registry path as examples — exactly
1762    /// what Azure cannot take, so this is the shape a customer is most likely to declare.
1763    #[test]
1764    fn an_image_azure_cannot_pull_is_refused_while_planning() {
1765        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1766        // Azure enforces no declared ceiling, so a sandbox carrying limits is refused before the
1767        // image is ever read.
1768        sandbox.limits = None;
1769
1770        for image in [
1771            "ubuntu:24.04",
1772            "ghcr.io/myorg/sandbox:latest",
1773            "ubuntu@sha256:abc",
1774            "",
1775            "   ",
1776            "ubuntu latest",
1777            "ubuntu?x",
1778        ] {
1779            sandbox.code = SandboxCode::Image {
1780                image: image.to_string(),
1781            };
1782            let error = sandbox
1783                .validate_for_platform(Platform::Azure)
1784                .expect_err("an image Azure has nowhere to put is refused");
1785            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "image '{image}'");
1786
1787            // The same declaration is ordinary everywhere that pulls a reference.
1788            sandbox
1789                .validate_for_platform(Platform::Kubernetes)
1790                .expect("a registry reference is what every other backend takes");
1791        }
1792
1793        for image in ["ubuntu", "ubuntu-22.04", "debian_slim"] {
1794            sandbox.code = SandboxCode::Image {
1795                image: image.to_string(),
1796            };
1797            sandbox
1798                .validate_for_platform(Platform::Azure)
1799                .unwrap_or_else(|error| panic!("'{image}' is a catalog name: {error}"));
1800        }
1801
1802        // Surrounding space is trimmed rather than carried into the create body.
1803        sandbox.code = SandboxCode::Image {
1804            image: " ubuntu ".to_string(),
1805        };
1806        assert_eq!(
1807            sandbox
1808                .azure_catalog_image()
1809                .expect("a padded name is still a name"),
1810            "ubuntu"
1811        );
1812    }
1813
1814    /// A deadline is accepted only where the platform itself terminates on it — the kubelet's
1815    /// `activeDeadlineSeconds` and Lambda's `maximumDurationInSeconds`. Everywhere else it would
1816    /// need a reaper that does not exist, so it is refused rather than accepted and dropped.
1817    #[test]
1818    fn a_sandbox_deadline_is_accepted_only_where_the_platform_applies_it() {
1819        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1820        sandbox.lifecycle.max_lifetime_seconds = Some(3600);
1821
1822        sandbox
1823            .validate_for_platform(Platform::Kubernetes)
1824            .expect("the kubelet enforces activeDeadlineSeconds");
1825        sandbox
1826            .validate_for_platform(Platform::Aws)
1827            .expect("Lambda terminates the MicroVM at maximumDurationInSeconds");
1828
1829        for platform in [Platform::Azure, Platform::Local] {
1830            let error = sandbox
1831                .validate_for_platform(platform)
1832                .expect_err("a deadline nothing applies must be refused");
1833            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1834        }
1835    }
1836
1837    /// A MicroVM bursts to four times its baseline with no way to opt out, so a ceiling is kept
1838    /// by choosing the size whose *peak* fits inside it. Sizing by baseline would hand back a
1839    /// sandbox that can reach four times what the customer declared.
1840    #[test]
1841    fn an_aws_size_is_chosen_so_its_peak_stays_inside_the_declared_ceiling() {
1842        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1843        let tier = sandbox
1844            .microvm_tier()
1845            .expect("2Gi/1cpu/20Gi is satisfiable");
1846
1847        assert_eq!(
1848            tier.peak_memory_mib, 2048,
1849            "the peak is the declared ceiling"
1850        );
1851        assert_eq!(
1852            tier.baseline_memory_mib, 512,
1853            "which is a quarter of it as the baseline"
1854        );
1855        assert!(tier.max_disk_mib <= 20 * 1024);
1856    }
1857
1858    /// AWS allocates one vCPU per 2GB, so a cpu ceiling below what the memory ceiling implies
1859    /// cannot be honoured together with it. Letting cpu choose the size instead would hand back a
1860    /// machine four times smaller than the memory asked for, with nothing to indicate it.
1861    #[test]
1862    fn a_cpu_ceiling_below_what_the_memory_implies_is_refused_not_quietly_downsized() {
1863        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1864        {
1865            let limits = sandbox
1866                .limits
1867                .as_mut()
1868                .expect("the fixture declares limits");
1869            limits.cpu = "1".to_string();
1870            limits.memory = "8Gi".to_string();
1871        }
1872
1873        let error = sandbox
1874            .microvm_tier()
1875            .expect_err("1 cpu and 8Gi cannot both be ceilings on AWS");
1876        assert!(
1877            error.to_string().contains("4 vCPU"),
1878            "the refusal must say what the memory ceiling implies: {error}"
1879        );
1880
1881        sandbox
1882            .limits
1883            .as_mut()
1884            .expect("the fixture declares limits")
1885            .cpu = "4".to_string();
1886        let tier = sandbox.microvm_tier().expect("4 cpu matches 8Gi");
1887        assert_eq!(tier.peak_memory_mib, 8192);
1888    }
1889
1890    /// Below AWS's smallest peak there is no size that holds the ceiling, and rounding up to the
1891    /// nearest one would silently exceed it.
1892    #[test]
1893    fn an_aws_ceiling_smaller_than_any_size_is_refused_rather_than_rounded() {
1894        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1895        sandbox
1896            .limits
1897            .as_mut()
1898            .expect("the fixture declares limits")
1899            .memory = "1Gi".to_string();
1900
1901        let error = sandbox
1902            .validate_for_platform(Platform::Aws)
1903            .expect_err("no MicroVM size peaks at or below 1Gi");
1904        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1905        assert!(
1906            error.to_string().contains("2Gi"),
1907            "the refusal must say what the smallest holdable ceiling is: {error}"
1908        );
1909    }
1910
1911    /// `alien build` builds an AWS sandbox's base image, so source is a declaration there. On
1912    /// every other platform the image is pulled as declared, and an unbuilt source would schedule
1913    /// a pod that can never run, so the refusal still has to happen at plan time.
1914    #[test]
1915    fn source_code_is_refused_off_aws_rather_than_producing_a_broken_manifest() {
1916        let sandbox = Sandbox::new("agent".to_string())
1917            .code(SandboxCode::Source {
1918                src: "./sandbox".to_string(),
1919                toolchain: ToolchainConfig::Docker {
1920                    dockerfile: None,
1921                    build_args: None,
1922                    target: None,
1923                },
1924            })
1925            .egress(SandboxEgress::Deny)
1926            .lifecycle(SandboxLifecyclePolicy {
1927                max_lifetime_seconds: None,
1928                idle_pause_seconds: None,
1929            })
1930            .build();
1931
1932        sandbox
1933            .validate_for_platform(Platform::Aws)
1934            .expect("an AWS sandbox base image is built by `alien build`");
1935
1936        for platform in [
1937            Platform::Azure,
1938            Platform::Gcp,
1939            Platform::Kubernetes,
1940            Platform::Local,
1941        ] {
1942            let error = sandbox
1943                .validate_for_platform(platform)
1944                .expect_err("no backend builds a sandbox image from source here");
1945            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1946            assert!(
1947                error.to_string().contains("code.image"),
1948                "the refusal must say what to write instead: {error}"
1949            );
1950            assert!(
1951                error.to_string().contains(&platform.to_string()),
1952                "the refusal must name the platform that cannot build it: {error}"
1953            );
1954        }
1955    }
1956
1957    /// `validate_quantity` accepts nine suffixes. Reading only `Gi` and `Mi` would size a
1958    /// declared `4G` as though it were `4Gi`, which for a ceiling means exceeding it.
1959    #[test]
1960    fn every_accepted_unit_converts_rather_than_falling_back() {
1961        assert_eq!(quantity_mib("2Gi"), Some(2048));
1962        assert_eq!(quantity_mib("512Mi"), Some(512));
1963        assert_eq!(quantity_mib("4G"), Some(3814));
1964        assert_eq!(quantity_mib("1Ti"), Some(1024 * 1024));
1965        assert_eq!(millicores("1"), Some(1000));
1966        assert_eq!(millicores("500m"), Some(500));
1967    }
1968
1969    #[test]
1970    fn unknown_fields_are_rejected() {
1971        let json = r#"{
1972            "id": "sbx",
1973            "code": {"type": "image", "image": "ubuntu:24.04"},
1974            "limits": {"cpu": "1", "memory": "2Gi", "disk": "20Gi"},
1975            "egress": {"mode": "deny"},
1976            "lifecycle": {},
1977            "unexpected": true
1978        }"#;
1979
1980        serde_json::from_str::<Sandbox>(json).expect_err("deny_unknown_fields must reject");
1981    }
1982
1983    #[test]
1984    fn serialization_roundtrips() {
1985        let sandbox = sandbox_with(
1986            SandboxEgress::AllowDomains {
1987                domains: vec!["example.com".to_string()],
1988            },
1989            vec![8080, 9090],
1990        );
1991
1992        let json = serde_json::to_string(&sandbox).expect("serializes");
1993        let restored: Sandbox = serde_json::from_str(&json).expect("deserializes");
1994        assert_eq!(sandbox, restored);
1995    }
1996
1997    #[test]
1998    fn id_is_immutable_across_updates() {
1999        let original = sandbox_with(SandboxEgress::Deny, vec![]);
2000        let renamed = Sandbox::new("other".to_string())
2001            .code(SandboxCode::Image {
2002                image: "ubuntu".to_string(),
2003            })
2004            .limits(
2005                original
2006                    .limits
2007                    .clone()
2008                    .expect("the fixture declares limits"),
2009            )
2010            .egress(SandboxEgress::Deny)
2011            .lifecycle(SandboxLifecyclePolicy {
2012                max_lifetime_seconds: None,
2013                idle_pause_seconds: None,
2014            })
2015            .build();
2016
2017        original
2018            .validate_update(&original.clone())
2019            .expect("an unchanged config is a valid update");
2020        original
2021            .validate_update(&renamed)
2022            .expect_err("renaming a sandbox is not an update");
2023    }
2024
2025    /// Azure declares an idle-pause policy but not a wall-clock ceiling.
2026    ///
2027    /// The two travel together in `SandboxLifecyclePolicy` and are gated separately on purpose:
2028    /// Azure pauses on idle and has no maximum lifetime, so accepting one and refusing the
2029    /// other is the honest split rather than an inconsistency.
2030    #[test]
2031    fn azure_takes_an_idle_policy_and_still_refuses_a_lifetime_ceiling() {
2032        let with_policy = |lifecycle: SandboxLifecyclePolicy| {
2033            Sandbox::new("sbx".to_string())
2034                .code(SandboxCode::Image {
2035                    image: "ubuntu".to_string(),
2036                })
2037                .egress(SandboxEgress::Allow)
2038                .lifecycle(lifecycle)
2039                .build()
2040                .validate_for_platform(Platform::Azure)
2041        };
2042
2043        with_policy(SandboxLifecyclePolicy {
2044            max_lifetime_seconds: None,
2045            idle_pause_seconds: Some(900),
2046        })
2047        .expect("Azure pauses a sandbox on idle");
2048
2049        let error = with_policy(SandboxLifecyclePolicy {
2050            max_lifetime_seconds: Some(3600),
2051            idle_pause_seconds: None,
2052        })
2053        .expect_err("Azure has no wall-clock ceiling to enforce one with");
2054        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2055        assert!(
2056            error.message.contains("sandboxLifetime"),
2057            "names the capability: {}",
2058            error.message
2059        );
2060    }
2061
2062    /// An allowlist naming nothing is a deny-all wearing an allowlist's label.
2063    ///
2064    /// It renders as a `Deny` default with no rules — the shape the Azure provider adds a
2065    /// catch-all to avoid — and a reader scanning the declaration sees "allowDomains" and reads
2066    /// the opposite of what it does.
2067    #[test]
2068    fn an_allowlist_with_no_domains_is_refused() {
2069        let declared = |domains: Vec<String>| {
2070            Sandbox::new("sbx".to_string())
2071                .code(SandboxCode::Image {
2072                    image: "ubuntu".to_string(),
2073                })
2074                .egress(SandboxEgress::AllowDomains { domains })
2075                .lifecycle(SandboxLifecyclePolicy {
2076                    max_lifetime_seconds: None,
2077                    idle_pause_seconds: None,
2078                })
2079                .build()
2080                .validate_for_platform(Platform::Azure)
2081        };
2082
2083        let error = declared(vec![]).expect_err("an empty allowlist must be refused");
2084        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2085
2086        declared(vec!["api.example.com".to_string()])
2087            .expect("a named domain is what an allowlist is for");
2088    }
2089
2090    /// The two expressible modes map to the boolean; a host list maps to nothing so the caller has
2091    /// to refuse rather than silently pick a side.
2092    #[test]
2093    fn internet_access_switch_maps_only_the_two_expressible_modes() {
2094        assert_eq!(SandboxEgress::Allow.internet_access_switch(), Some(true));
2095        assert_eq!(SandboxEgress::Deny.internet_access_switch(), Some(false));
2096        assert_eq!(
2097            SandboxEgress::AllowDomains {
2098                domains: vec!["api.example.com".to_string()]
2099            }
2100            .internet_access_switch(),
2101            None,
2102            "a host list has no boolean and must not be approximated"
2103        );
2104    }
2105}