Skip to main content

alien_core/resources/
sandbox.rs

1//! Sandbox resource for running untrusted code in an isolated environment.
2//!
3//! The declaration provisions a durable parent, and the application creates and destroys
4//! individual sandboxes through its binding at runtime.
5//!
6//! The capability set differs per platform and is published rather than assumed. Calling an
7//! unsupported capability is a typed error naming both the platform and the capability, so a
8//! portable application can branch on `SandboxCapabilities` before it calls.
9
10use crate::error::{ErrorData, Result};
11use crate::resource::{ResourceDefinition, ResourceOutputsDefinition, ResourceRef, ResourceType};
12use crate::resources::ToolchainConfig;
13use crate::Platform;
14use alien_error::AlienError;
15use bon::Builder;
16use serde::{Deserialize, Serialize};
17use sha2::{Digest, Sha256};
18use std::any::Any;
19use std::fmt::Debug;
20
21/// Specifies where the sandbox's root filesystem comes from.
22#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
23#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
24#[serde(rename_all = "camelCase", tag = "type")]
25pub enum SandboxCode {
26    /// A prebuilt container image used as the sandbox root filesystem.
27    #[serde(rename_all = "camelCase")]
28    Image {
29        /// Image reference (e.g. `ubuntu:24.04`, `ghcr.io/myorg/sandbox:latest`). AWS wants an
30        /// `s3://` bundle; Azure takes a catalog name such as `ubuntu` or an amd64 registry
31        /// image, told apart by syntax. Each refuses what it cannot take while planning.
32        image: String,
33    },
34    /// A Dockerfile `alien build` builds into the sandbox's base image.
35    ///
36    /// AWS only, and docker only: the base image is a root filesystem, not a binary laid on one.
37    /// `alien release` pushes it and the bundle layers the sandbox agent on afterwards.
38    #[serde(rename_all = "camelCase")]
39    Source {
40        /// The source directory to build from
41        src: String,
42        /// Toolchain configuration with type-safe options
43        toolchain: ToolchainConfig,
44    },
45}
46
47/// Hard ceilings enforced on a sandbox.
48///
49/// These are limits, not scheduling requests. Untrusted code does not respect a hint, so every
50/// field is enforced by the platform and a platform that cannot enforce one is rejected at plan
51/// time rather than silently ignoring it.
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
54#[serde(rename_all = "camelCase", deny_unknown_fields)]
55pub struct SandboxLimits {
56    /// CPU ceiling in cores or millicores (e.g. `"1"`, `"500m"`)
57    pub cpu: String,
58    /// Memory ceiling (e.g. `"2Gi"`, `"512Mi"`)
59    pub memory: String,
60    /// Disk ceiling (e.g. `"20Gi"`)
61    pub disk: String,
62    /// Maximum number of processes, which bounds fork bombs.
63    ///
64    /// Optional because only a container runtime has the primitive: Kubernetes sets a pid ceiling
65    /// per node, not per pod, and neither AWS MicroVMs nor Azure sandboxes expose one. Declaring
66    /// it on a platform that cannot apply it is refused at plan time.
67    #[serde(default, skip_serializing_if = "Option::is_none")]
68    pub max_processes: Option<u32>,
69}
70
71/// One of the five sizes a Lambda MicroVM can be built at.
72///
73/// AWS has no ceiling knob: `minimumMemoryInMiB` sets a *baseline* and a running MicroVM bursts
74/// vertically to four times it with no way to opt out. A declared ceiling is therefore honoured by
75/// picking the tier whose **peak** stays inside it, not the tier whose baseline matches it.
76#[derive(Debug, Clone, Copy, PartialEq, Eq)]
77pub struct MicrovmTier {
78    /// What `minimumMemoryInMiB` is set to.
79    pub baseline_memory_mib: i64,
80    /// The most memory the MicroVM can reach, in MiB.
81    pub peak_memory_mib: i64,
82    /// The most vCPU the MicroVM can reach.
83    pub peak_vcpu: u32,
84    /// The most disk the MicroVM can use, in MiB.
85    pub max_disk_mib: i64,
86}
87
88/// The published sizes, smallest first. Baseline memory to vCPU is 2 GB per vCPU, peak is four
89/// times baseline, and disk is fixed per tier rather than independently selectable.
90/// Longest life AWS will run a MicroVM for, from `RunMicrovm`'s `maximumDurationInSeconds`.
91const AWS_MAX_LIFETIME_SECONDS: u32 = 28_800;
92
93/// Azure's sandbox sizing rule, quoted from the data plane's own refusal of an oversized request:
94/// *CPU must be n×250m for n=1..64 (0.25–16 cores); Memory ≤ cores × 2Gi; Disk ≤ cores × 20Gi*.
95const AZURE_CPU_STEP_MILLICORES: i64 = 250;
96const AZURE_MAX_CPU_MILLICORES: i64 = 16_000;
97const AZURE_MEMORY_MIB_PER_CORE: i64 = 2 * 1024;
98const AZURE_DISK_MIB_PER_CORE: i64 = 20 * 1024;
99
100const MICROVM_TIERS: &[MicrovmTier] = &[
101    MicrovmTier {
102        baseline_memory_mib: 512,
103        peak_memory_mib: 2048,
104        peak_vcpu: 1,
105        max_disk_mib: 8192,
106    },
107    MicrovmTier {
108        baseline_memory_mib: 1024,
109        peak_memory_mib: 4096,
110        peak_vcpu: 2,
111        max_disk_mib: 8192,
112    },
113    MicrovmTier {
114        baseline_memory_mib: 2048,
115        peak_memory_mib: 8192,
116        peak_vcpu: 4,
117        max_disk_mib: 8192,
118    },
119    MicrovmTier {
120        baseline_memory_mib: 4096,
121        peak_memory_mib: 16384,
122        peak_vcpu: 8,
123        max_disk_mib: 16384,
124    },
125    MicrovmTier {
126        baseline_memory_mib: 8192,
127        peak_memory_mib: 32768,
128        peak_vcpu: 16,
129        max_disk_mib: 32768,
130    },
131];
132
133/// Outbound network policy for a sandbox.
134#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
135#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
136#[serde(rename_all = "camelCase", tag = "mode")]
137pub enum SandboxEgress {
138    /// No outbound network access.
139    ///
140    /// Routed traffic only. Link-local is not outbound and no backend's egress control reaches
141    /// it, so this is not a boundary against instance metadata.
142    ///
143    /// Nor, on AWS, against DNS: while the connector VPC has DNS support on, a session resolves
144    /// names through that VPC's resolver, which no security group filters, so a query name can
145    /// carry data out.
146    Deny,
147    /// Unrestricted outbound access to the public internet, and none to private ranges or the
148    /// deployment's own network.
149    ///
150    /// Link-local carries the same exception as `Deny`. AWS and Kubernetes deliver both halves.
151    /// Azure and GCP deliver the first only: one matches host patterns and the other is a single
152    /// switch, so neither can name an address range to exclude.
153    Allow,
154    /// Outbound access only to the listed hostnames.
155    ///
156    /// Azure matches host patterns. With `privilegedSupervisor`, AWS resolves the names at
157    /// startup, pins their public IPv4 addresses in /etc/hosts, and filters by those addresses.
158    /// Other services sharing an allowed address are reachable; addresses remain pinned for the
159    /// session lifetime. Backends without either enforcement path refuse this mode.
160    #[serde(rename_all = "camelCase")]
161    AllowDomains {
162        /// Hostnames the sandbox may reach
163        domains: Vec<String>,
164    },
165}
166
167impl SandboxEgress {
168    /// The single outbound switch for a backend that has no host matcher, or `None` for a mode a
169    /// boolean cannot carry.
170    ///
171    /// `AllowDomains` needs a host list, so it maps to nothing and each caller refuses it in its
172    /// own error naming the sandbox. One source for what a mode means, so a template and a sandbox
173    /// cannot disagree on it.
174    pub fn internet_access_switch(&self) -> Option<bool> {
175        match self {
176            SandboxEgress::Allow => Some(true),
177            SandboxEgress::Deny => Some(false),
178            SandboxEgress::AllowDomains { .. } => None,
179        }
180    }
181}
182
183/// How long a sandbox may live and when it is paused.
184///
185/// Declaration-time ceilings, not per-request values: every sandbox created through this
186/// declaration's binding is held to them, whatever a caller asks for at runtime.
187#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
188#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
189#[serde(rename_all = "camelCase", deny_unknown_fields)]
190pub struct SandboxLifecyclePolicy {
191    /// Wall-clock ceiling on a single sandbox, after which the platform terminates it.
192    ///
193    /// Optional because not every backend has the primitive: Kubernetes has
194    /// `activeDeadlineSeconds` and AWS `maximumDurationInSeconds`, while neither Azure nor Local
195    /// expose one, so declaring a ceiling there is refused at plan time rather than accepted and
196    /// never applied. AWS caps it at 8 hours.
197    #[serde(default, skip_serializing_if = "Option::is_none")]
198    pub max_lifetime_seconds: Option<u32>,
199    /// Idle period after which the sandbox is paused, where the platform supports it.
200    ///
201    /// Stored state written before the rename calls this `idleSuspendSeconds`.
202    #[serde(alias = "idleSuspendSeconds", skip_serializing_if = "Option::is_none")]
203    pub idle_pause_seconds: Option<u32>,
204}
205
206/// What a platform's sandbox backend can actually do.
207///
208/// Published so portable code can branch before calling rather than discovering a gap through
209/// an error. Every field here corresponds to a capability that at least one platform lacks;
210/// create, exec and terminate are the guaranteed floor and are therefore not listed.
211#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
212#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
213#[serde(rename_all = "camelCase", deny_unknown_fields)]
214pub struct SandboxCapabilities {
215    /// Files can be moved in and out of a sandbox
216    pub files: bool,
217    /// A later call can reach a sandbox created by an earlier one
218    pub reconnect: bool,
219    /// A command can be started, polled and cancelled across separate calls, so it outlives the
220    /// one that started it. False where nothing inside the sandbox owns the process in between.
221    pub jobs: bool,
222    /// An authenticated, port-scoped capability to reach a service inside the sandbox
223    pub preview: bool,
224    /// Sandbox state can be paused and resumed
225    pub pause_resume: bool,
226    /// A sandbox's full state can be captured and used to create another
227    pub snapshot: bool,
228    /// Egress can be restricted to a hostname allowlist
229    pub domain_egress_rules: bool,
230    /// Whether a declared `deny` is actually enforced, rather than accepted and dropped.
231    /// It covers routed traffic; the `deny` mode says where DNS still resolves.
232    pub egress_deny: bool,
233    /// The platform enforces the declared cpu, memory and disk ceilings
234    pub enforced_limits: bool,
235    /// The platform can cap how many processes a sandbox runs
236    pub process_limit: bool,
237    /// The platform terminates a sandbox at a declared wall-clock deadline
238    pub sandbox_lifetime: bool,
239    /// A command runs in its own PID namespace and cannot see or signal the agent's processes.
240    ///
241    /// Only where an agent runs as root. Creating the namespace needs `CAP_SYS_ADMIN`, and the
242    /// Kubernetes sandbox pod drops every capability — which is also what denies `ptrace` by
243    /// construction, so granting it there would remove a lock to add one.
244    pub supervisor_pid_namespace: bool,
245    /// The process supervising a command is a different identity from the command.
246    ///
247    /// False where a command runs as the agent's own user: it can then read the supervisor's
248    /// environment and signal it. Separate from `supervisorPidNamespace`, which is about
249    /// visibility rather than identity — a backend can have one without the other.
250    pub supervisor_isolation: bool,
251}
252
253impl SandboxCapabilities {
254    /// Returns what the given platform's sandbox backend supports.
255    ///
256    /// Errors for platforms with no sandbox backend, rather than returning an all-false set —
257    /// "every capability is missing" and "this platform has no sandboxes" are different
258    /// conditions and an application should not have to tell them apart by inspection.
259    pub fn for_platform(platform: Platform) -> Result<Self> {
260        match platform {
261            Platform::Aws => Ok(Self {
262                files: true,
263                reconnect: true,
264                jobs: true,
265                preview: true,
266                pause_resume: true,
267                snapshot: false,
268                domain_egress_rules: false,
269                // Routed egress only: while the connector VPC has DNS support on, a session
270                // resolves names through its resolver, which no security group filters.
271                egress_deny: true,
272                enforced_limits: true,
273                // Nothing in the API bounds process count.
274                process_limit: false,
275                // `maximumDurationInSeconds` on `RunMicrovm`, which Lambda enforces by
276                // terminating the MicroVM. Capped at 8 hours by the service.
277                sandbox_lifetime: true,
278                // Measured, not assumed: the agent inside a Lambda MicroVM runs as uid 0 with
279                // `CapEff: 00000000a80425fb`, the standard container default set, which excludes
280                // `CAP_SYS_ADMIN`. It can drop privilege (`CAP_SETUID`/`CAP_SETGID` are held) and
281                // it cannot create a namespace. No backend offers this today.
282                supervisor_pid_namespace: false,
283                // The agent runs as uid 0 and `setuid`s the command to uid 60000, so the command
284                // runs under a different identity than the process supervising it.
285                supervisor_isolation: true,
286            }),
287            Platform::Azure => Ok(Self::azure()),
288            Platform::Gcp => Ok(Self::gcp_agent_platform()),
289            // Preview needs a gateway that validates a sandbox-and-port capability, which this
290            // backend has none of.
291            Platform::Kubernetes => Ok(Self {
292                files: true,
293                reconnect: true,
294                jobs: true,
295                preview: false,
296                pause_resume: false,
297                snapshot: false,
298                domain_egress_rules: false,
299                egress_deny: true,
300                enforced_limits: true,
301                // A pid ceiling is a kubelet setting per node, not a pod field.
302                process_limit: false,
303                // `activeDeadlineSeconds` on the pod, which the kubelet enforces.
304                sandbox_lifetime: true,
305                // The pod drops every capability, including the `CAP_SYS_ADMIN` the agent would
306                // need to unshare. That is also what denies `ptrace`, so this stays false rather
307                // than the pod being weakened to make it true.
308                supervisor_pid_namespace: false,
309                // The pod pins one uid (`run_as_user: 65534` on both pod and container) with
310                // `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no
311                // process can setuid to split the command off from a supervisor. No uid split is
312                // possible, so none exists.
313                supervisor_isolation: false,
314            }),
315            Platform::Local => Ok(Self {
316                files: true,
317                reconnect: true,
318                // Nothing runs inside the sandbox: the manager drives Docker from outside it.
319                jobs: false,
320                preview: true,
321                pause_resume: false,
322                snapshot: false,
323                domain_egress_rules: false,
324                egress_deny: true,
325                enforced_limits: true,
326                // Docker's `--pids-limit`.
327                process_limit: true,
328                sandbox_lifetime: false,
329                // Local has no in-sandbox agent: the manager drives Docker from outside, so
330                // there is no supervisor inside the sandbox to isolate from.
331                supervisor_pid_namespace: false,
332                // The supervisor is the manager on the host, outside the container entirely, and
333                // `docker exec` runs the command as the workload uid — a different identity by
334                // construction.
335                supervisor_isolation: true,
336            }),
337            Platform::Machines | Platform::Test => {
338                Err(AlienError::new(ErrorData::SandboxPlatformUnsupported {
339                    platform: platform.to_string(),
340                }))
341            }
342        }
343    }
344
345    /// What the Azure sandbox backend supports; the body of the `Platform::Azure` arm.
346    pub fn azure() -> Self {
347        Self {
348            files: true,
349            reconnect: true,
350            // No Alien process runs inside the sandbox to own a command between two calls.
351            jobs: false,
352            // A sandbox port carries a URL and an auth config, and the auth config offers two
353            // things: anonymous, or Entra ID with an allowlist of human email addresses.
354            // Neither is a credential scoped to a port for a fixed time, which is what a
355            // preview capability is. Returning the anonymous URL would publish the port.
356            preview: false,
357            pause_resume: true,
358            // False for a client reason, not a cloud one: this client has no snapshot call, and
359            // `CreateSandboxRequest` has no field to consume the id it would return. Also
360            // unclaimed: Microsoft does not garbage-collect snapshots, so an id is a bill that grows.
361            snapshot: false,
362            domain_egress_rules: true,
363            egress_deny: true,
364            // Enforced inside the sandbox, not at create: an over-allocation raises `MemoryError`
365            // while the sandbox keeps running. `azure_sandbox_limits` checks the continuous
366            // sizing rule at plan time instead of matching a tier.
367            enforced_limits: true,
368            process_limit: false,
369            // Auto-suspend and auto-delete exist; a wall-clock ceiling does not. Accepting
370            // `maxLifetimeSeconds` here would be the silent no-op the capability set exists
371            // to prevent, so this is a decision rather than a gap.
372            sandbox_lifetime: false,
373            // No Alien process inside an Azure sandbox, so there is no supervisor to isolate.
374            supervisor_pid_namespace: false,
375            // No Alien process runs the command at all — the platform's own data plane does,
376            // so there is no separate supervisor identity to speak of.
377            supervisor_isolation: false,
378        }
379    }
380
381    /// What the GCP Agent Platform sandbox backend supports; the body of the `Platform::Gcp` arm.
382    pub fn gcp_agent_platform() -> Self {
383        Self {
384            // Agent file operations move over the sandbox envelope.
385            files: true,
386            // Reaching a sandbox across processes is safe because `generation` is derived from the
387            // container boot id read through the agent's health op, so a caller detects a container
388            // replaced under a stable sandbox name rather than reconnecting to a blank one.
389            reconnect: true,
390            jobs: true,
391            // No method mints a port-scoped ingress capability; the only ingress is `:execute`.
392            preview: false,
393            // `:resume` can return a fresh container while reporting success, so a pause does not
394            // keep the sandbox's state.
395            pause_resume: false,
396            // The create path never sends `sandbox_environment_snapshot`, so no sandbox state is
397            // reachable through the trait; declared false until the client carries it.
398            snapshot: false,
399            // Egress is shaped by VPC and DNS peering, which is not a hostname allowlist.
400            domain_egress_rules: false,
401            // A declared `deny` blocks both routed egress and DNS.
402            egress_deny: true,
403            // The declared ceilings are enforced, but by terminating the sandbox on breach rather
404            // than by refusing the allocation — a caller reading `true` should expect the sandbox
405            // to die, not a clean error at the point of the request.
406            enforced_limits: true,
407            // No ceiling on process count is observed.
408            process_limit: false,
409            // `ttl` maps to a sandbox `expireTime` the platform terminates at.
410            sandbox_lifetime: true,
411            // No PID-namespace isolation between the command and anything supervising it.
412            supervisor_pid_namespace: false,
413            // No separate supervisor identity: the command is not run under a different identity
414            // than the process supervising it.
415            supervisor_isolation: false,
416        }
417    }
418
419    /// Returns a typed error if the named capability is absent on this platform.
420    pub fn require(&self, capability: SandboxCapability, platform: Platform) -> Result<()> {
421        let available = match capability {
422            SandboxCapability::Files => self.files,
423            SandboxCapability::Reconnect => self.reconnect,
424            SandboxCapability::Jobs => self.jobs,
425            SandboxCapability::Preview => self.preview,
426            SandboxCapability::PauseResume => self.pause_resume,
427            SandboxCapability::Snapshot => self.snapshot,
428            SandboxCapability::DomainEgressRules => self.domain_egress_rules,
429            SandboxCapability::EgressDeny => self.egress_deny,
430            SandboxCapability::EnforcedLimits => self.enforced_limits,
431            SandboxCapability::ProcessLimit => self.process_limit,
432            SandboxCapability::SandboxLifetime => self.sandbox_lifetime,
433            SandboxCapability::SupervisorPidNamespace => self.supervisor_pid_namespace,
434            SandboxCapability::SupervisorIsolation => self.supervisor_isolation,
435        };
436
437        if available {
438            return Ok(());
439        }
440
441        Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
442            capability: capability.as_str().to_string(),
443            platform: platform.to_string(),
444        }))
445    }
446}
447
448/// Names a single sandbox capability, so an unsupported call can report which one it needed.
449#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
450#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
451#[serde(rename_all = "camelCase")]
452pub enum SandboxCapability {
453    /// Moving files in and out of a sandbox
454    Files,
455    /// Reaching a sandbox created by an earlier call
456    Reconnect,
457    /// Starting, polling and cancelling a command across separate calls
458    Jobs,
459    /// An authenticated, port-scoped ingress capability
460    Preview,
461    /// Pausing and resuming sandbox state
462    PauseResume,
463    /// Capturing full sandbox state
464    Snapshot,
465    /// Restricting egress to a hostname allowlist
466    DomainEgressRules,
467    /// Refusing outbound access when a sandbox declares none
468    EgressDeny,
469    /// Platform-enforced resource ceilings
470    EnforcedLimits,
471    /// A ceiling on the number of processes a sandbox may run
472    ProcessLimit,
473    /// A wall-clock ceiling on a sandbox, applied by the platform rather than by a caller
474    SandboxLifetime,
475    /// A command runs in its own PID namespace, isolated from the agent supervising it
476    SupervisorPidNamespace,
477    /// A command runs under a different identity than the process supervising it
478    SupervisorIsolation,
479}
480
481impl SandboxCapability {
482    /// Returns the stable identifier used in errors and capability queries.
483    pub fn as_str(&self) -> &'static str {
484        match self {
485            Self::Files => "files",
486            Self::Reconnect => "reconnect",
487            Self::Jobs => "jobs",
488            Self::Preview => "preview",
489            Self::PauseResume => "pauseResume",
490            Self::Snapshot => "snapshot",
491            Self::DomainEgressRules => "domainEgressRules",
492            Self::EgressDeny => "egressDeny",
493            Self::EnforcedLimits => "enforcedLimits",
494            Self::ProcessLimit => "processLimit",
495            Self::SandboxLifetime => "sandboxLifetime",
496            Self::SupervisorPidNamespace => "supervisorPidNamespace",
497            Self::SupervisorIsolation => "supervisorIsolation",
498        }
499    }
500}
501
502/// Opt-in supervisor-owned network enforcement. Caller requests cannot change this identity.
503#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
504#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
505#[serde(rename_all = "camelCase", deny_unknown_fields)]
506pub struct SandboxPrivilegedSupervisor {
507    /// Nonzero numeric uid and primary gid for every command, including the image entrypoint.
508    pub command_uid: u32,
509}
510
511/// An isolated environment for running untrusted code, created at runtime.
512#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Builder)]
513#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
514#[serde(rename_all = "camelCase", deny_unknown_fields)]
515#[builder(start_fn = new)]
516pub struct Sandbox {
517    /// Identifier for the sandbox. Must contain only alphanumeric characters, hyphens, and
518    /// underscores ([A-Za-z0-9-_]). Maximum 64 characters.
519    #[builder(start_fn)]
520    pub id: String,
521    /// Where the sandbox's root filesystem comes from
522    pub code: SandboxCode,
523    /// Private ECR base image an AWS build pulls; `code.image` names only the S3 bundle, so this is
524    /// what the cross-account grant opens. Live only: the grant needs the customer account,
525    /// which registration reports. Absent means the base image is pulled anonymously.
526    #[serde(skip_serializing_if = "Option::is_none")]
527    pub private_base_image: Option<String>,
528    /// Enforced resource ceilings.
529    ///
530    /// Optional because not every platform can enforce them, and a declaration that names none
531    /// takes the platform's own defaults. Naming them on a platform that cannot enforce them is
532    /// rejected at plan time rather than silently ignored.
533    #[serde(skip_serializing_if = "Option::is_none")]
534    pub limits: Option<SandboxLimits>,
535    /// Outbound network policy
536    pub egress: SandboxEgress,
537    /// Have Alien's agent install the declared egress policy before running any image code.
538    /// Unsupported backends refuse this at plan time.
539    #[serde(default, skip_serializing_if = "Option::is_none")]
540    pub privileged_supervisor: Option<SandboxPrivilegedSupervisor>,
541    /// Sandbox lifetime ceiling and idle behaviour.
542    ///
543    /// Stored state written before the rename calls this `session`. A stack state or release
544    /// that old must stay readable, otherwise its deployment can no longer be updated or deleted.
545    #[serde(alias = "session")]
546    pub lifecycle: SandboxLifecyclePolicy,
547    /// Ports eligible for a preview capability. An application reaches its sandbox through the
548    /// provider, so it cannot widen its own ingress at runtime; a holder of a remote binding's
549    /// credentials is bounded by no port condition, which is why a remote sandbox declares none.
550    #[builder(default)]
551    #[serde(default, skip_serializing_if = "Vec::is_empty")]
552    pub preview_ports: Vec<u16>,
553}
554
555/// Whether the artifact being rendered restricts which network modes it accepts.
556///
557/// Cloud setup needs explicit subnets for restricted sandbox connectors. Kubernetes targets do
558/// not emit these cloud backends.
559pub fn restricts_network_mode(stack: &crate::Stack, targets_kubernetes: bool) -> bool {
560    !targets_kubernetes && stack_needs_named_subnets_at_setup(stack)
561}
562
563/// Whether any setup-owned resource forces setup to name subnets.
564///
565/// Restricted sandbox connectors require subnet IDs. Callers rendering an artifact want
566/// [`restricts_network_mode`] instead:
567/// this one answers for the declaration, which on a Kubernetes target is not what gets emitted.
568pub fn stack_needs_named_subnets_at_setup(stack: &crate::Stack) -> bool {
569    stack.resources().any(|(_resource_id, resource)| {
570        resource
571            .config
572            .downcast_ref::<Sandbox>()
573            .is_some_and(|sandbox| !matches!(sandbox.cloud_egress(), SandboxEgress::Allow))
574    })
575}
576
577impl Sandbox {
578    /// The resource type identifier for Sandbox
579    pub const RESOURCE_TYPE: ResourceType = ResourceType::from_static("sandbox");
580
581    /// Returns the sandbox's unique identifier.
582    pub fn id(&self) -> &str {
583        &self.id
584    }
585
586    /// Cloud routing stays open when the agent owns enforcement.
587    pub fn cloud_egress(&self) -> &SandboxEgress {
588        if self.privileged_supervisor.is_some() {
589            &SandboxEgress::Allow
590        } else {
591            &self.egress
592        }
593    }
594
595    /// Startup contract for the privileged agent; never supplied by an exec caller.
596    pub fn supervisor_environment(&self) -> std::collections::BTreeMap<String, String> {
597        let mut env = std::collections::BTreeMap::new();
598        if let Some(supervisor) = &self.privileged_supervisor {
599            env.insert(
600                "ALIEN_SANDBOX_EXEC_UID".to_string(),
601                supervisor.command_uid.to_string(),
602            );
603            env.insert(
604                "ALIEN_SANDBOX_EXEC_GID".to_string(),
605                supervisor.command_uid.to_string(),
606            );
607            env.insert(
608                "ALIEN_SANDBOX_EGRESS".to_string(),
609                serde_json::to_string(&self.egress).expect("egress serializes"),
610            );
611        }
612        env
613    }
614
615    /// The declared ceilings, or the defaults a platform applies when none were named.
616    ///
617    /// Backends want a concrete set: a sandbox with no declared ceilings still runs inside
618    /// whatever the platform gives it, and a backend that had to branch on `None` would end up
619    /// inventing its own default anyway.
620    pub fn resolved_limits(&self) -> SandboxLimits {
621        self.limits.clone().unwrap_or_else(default_limits)
622    }
623
624    /// Validates the declaration against what the target platform can enforce.
625    ///
626    /// Runs at plan time so an unenforceable limit or an unsupported egress mode fails before
627    /// anything is provisioned, rather than at the first exec.
628    pub fn validate_for_platform(&self, platform: Platform) -> Result<()> {
629        let mut capabilities = SandboxCapabilities::for_platform(platform)?;
630        if let Some(supervisor) = &self.privileged_supervisor {
631            if platform != Platform::Aws {
632                return Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
633                    capability: "privilegedSupervisor".to_string(),
634                    platform: platform.to_string(),
635                }));
636            }
637            if supervisor.command_uid == 0 || supervisor.command_uid == u32::MAX {
638                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
639                    resource_id: self.id.clone(),
640                    field: "privilegedSupervisor.commandUid".to_string(),
641                    value: supervisor.command_uid.to_string(),
642                    reason: "must be a non-root Linux uid other than the invalid uid sentinel"
643                        .to_string(),
644                }));
645            }
646            if let SandboxEgress::AllowDomains { domains } = &self.egress {
647                for domain in domains {
648                    let hostname = domain.strip_suffix('.').unwrap_or(domain);
649                    if hostname.len() > 253
650                        || !hostname.split('.').all(|label| {
651                            !label.is_empty()
652                                && label.len() <= 63
653                                && label
654                                    .bytes()
655                                    .all(|c| c.is_ascii_alphanumeric() || c == b'-')
656                                && !label.starts_with('-')
657                                && !label.ends_with('-')
658                        })
659                    {
660                        return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
661                            resource_id: self.id.clone(), field: "egress.domains".to_string(), value: domain.clone(),
662                            reason: "privileged supervisor allowlists require exact DNS hostnames; wildcards are unsupported".to_string(),
663                        }));
664                    }
665                }
666            }
667            capabilities.domain_egress_rules = true;
668        }
669
670        // `alien build` builds an AWS sandbox's base image, so source is a declaration there and
671        // the emitters refuse it only if it reaches them unbuilt. Everywhere else the image is
672        // pulled as declared, and an empty image string would schedule a pod that can never run.
673        if matches!(&self.code, SandboxCode::Source { .. }) && platform != Platform::Aws {
674            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
675                resource_id: self.id.clone(),
676                field: "code".to_string(),
677                value: "source".to_string(),
678                reason: format!(
679                    "no sandbox backend builds an image from source on {platform}; give \
680                     code.image a prebuilt reference"
681                ),
682            }));
683        }
684
685        // Elsewhere `code.image` is pulled directly, so a second reference would be a grant
686        // nothing reads.
687        if self.private_base_image.is_some() && platform != Platform::Aws {
688            return Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
689                capability: "privateBaseImage".to_string(),
690                platform: platform.to_string(),
691            }));
692        }
693
694        // Read before the limits, because the image is declared whether or not any are.
695        if platform == Platform::Azure {
696            self.azure_image()?;
697        }
698
699        let Some(limits) = self.limits.as_ref() else {
700            // Nothing declared, so nothing to enforce and nothing to reject.
701            return self.validate_capabilities(&capabilities, platform);
702        };
703
704        validate_quantity(&self.id, "cpu", &limits.cpu)?;
705        validate_quantity(&self.id, "memory", &limits.memory)?;
706        validate_quantity(&self.id, "disk", &limits.disk)?;
707
708        if let Some(max_processes) = limits.max_processes {
709            if max_processes == 0 {
710                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
711                    resource_id: self.id.clone(),
712                    field: "maxProcesses".to_string(),
713                    value: "0".to_string(),
714                    reason: "a sandbox that may run no processes cannot run code".to_string(),
715                }));
716            }
717            capabilities.require(SandboxCapability::ProcessLimit, platform)?;
718        }
719
720        // Declaring limits a platform ignores is worse than not declaring them: the stack reads
721        // as bounded while the sandbox is not.
722        capabilities.require(SandboxCapability::EnforcedLimits, platform)?;
723
724        if platform == Platform::Azure {
725            self.azure_sandbox_limits()?;
726        }
727
728        if platform == Platform::Aws {
729            // Refused here rather than at emit so a customer sees it while planning, and so both
730            // package formats inherit the same answer.
731            self.microvm_tier()?;
732
733            // The ceiling is Lambda's, and it rejects the run rather than clamping — so a value
734            // outside it would pass planning, render into the package, and fail at the first
735            // sandbox. Kubernetes takes the same field with no such bound, which is why this
736            // sits under the AWS gate rather than on the type.
737            if let Some(seconds) = self.lifecycle.max_lifetime_seconds {
738                if !(1..=AWS_MAX_LIFETIME_SECONDS).contains(&seconds) {
739                    return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
740                        resource_id: self.id.clone(),
741                        field: "maxLifetimeSeconds".to_string(),
742                        value: seconds.to_string(),
743                        reason: format!(
744                            "AWS runs a MicroVM for between 1 and \
745                             {AWS_MAX_LIFETIME_SECONDS} seconds"
746                        ),
747                    }));
748                }
749            }
750        }
751
752        self.validate_capabilities(&capabilities, platform)
753    }
754
755    /// What Azure creates this sandbox from: a catalog name or a registry image, told apart by
756    /// [`classify_azure_sandbox_image`]. Refused while planning, because a value the data plane
757    /// rejects would otherwise surface at the first sandbox, long after the apply.
758    pub fn azure_image(&self) -> Result<AzureSandboxImage<'_>> {
759        let SandboxCode::Image { image } = &self.code else {
760            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
761                resource_id: self.id.clone(),
762                field: "code".to_string(),
763                value: "source".to_string(),
764                reason: "no sandbox backend builds an image from source yet".to_string(),
765            }));
766        };
767
768        classify_azure_sandbox_image(image).ok_or_else(|| {
769            let image = image.trim();
770            AlienError::new(ErrorData::SandboxLimitInvalid {
771                resource_id: self.id.clone(),
772                field: "code.image".to_string(),
773                value: image.to_string(),
774                reason: if image.is_empty() {
775                    "a sandbox has to name an image".to_string()
776                } else {
777                    "Azure creates a sandbox from a catalog name such as 'ubuntu' or from a \
778                     registry image such as 'docker.io/library/python:3.14-slim'"
779                        .to_string()
780                },
781            })
782        })
783    }
784
785    /// Checks the declared ceilings against Azure's sizing rule (the `AZURE_*` constants above).
786    /// Refused at plan time, like [`Self::microvm_tier`], so a bad value is a declaration to fix
787    /// rather than a runtime fault at create.
788    pub fn azure_sandbox_limits(&self) -> Result<()> {
789        let Some(limits) = self.limits.as_ref() else {
790            // Nothing declared means the binding substitutes Alien's own default sizing, which is
791            // inside the rule — asserted where those constants live, since this cannot see them.
792            return Ok(());
793        };
794
795        let refused = |field: &str, value: &str, reason: &str| {
796            AlienError::new(ErrorData::SandboxLimitInvalid {
797                resource_id: self.id.clone(),
798                field: field.to_string(),
799                value: value.to_string(),
800                reason: reason.to_string(),
801            })
802        };
803
804        let cpu_millicores = millicores(&limits.cpu)
805            .ok_or_else(|| refused("cpu", &limits.cpu, "expected cores or millicores"))?;
806
807        // The multiple is checked, not just the range: `333m` sits inside 0.25–16 cores and is
808        // still refused on the wire, so a bounds-only check would pass a declaration that fails
809        // at create.
810        if cpu_millicores % AZURE_CPU_STEP_MILLICORES != 0
811            || !(AZURE_CPU_STEP_MILLICORES..=AZURE_MAX_CPU_MILLICORES).contains(&cpu_millicores)
812        {
813            return Err(refused(
814                "cpu",
815                &limits.cpu,
816                "Azure allocates cpu in steps of 250m from 250m to 16000m",
817            ));
818        }
819
820        // Both ceilings are derived from the cpu, so they cannot be checked before it is known.
821        let memory_ceiling_mib = cpu_millicores * AZURE_MEMORY_MIB_PER_CORE / 1000;
822        let disk_ceiling_mib = cpu_millicores * AZURE_DISK_MIB_PER_CORE / 1000;
823
824        let memory_mib = quantity_mib(&limits.memory)
825            .ok_or_else(|| refused("memory", &limits.memory, "Azure sizes memory in whole MiB"))?;
826        if memory_mib > memory_ceiling_mib {
827            return Err(refused(
828                "memory",
829                &limits.memory,
830                &format!(
831                    "Azure allows at most 2Gi of memory per core, or {memory_ceiling_mib}Mi \
832                          at the declared cpu"
833                ),
834            ));
835        }
836
837        let disk_mib = quantity_mib(&limits.disk)
838            .ok_or_else(|| refused("disk", &limits.disk, "Azure sizes disk in whole MiB"))?;
839        if disk_mib > disk_ceiling_mib {
840            return Err(refused(
841                "disk",
842                &limits.disk,
843                &format!(
844                    "Azure allows at most 20Gi of disk per core, or {disk_ceiling_mib}Mi at \
845                          the declared cpu"
846                ),
847            ));
848        }
849
850        Ok(())
851    }
852
853    /// The MicroVM size that keeps every declared ceiling, or why none does.
854    ///
855    /// AWS sizes are discrete and a running MicroVM bursts to four times its baseline, so the
856    /// only tier that honours a ceiling is one whose peak fits inside it. A declaration no tier
857    /// satisfies is refused: shipping the nearest size would give the customer a sandbox that
858    /// exceeds the bound they wrote down.
859    pub fn microvm_tier(&self) -> Result<MicrovmTier> {
860        let Some(limits) = self.limits.as_ref() else {
861            // Nothing declared: AWS's own default baseline, which is also `default_limits`.
862            return Ok(MICROVM_TIERS[2]);
863        };
864
865        let memory_mib = quantity_mib(&limits.memory).ok_or_else(|| {
866            AlienError::new(ErrorData::SandboxLimitInvalid {
867                resource_id: self.id.clone(),
868                field: "memory".to_string(),
869                value: limits.memory.clone(),
870                reason: "AWS sizes a MicroVM in whole MiB".to_string(),
871            })
872        })?;
873        let disk_mib = quantity_mib(&limits.disk).ok_or_else(|| {
874            AlienError::new(ErrorData::SandboxLimitInvalid {
875                resource_id: self.id.clone(),
876                field: "disk".to_string(),
877                value: limits.disk.clone(),
878                reason: "AWS sizes a MicroVM's disk in whole MiB".to_string(),
879            })
880        })?;
881        let cpu_millicores = millicores(&limits.cpu).ok_or_else(|| {
882            AlienError::new(ErrorData::SandboxLimitInvalid {
883                resource_id: self.id.clone(),
884                field: "cpu".to_string(),
885                value: limits.cpu.clone(),
886                reason: "expected cores or millicores".to_string(),
887            })
888        })?;
889
890        // Memory and disk choose the size; cpu is then checked rather than used to choose.
891        // AWS couples cpu to memory at 2 GB per vCPU, so letting a low cpu ceiling select the
892        // size too would quietly hand back a machine four times smaller than the memory ceiling
893        // asked for, with nothing to indicate it.
894        let sized = |tier: &&MicrovmTier| {
895            tier.peak_memory_mib <= memory_mib && tier.max_disk_mib <= disk_mib
896        };
897
898        let tier = MICROVM_TIERS
899            .iter()
900            .rev()
901            .find(sized)
902            .copied()
903            .ok_or_else(|| {
904                AlienError::new(ErrorData::SandboxLimitInvalid {
905                    resource_id: self.id.clone(),
906                    field: "memory".to_string(),
907                    value: limits.memory.clone(),
908                    reason: format!(
909                        "a Lambda MicroVM bursts to four times its baseline, so the smallest \
910                         ceiling AWS can hold is 2Gi memory with 8Gi disk; '{}' memory and '{}' \
911                         disk fit no size",
912                        limits.memory, limits.disk
913                    ),
914                })
915            })?;
916
917        let required_millicores = i64::from(tier.peak_vcpu) * 1000;
918        if cpu_millicores < required_millicores {
919            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
920                resource_id: self.id.clone(),
921                field: "cpu".to_string(),
922                value: limits.cpu.clone(),
923                reason: format!(
924                    "AWS allocates one vCPU per 2GB, so a MicroVM sized to a '{}' memory ceiling \
925                     reaches {} vCPU; declare cpu '{}' or lower the memory ceiling",
926                    limits.memory, tier.peak_vcpu, tier.peak_vcpu
927                ),
928            }));
929        }
930
931        Ok(tier)
932    }
933
934    /// The capability checks that do not depend on declared limits.
935    fn validate_capabilities(
936        &self,
937        capabilities: &SandboxCapabilities,
938        platform: Platform,
939    ) -> Result<()> {
940        if matches!(self.egress, SandboxEgress::AllowDomains { .. }) {
941            capabilities.require(SandboxCapability::DomainEgressRules, platform)?;
942        }
943
944        // `allow` asks for no restriction, so a backend that ignores it fails loudly on the first
945        // blocked connection. `deny` asks for one, and a backend that ignores it puts untrusted
946        // code on the internet with nothing to notice — so only this direction is gated.
947        // An empty list is not a restriction anyone wrote down: it renders as a deny-all wearing
948        // an allowlist's label, which reads at a glance as the opposite of what it does.
949        if let SandboxEgress::AllowDomains { domains } = &self.egress {
950            if domains.is_empty() {
951                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
952                    resource_id: self.id.clone(),
953                    field: "egress.domains".to_string(),
954                    value: "[]".to_string(),
955                    reason: "an allowlist naming no domain denies everything; declare \
956                             egress: deny if that is what was meant"
957                        .to_string(),
958                }));
959            }
960        }
961
962        if matches!(self.egress, SandboxEgress::Deny) {
963            capabilities.require(SandboxCapability::EgressDeny, platform)?;
964        }
965
966        if !self.preview_ports.is_empty() {
967            capabilities.require(SandboxCapability::Preview, platform)?;
968        }
969
970        if self.lifecycle.idle_pause_seconds.is_some() {
971            capabilities.require(SandboxCapability::PauseResume, platform)?;
972        }
973
974        if self.lifecycle.max_lifetime_seconds.is_some() {
975            capabilities.require(SandboxCapability::SandboxLifetime, platform)?;
976        }
977
978        Ok(())
979    }
980}
981
982/// Ceilings applied when a declaration names none.
983///
984/// Modest on purpose: an undeclared sandbox is one whose author did not think about sizing, and
985/// the safe reading of that is a small box rather than a generous one.
986fn default_limits() -> SandboxLimits {
987    SandboxLimits {
988        cpu: "1".to_string(),
989        memory: "2Gi".to_string(),
990        disk: "8Gi".to_string(),
991        max_processes: None,
992    }
993}
994
995/// Validates a Kubernetes-style resource quantity such as `500m`, `2Gi` or `1`.
996fn validate_quantity(resource_id: &str, field: &str, value: &str) -> Result<()> {
997    let invalid = |reason: &str| {
998        AlienError::new(ErrorData::SandboxLimitInvalid {
999            resource_id: resource_id.to_string(),
1000            field: field.to_string(),
1001            value: value.to_string(),
1002            reason: reason.to_string(),
1003        })
1004    };
1005
1006    let digits_end = value
1007        .find(|c: char| !c.is_ascii_digit() && c != '.')
1008        .unwrap_or(value.len());
1009    let (number, suffix) = value.split_at(digits_end);
1010
1011    let parsed: f64 = number
1012        .parse()
1013        .map_err(|_| invalid("expected a number, optionally followed by a unit suffix"))?;
1014
1015    if parsed <= 0.0 {
1016        return Err(invalid("must be greater than zero"));
1017    }
1018
1019    const SUFFIXES: &[&str] = &["", "m", "k", "M", "G", "T", "Ki", "Mi", "Gi", "Ti"];
1020    if !SUFFIXES.contains(&suffix) {
1021        return Err(invalid(
1022            "unit must be one of m, k, M, G, T, Ki, Mi, Gi, Ti, or absent",
1023        ));
1024    }
1025
1026    Ok(())
1027}
1028
1029/// Splits a quantity into its number and unit suffix.
1030fn split_quantity(value: &str) -> Option<(f64, &str)> {
1031    let trimmed = value.trim();
1032    let digits_end = trimmed
1033        .find(|c: char| !c.is_ascii_digit() && c != '.')
1034        .unwrap_or(trimmed.len());
1035    let (number, suffix) = trimmed.split_at(digits_end);
1036    number.parse().ok().map(|number| (number, suffix))
1037}
1038
1039/// A memory or disk quantity in whole MiB, rounded down.
1040///
1041/// Every suffix `validate_quantity` accepts is handled here. Reading only `Gi` and `Mi` and
1042/// falling back for the rest would turn a declared `4G` into a different size than the customer
1043/// asked for, which for a ceiling means a sandbox larger than its bound.
1044pub fn quantity_mib(value: &str) -> Option<i64> {
1045    let (number, suffix) = split_quantity(value)?;
1046    let bytes = match suffix {
1047        "" => number,
1048        "k" => number * 1e3,
1049        "M" => number * 1e6,
1050        "G" => number * 1e9,
1051        "T" => number * 1e12,
1052        "Ki" => number * 1024.0,
1053        "Mi" => number * 1024.0 * 1024.0,
1054        "Gi" => number * 1024.0 * 1024.0 * 1024.0,
1055        "Ti" => number * 1024.0 * 1024.0 * 1024.0 * 1024.0,
1056        // `m` is a millicore suffix; memory has no use for it.
1057        _ => return None,
1058    };
1059    Some((bytes / (1024.0 * 1024.0)) as i64)
1060}
1061
1062/// A CPU quantity in millicores.
1063pub fn millicores(value: &str) -> Option<i64> {
1064    let (number, suffix) = split_quantity(value)?;
1065    match suffix {
1066        "" => Some((number * 1000.0) as i64),
1067        "m" => Some(number as i64),
1068        _ => None,
1069    }
1070}
1071
1072/// Outputs generated by a successfully provisioned Sandbox parent.
1073#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
1074#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
1075#[serde(rename_all = "camelCase")]
1076pub struct SandboxOutputs {
1077    /// Name of the durable parent that sandboxes are created inside
1078    pub parent_name: String,
1079    /// Platform-specific identifier for the parent (image ARN, sandbox group id, namespace)
1080    #[serde(skip_serializing_if = "Option::is_none")]
1081    pub identifier: Option<String>,
1082    /// Data-plane endpoint sandboxes are created through, where the platform has one
1083    #[serde(skip_serializing_if = "Option::is_none")]
1084    pub endpoint: Option<String>,
1085}
1086
1087impl ResourceOutputsDefinition for SandboxOutputs {
1088    fn get_resource_type(&self) -> ResourceType {
1089        Sandbox::RESOURCE_TYPE
1090    }
1091
1092    fn as_any(&self) -> &dyn Any {
1093        self
1094    }
1095
1096    fn box_clone(&self) -> Box<dyn ResourceOutputsDefinition> {
1097        Box::new(self.clone())
1098    }
1099
1100    fn outputs_eq(&self, other: &dyn ResourceOutputsDefinition) -> bool {
1101        other.as_any().downcast_ref::<SandboxOutputs>() == Some(self)
1102    }
1103
1104    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1105        serde_json::to_value(self)
1106    }
1107}
1108
1109impl ResourceDefinition for Sandbox {
1110    fn get_resource_type(&self) -> ResourceType {
1111        Self::RESOURCE_TYPE
1112    }
1113
1114    fn id(&self) -> &str {
1115        &self.id
1116    }
1117
1118    fn get_dependencies(&self) -> Vec<ResourceRef> {
1119        Vec::new()
1120    }
1121
1122    fn validate_update(&self, new_config: &dyn ResourceDefinition) -> Result<()> {
1123        let new_sandbox = new_config
1124            .as_any()
1125            .downcast_ref::<Sandbox>()
1126            .ok_or_else(|| {
1127                AlienError::new(ErrorData::UnexpectedResourceType {
1128                    resource_id: self.id.clone(),
1129                    expected: Self::RESOURCE_TYPE,
1130                    actual: new_config.get_resource_type(),
1131                })
1132            })?;
1133
1134        if self.id != new_sandbox.id {
1135            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
1136                resource_id: self.id.clone(),
1137                reason: "the 'id' field is immutable".to_string(),
1138            }));
1139        }
1140
1141        Ok(())
1142    }
1143
1144    fn as_any(&self) -> &dyn Any {
1145        self
1146    }
1147
1148    fn as_any_mut(&mut self) -> &mut dyn Any {
1149        self
1150    }
1151
1152    fn box_clone(&self) -> Box<dyn ResourceDefinition> {
1153        Box::new(self.clone())
1154    }
1155
1156    fn resource_eq(&self, other: &dyn ResourceDefinition) -> bool {
1157        other.as_any().downcast_ref::<Sandbox>() == Some(self)
1158    }
1159
1160    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1161        serde_json::to_value(self)
1162    }
1163}
1164
1165/// What an Azure sandbox starts from.
1166#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1167pub enum AzureSandboxImage<'a> {
1168    /// A public catalog disk image, such as `ubuntu`, which the data plane names directly.
1169    Catalog(&'a str),
1170    /// A registry image, which the controller builds into a disk image in the sandbox's group
1171    /// before any sandbox can start from it.
1172    Registry(&'a str),
1173}
1174
1175impl<'a> AzureSandboxImage<'a> {
1176    /// The declared value, trimmed.
1177    pub fn as_str(&self) -> &'a str {
1178        match self {
1179            Self::Catalog(value) | Self::Registry(value) => value,
1180        }
1181    }
1182}
1183
1184/// Label key the controller writes on every disk image it builds, and the provider finds it by.
1185pub const AZURE_DISK_IMAGE_LABEL: &str = "alienImage";
1186
1187/// Classifies a declared `code.image` for Azure, or `None` when it is neither kind. A bare
1188/// `[A-Za-z0-9._-]+` is checked first and always a catalog name; anything else must carry `/`,
1189/// `:` or `@` and parse as an OCI reference.
1190pub fn classify_azure_sandbox_image(image: &str) -> Option<AzureSandboxImage<'_>> {
1191    let image = image.trim();
1192    if image.is_empty() {
1193        return None;
1194    }
1195    if image
1196        .chars()
1197        .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '-'))
1198    {
1199        return Some(AzureSandboxImage::Catalog(image));
1200    }
1201    (image.contains(|c| matches!(c, '/' | ':' | '@')) && OCI_REFERENCE.is_match(image))
1202        .then_some(AzureSandboxImage::Registry(image))
1203}
1204
1205/// The label value naming the disk image built from `reference`: a digest, since a label value
1206/// may not carry a reference's `/`, `:` and `@`.
1207pub fn azure_disk_image_label(reference: &str) -> String {
1208    let digest = format!("{:x}", Sha256::digest(reference.trim().as_bytes()));
1209    digest[..32].to_string()
1210}
1211
1212/// The distribution reference grammar: `[host[:port]/]path[:tag][@digest]`.
1213static OCI_REFERENCE: std::sync::LazyLock<regex::Regex> = std::sync::LazyLock::new(|| {
1214    let domain_component = r"(?:[a-zA-Z0-9]|[a-zA-Z0-9][a-zA-Z0-9-]*[a-zA-Z0-9])";
1215    let domain = format!(r"{domain_component}(?:\.{domain_component})*(?::[0-9]+)?");
1216    let path_component = r"[a-z0-9]+(?:(?:[._]|__|-+)[a-z0-9]+)*";
1217    let name = format!(r"(?:{domain}/)?{path_component}(?:/{path_component})*");
1218    let tag = r"[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}";
1219    let digest = r"[A-Za-z][A-Za-z0-9]*(?:[-_+.][A-Za-z][A-Za-z0-9]*)*:[0-9a-fA-F]{32,}";
1220    regex::Regex::new(&format!(r"^{name}(?::{tag})?(?:@{digest})?$"))
1221        .expect("the OCI reference grammar compiles")
1222});
1223
1224/// The one token a sandbox bundle URI may carry, replaced with the deploying region.
1225///
1226/// AWS builds a MicroVM image only from a bucket in the image's own region, so a vendor
1227/// publishing to every supported region needs one stored URI that resolves per region.
1228pub const BUNDLE_REGION_TOKEN: &str = "{region}";
1229
1230/// A bundle URI split around its region token, or carried whole when it has none.
1231#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1232pub enum BundleUri<'a> {
1233    /// No token: emitted exactly as it is today.
1234    Literal(&'a str),
1235    /// The text either side of the token, for an emitter to rejoin around its own region
1236    /// expression.
1237    Regional { before: &'a str, after: &'a str },
1238}
1239
1240/// The prefix a rebuild's new key still sits under: everything above the file name and the
1241/// version segment beneath it. `None` when nothing sits there, meaning no prefix can be granted
1242/// without also granting objects a rebuild never reads. Shared so both emitters agree on it.
1243pub fn stable_bundle_key_prefix(key: &str) -> Option<&str> {
1244    let (above_file, _) = key.rsplit_once('/')?;
1245    let (above_version, _) = above_file.rsplit_once('/')?;
1246    Some(above_version)
1247}
1248
1249/// Reads a sandbox bundle URI, refusing anything an image build would only reject later.
1250///
1251/// The token is accepted in the bucket alone. A key-position token would name an object that does
1252/// not exist, and any other brace is a typo that would otherwise reach S3 verbatim and fail ~160s
1253/// into the build — which is the failure this whole check exists to move to plan time.
1254pub fn parse_bundle_uri(uri: &str) -> std::result::Result<BundleUri<'_>, String> {
1255    let path = uri
1256        .strip_prefix("s3://")
1257        .ok_or_else(|| format!("'{uri}' is not an s3:// URI"))?;
1258    let (bucket, key) = path
1259        .split_once('/')
1260        .ok_or_else(|| format!("'{uri}' names a bucket with no object key"))?;
1261
1262    // Both emitters interpolate this path into the build role's resource ARN, where `*` and `?`
1263    // are IAM wildcards rather than literal characters. S3 accepts them in a key, so a bundle
1264    // published under one would silently widen the grant past the bundle it names.
1265    if path.contains('*') || path.contains('?') {
1266        return Err(format!(
1267            "'{uri}' carries an IAM wildcard; the bundle's path is interpolated into the build \
1268             role's grant, so '*' and '?' would widen it past the bundle"
1269        ));
1270    }
1271
1272    if key.contains('{') || key.contains('}') {
1273        return Err(format!(
1274            "'{uri}' places a token in the object key; {BUNDLE_REGION_TOKEN} is accepted in the \
1275             bucket name alone"
1276        ));
1277    }
1278
1279    let Some((before, after)) = bucket.split_once(BUNDLE_REGION_TOKEN) else {
1280        if bucket.contains('{') || bucket.contains('}') {
1281            return Err(format!(
1282                "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the \
1283                 only one"
1284            ));
1285        }
1286        return Ok(BundleUri::Literal(uri));
1287    };
1288
1289    if after.contains(BUNDLE_REGION_TOKEN) {
1290        return Err(format!("'{uri}' repeats {BUNDLE_REGION_TOKEN}"));
1291    }
1292    if before.contains('{') || before.contains('}') || after.contains('{') || after.contains('}') {
1293        return Err(format!(
1294            "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the only one"
1295        ));
1296    }
1297
1298    Ok(BundleUri::Regional {
1299        before: &uri[.."s3://".len() + before.len()],
1300        after: &uri["s3://".len() + before.len() + BUNDLE_REGION_TOKEN.len()..],
1301    })
1302}
1303
1304/// Where a private ECR image's region comes from.
1305#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1306pub enum EcrImageRegion<'a> {
1307    /// A region named in the host.
1308    Literal(&'a str),
1309    /// [`BUNDLE_REGION_TOKEN`] in the host: the region the deployment renders.
1310    Deployment,
1311}
1312
1313/// The ECR repository a private image reference is pulled from.
1314#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1315pub struct EcrImageRepository<'a> {
1316    pub account_id: &'a str,
1317    pub region: EcrImageRegion<'a>,
1318    /// The repository name, which may carry `/`; never a tag or digest.
1319    pub repository: &'a str,
1320}
1321
1322impl EcrImageRepository<'_> {
1323    /// The repository's ARN, with `region` standing in for a region the host leaves to the
1324    /// deployment. The partition is the deployment's: a GovCloud host ends `.amazonaws.com` too.
1325    pub fn arn(&self, partition: &str, region: &str) -> String {
1326        let region = match self.region {
1327            EcrImageRegion::Literal(region) => region,
1328            EcrImageRegion::Deployment => region,
1329        };
1330        format!(
1331            "arn:{partition}:ecr:{region}:{}:repository/{}",
1332            self.account_id, self.repository
1333        )
1334    }
1335}
1336
1337/// Reads `privateBaseImage` as the one repository a build role may pull from.
1338///
1339/// The name is interpolated into an IAM ARN, so it is held to ECR's own repository grammar: that
1340/// refuses `*` and `?`, which would widen the grant, and `$` and braces, which a CloudFormation
1341/// `Sub` or a Terraform template would read as an expression.
1342pub fn parse_ecr_image_repository(
1343    image: &str,
1344) -> std::result::Result<EcrImageRepository<'_>, String> {
1345    let refuse = |reason: &str| format!("privateBaseImage '{image}' {reason}");
1346    let (host, path) = image
1347        .split_once('/')
1348        .ok_or_else(|| refuse("names no repository"))?;
1349    let (account_id, rest) = host.split_once(".dkr.ecr.").ok_or_else(|| {
1350        refuse("is not served by a private ECR registry (<account>.dkr.ecr.<region>.amazonaws.com)")
1351    })?;
1352    // `.com.cn` first: a China host ends with the shorter suffix too.
1353    let region = rest
1354        .strip_suffix(".amazonaws.com.cn")
1355        .or_else(|| rest.strip_suffix(".amazonaws.com"))
1356        .ok_or_else(|| refuse("is not served by a private ECR registry (<account>.dkr.ecr.<region>.amazonaws.com)"))?;
1357    if account_id.len() != 12 || !account_id.bytes().all(|b| b.is_ascii_digit()) {
1358        return Err(refuse("names no 12-digit account in its registry host"));
1359    }
1360    let region = if region == BUNDLE_REGION_TOKEN {
1361        EcrImageRegion::Deployment
1362    } else if !region.is_empty()
1363        && region
1364            .bytes()
1365            .all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'-')
1366    {
1367        EcrImageRegion::Literal(region)
1368    } else {
1369        return Err(refuse(&format!(
1370            "names no region in its registry host; give one or {BUNDLE_REGION_TOKEN}"
1371        )));
1372    };
1373
1374    let repository = match path.split_once('@') {
1375        Some((repository, _digest)) => repository,
1376        None => match path.rsplit_once('/') {
1377            Some((parent, last)) => match last.split_once(':') {
1378                Some((name, _tag)) => &path[..parent.len() + 1 + name.len()],
1379                None => path,
1380            },
1381            None => path.split_once(':').map_or(path, |(name, _tag)| name),
1382        },
1383    };
1384    let valid_segment = |segment: &str| {
1385        !segment.is_empty()
1386            && segment.bytes().all(|b| {
1387                b.is_ascii_lowercase() || b.is_ascii_digit() || matches!(b, b'.' | b'_' | b'-')
1388            })
1389    };
1390    if !repository.split('/').all(valid_segment) {
1391        return Err(refuse(
1392            "names a repository outside ECR's grammar (lowercase letters, digits, '.', '_', '-', and '/' between them)",
1393        ));
1394    }
1395    Ok(EcrImageRepository {
1396        account_id,
1397        region,
1398        repository,
1399    })
1400}
1401
1402#[cfg(test)]
1403mod tests {
1404    use super::*;
1405
1406    #[test]
1407    fn private_database_setup_accepts_the_default_network() {
1408        for lifecycle in [
1409            crate::ResourceLifecycle::Frozen,
1410            crate::ResourceLifecycle::Live,
1411        ] {
1412            let stack = crate::Stack::new("database".to_string())
1413                .add(
1414                    crate::Postgres::new("metadata".to_string()).build(),
1415                    lifecycle,
1416                )
1417                .build();
1418            assert!(!restricts_network_mode(&stack, false));
1419            assert!(!restricts_network_mode(&stack, true));
1420        }
1421        assert!(!restricts_network_mode(
1422            &crate::Stack::new("empty".to_string()).build(),
1423            false,
1424        ));
1425    }
1426
1427    /// The vectors' `repositoryKey`: registry host and repository, never the tag or digest, so a
1428    /// new tag keeps the key. A reference the parser refuses stays whole.
1429    fn repository_key(image: &str) -> String {
1430        match parse_ecr_image_repository(image) {
1431            Ok(parsed) => {
1432                let host = image.split_once('/').map_or(image, |(host, _)| host);
1433                format!("{host}/{}", parsed.repository)
1434            }
1435            Err(_) => image.to_string(),
1436        }
1437    }
1438
1439    #[test]
1440    fn a_private_base_image_names_one_repository() {
1441        let vectors: serde_json::Value = serde_json::from_str(include_str!(
1442            "../../tests/fixtures/ecr-image-repository-parity.json"
1443        ))
1444        .expect("the ECR repository vectors must be JSON");
1445        let field = |case: &serde_json::Value, name: &str| {
1446            case[name]
1447                .as_str()
1448                .unwrap_or_else(|| panic!("vector must carry {name}: {case}"))
1449                .to_string()
1450        };
1451
1452        for case in vectors["accepted"].as_array().expect("accepted vectors") {
1453            let image = field(case, "image");
1454            let region = field(case, "region");
1455            let region = if region == BUNDLE_REGION_TOKEN {
1456                EcrImageRegion::Deployment
1457            } else {
1458                EcrImageRegion::Literal(&region)
1459            };
1460            assert_eq!(
1461                parse_ecr_image_repository(&image),
1462                Ok(EcrImageRepository {
1463                    account_id: &field(case, "accountId"),
1464                    region,
1465                    repository: &field(case, "repository"),
1466                }),
1467                "{image}"
1468            );
1469            assert_eq!(
1470                repository_key(&image),
1471                field(case, "repositoryKey"),
1472                "{image}"
1473            );
1474        }
1475
1476        for case in vectors["refused"].as_array().expect("refused vectors") {
1477            let image = field(case, "image");
1478            assert!(
1479                parse_ecr_image_repository(&image).is_err(),
1480                "{image} must be refused"
1481            );
1482            assert_eq!(
1483                repository_key(&image),
1484                field(case, "repositoryKey"),
1485                "{image}"
1486            );
1487        }
1488    }
1489
1490    #[test]
1491    fn a_repository_arn_takes_the_deployment_region_only_where_the_host_leaves_it() {
1492        let regional =
1493            parse_ecr_image_repository("123456789012.dkr.ecr.{region}.amazonaws.com/team/base:1")
1494                .expect("parses");
1495        let pinned =
1496            parse_ecr_image_repository("123456789012.dkr.ecr.eu-west-1.amazonaws.com/team/base:1")
1497                .expect("parses");
1498        assert_eq!(
1499            regional.arn("aws-us-gov", "us-gov-west-1"),
1500            "arn:aws-us-gov:ecr:us-gov-west-1:123456789012:repository/team/base"
1501        );
1502        assert_eq!(
1503            pinned.arn("aws", "us-east-1"),
1504            "arn:aws:ecr:eu-west-1:123456789012:repository/team/base"
1505        );
1506    }
1507
1508    /// A wildcard reaching the grant would widen it past the bundle, and it widens the Frozen
1509    /// object grant as readily as the Live prefix — both interpolate the path into the ARN.
1510    #[test]
1511    fn a_uri_carrying_an_iam_wildcard_is_refused() {
1512        for uri in [
1513            "s3://acme/team-*/v1/bundle.zip",
1514            "s3://acme/sandbox-bundle/f00d/bundle?.zip",
1515            "s3://acme-*/sandbox-bundle/f00d/bundle.zip",
1516        ] {
1517            let error = parse_bundle_uri(uri).expect_err("a wildcard must be refused");
1518            assert!(error.contains("IAM wildcard"), "for {uri}: {error}");
1519        }
1520
1521        parse_bundle_uri("s3://acme/sandbox-bundle/f00d/bundle.zip")
1522            .expect("an ordinary key still parses");
1523    }
1524
1525    /// The rule both package formats grant by, pinned here rather than in either. The
1526    /// near-misses the cases separate: the key's first segment grants objects a rebuild never
1527    /// reads, and the object's own directory pins the version segment that moves.
1528    #[test]
1529    fn a_grantable_prefix_stops_above_the_segment_that_moves() {
1530        assert_eq!(
1531            stable_bundle_key_prefix("sandbox-bundle/f00dcafe/bundle.zip"),
1532            Some("sandbox-bundle")
1533        );
1534        assert_eq!(
1535            stable_bundle_key_prefix("artifacts/team-a/sandbox/f00dcafe/bundle.zip"),
1536            Some("artifacts/team-a/sandbox"),
1537            "a deeper key narrows the prefix, it never widens to the first segment"
1538        );
1539
1540        // Nothing sits above the version segment, so no prefix a moved bundle stays inside
1541        // exists. Emitting the object grant instead installs a role that denies the next rebuild.
1542        assert_eq!(stable_bundle_key_prefix("agents/bundle.zip"), None);
1543        assert_eq!(stable_bundle_key_prefix("bundle.zip"), None);
1544    }
1545
1546    fn sandbox_with(egress: SandboxEgress, preview_ports: Vec<u16>) -> Sandbox {
1547        Sandbox::new("agent-sbx".to_string())
1548            .code(SandboxCode::Image {
1549                image: "ubuntu".to_string(),
1550            })
1551            .limits(SandboxLimits {
1552                cpu: "1".to_string(),
1553                memory: "2Gi".to_string(),
1554                disk: "20Gi".to_string(),
1555                max_processes: None,
1556            })
1557            .egress(egress)
1558            .lifecycle(SandboxLifecyclePolicy {
1559                max_lifetime_seconds: None,
1560                idle_pause_seconds: None,
1561            })
1562            .preview_ports(preview_ports)
1563            .build()
1564    }
1565
1566    /// A URI with no token must come back whole, because every bundle configured today has none
1567    /// and emitting one differently would change every existing customer's template.
1568    #[test]
1569    fn a_uri_without_a_token_is_carried_whole() {
1570        assert_eq!(
1571            parse_bundle_uri("s3://acme-artifacts-us-east-2/agents/bundle.zip"),
1572            Ok(BundleUri::Literal(
1573                "s3://acme-artifacts-us-east-2/agents/bundle.zip"
1574            ))
1575        );
1576    }
1577
1578    /// The split has to rejoin to the original with the region in place, or an emitter builds a
1579    /// URI that is subtly not the one the vendor configured.
1580    #[test]
1581    fn a_regional_uri_splits_either_side_of_the_token() {
1582        let BundleUri::Regional { before, after } =
1583            parse_bundle_uri("s3://acme-artifacts-{region}/agents/bundle.zip")
1584                .expect("the token is accepted in the bucket")
1585        else {
1586            panic!("a bucket-position token must split");
1587        };
1588
1589        assert_eq!(before, "s3://acme-artifacts-");
1590        assert_eq!(after, "/agents/bundle.zip");
1591        assert_eq!(
1592            format!("{before}us-east-2{after}"),
1593            "s3://acme-artifacts-us-east-2/agents/bundle.zip",
1594            "the halves must rejoin to the URI the vendor meant"
1595        );
1596    }
1597
1598    /// Each of these reaches S3 verbatim and dies ~160s into an image build if it is not refused
1599    /// here, which is the whole reason this runs at plan time.
1600    #[test]
1601    fn a_token_this_build_cannot_resolve_is_refused() {
1602        for uri in [
1603            "s3://acme-artifacts-{regio}/bundle.zip",
1604            "s3://acme-artifacts/{region}/bundle.zip",
1605            "s3://acme-artifacts-{region}-{region}/bundle.zip",
1606            "s3://acme-artifacts/bundle-{version}.zip",
1607            "s3://acme}-artifacts-{region}/bundle.zip",
1608            "s3://acme{-artifacts-{region}/bundle.zip",
1609        ] {
1610            assert!(
1611                parse_bundle_uri(uri).is_err(),
1612                "'{uri}' must be refused before it can reach an image build"
1613            );
1614        }
1615    }
1616
1617    #[test]
1618    fn resource_type_is_stable() {
1619        assert_eq!(Sandbox::RESOURCE_TYPE.as_ref(), "sandbox");
1620    }
1621
1622    #[test]
1623    fn capability_sets_are_per_platform() {
1624        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1625        assert!(
1626            gcp.reconnect,
1627            "generation from the container boot id makes a sandbox reachable across processes"
1628        );
1629        assert!(!gcp.preview);
1630        assert!(gcp.enforced_limits);
1631
1632        let azure = SandboxCapabilities::for_platform(Platform::Azure).expect("azure is supported");
1633        assert!(azure.files, "every backend moves files");
1634        assert!(gcp.files);
1635        // Azure is the only backend whose egress policy matches on host pattern, and the only
1636        // one where `deny` and a hostname list are the same object.
1637        assert!(azure.domain_egress_rules);
1638        assert!(azure.egress_deny);
1639        // The data plane honours a continuous cpu/memory/disk surface and refuses anything
1640        // outside `n×250m`, with the rule in the message.
1641        assert!(azure.enforced_limits);
1642        assert!(azure.pause_resume);
1643        // Both stay false for reasons that are not "unbuilt": a snapshot id has nothing to
1644        // consume it on any backend, and an Azure port's auth is anonymous or a human allowlist,
1645        // neither of which is a port-scoped credential.
1646        assert!(!azure.snapshot);
1647        assert!(!azure.preview);
1648
1649        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1650        assert!(!aws.snapshot, "AWS has no user-callable sandbox snapshot");
1651        assert!(aws.pause_resume);
1652
1653        let k8s =
1654            SandboxCapabilities::for_platform(Platform::Kubernetes).expect("k8s is supported");
1655        assert!(
1656            !k8s.preview,
1657            "the sandbox-scoped ingress gateway does not exist yet"
1658        );
1659    }
1660
1661    /// Whether the process supervising a command is a separate identity from the command.
1662    ///
1663    /// Values are measured, not inferred. AWS: the agent runs as uid 0 with
1664    /// `CapEff: 00000000a80425fb` and `setuid`s the command to uid 60000, so the two differ.
1665    /// Kubernetes: the sandbox pod pins `run_as_user: 65534` on both pod and container with
1666    /// `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no uid split is
1667    /// possible (`kubernetes_spec.rs`). Local: `docker exec` runs as the workload uid while the
1668    /// manager supervises from the host. Azure and Agent Platform have no in-sandbox supervisor.
1669    #[test]
1670    fn supervisor_isolation_is_per_platform() {
1671        let value = |platform| {
1672            SandboxCapabilities::for_platform(platform)
1673                .expect("supported")
1674                .supervisor_isolation
1675        };
1676
1677        assert!(
1678            value(Platform::Aws),
1679            "root agent setuids the command to 60000"
1680        );
1681        assert!(
1682            value(Platform::Local),
1683            "the supervisor is on the host, outside the container"
1684        );
1685        assert!(
1686            !value(Platform::Kubernetes),
1687            "a single pinned uid cannot be split"
1688        );
1689        assert!(!value(Platform::Azure), "no Alien process runs the command");
1690        assert!(
1691            !value(Platform::Gcp),
1692            "no separate supervisor identity runs the command"
1693        );
1694    }
1695
1696    /// The point of the field: AWS and GCP report the *same* `supervisor_pid_namespace` (neither
1697    /// has `CAP_SYS_ADMIN`), so that axis alone reads them as equivalent. They are not — AWS
1698    /// separates the command's identity from the supervisor's and Agent Platform does not.
1699    #[test]
1700    fn supervisor_isolation_separates_aws_from_a_subprocess_backend() {
1701        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1702        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1703
1704        assert_eq!(
1705            aws.supervisor_pid_namespace, gcp.supervisor_pid_namespace,
1706            "the older axis cannot tell them apart"
1707        );
1708        assert!(
1709            aws.supervisor_isolation,
1710            "AWS setuids the command off the supervisor"
1711        );
1712        assert!(
1713            !gcp.supervisor_isolation,
1714            "the command runs under no separate supervisor identity"
1715        );
1716    }
1717
1718    /// The Agent Platform row, each value against the behaviour it was measured from. `reconnect`
1719    /// is the tripwire: it is `true` only because `generation` is derived from the container boot
1720    /// id read through the agent's health op, so a caller detects a replaced container instead of
1721    /// reconnecting to a blank one. It is also the body of the `Platform::Gcp` arm, asserted below.
1722    #[test]
1723    fn gcp_agent_platform_row_matches_measured_backend() {
1724        let row = SandboxCapabilities::gcp_agent_platform();
1725
1726        assert!(row.files, "agent file ops move over the sandbox envelope");
1727        assert!(
1728            row.reconnect,
1729            "generation is derived from the container boot id, so a sandbox is reachable across \
1730             processes"
1731        );
1732        assert!(
1733            !row.preview,
1734            "the only ingress is :execute; no port-scoped capability"
1735        );
1736        assert!(
1737            !row.pause_resume,
1738            ":resume can return a fresh container, so a pause keeps no state"
1739        );
1740        assert!(
1741            !row.snapshot,
1742            "the create path never sends a snapshot, so none is reachable through the trait"
1743        );
1744        assert!(
1745            !row.domain_egress_rules,
1746            "VPC and DNS peering is not a hostname allowlist"
1747        );
1748        assert!(
1749            row.egress_deny,
1750            "a declared deny blocks both egress and DNS"
1751        );
1752        assert!(
1753            row.enforced_limits,
1754            "ceilings are enforced, by terminating the sandbox on breach"
1755        );
1756        assert!(!row.process_limit, "no process-count ceiling is observed");
1757        assert!(row.sandbox_lifetime, "ttl maps to a sandbox expireTime");
1758        assert!(!row.supervisor_pid_namespace, "no PID-namespace isolation");
1759        assert!(
1760            !row.supervisor_isolation,
1761            "the command is not run under a separate supervisor identity"
1762        );
1763
1764        // Agent Platform is the registered GCP backend, so the arm returns exactly this row.
1765        let live = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1766        assert_eq!(
1767            live, row,
1768            "the Platform::Gcp arm is the Agent Platform capability row"
1769        );
1770    }
1771
1772    #[test]
1773    fn platforms_without_a_backend_are_an_error_not_an_empty_set() {
1774        let error = SandboxCapabilities::for_platform(Platform::Machines)
1775            .expect_err("Machines has no sandbox backend");
1776        assert_eq!(error.code, "SANDBOX_PLATFORM_UNSUPPORTED");
1777    }
1778
1779    #[test]
1780    fn unsupported_capability_names_platform_and_capability() {
1781        let capabilities = SandboxCapabilities::for_platform(Platform::Gcp).expect("supported");
1782        let error = capabilities
1783            .require(SandboxCapability::Preview, Platform::Gcp)
1784            .expect_err("GCP has no preview");
1785
1786        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1787        let rendered = error.to_string();
1788        assert!(
1789            rendered.contains("preview"),
1790            "names the capability: {rendered}"
1791        );
1792        assert!(rendered.contains("gcp"), "names the platform: {rendered}");
1793    }
1794
1795    /// Azure matches on hostname; AWS and Kubernetes match CIDRs, and Local and GCP have a
1796    /// switch rather than a filter. Accepting a hostname list on those four would leave a stack
1797    /// reading as restricted while the sandbox reaches the whole internet.
1798    #[test]
1799    fn a_hostname_allowlist_is_refused_everywhere_it_would_be_approximated() {
1800        let sandbox = sandbox_with(
1801            SandboxEgress::AllowDomains {
1802                domains: vec!["example.com".to_string()],
1803            },
1804            vec![],
1805        );
1806
1807        for platform in [
1808            Platform::Aws,
1809            Platform::Gcp,
1810            Platform::Kubernetes,
1811            Platform::Local,
1812        ] {
1813            let error = sandbox
1814                .validate_for_platform(platform)
1815                .expect_err("only Azure expresses a hostname allowlist");
1816            assert_eq!(
1817                error.code, "SANDBOX_CAPABILITY_UNSUPPORTED",
1818                "on {platform:?}"
1819            );
1820        }
1821
1822        assert!(
1823            SandboxCapabilities::for_platform(Platform::Azure)
1824                .expect("supported")
1825                .domain_egress_rules,
1826            "Azure's egress policy matches on host pattern"
1827        );
1828    }
1829
1830    /// `deny` is the declaration that carries a security promise, so a backend that cannot keep
1831    /// it has to refuse rather than accept it and run the code with open egress.
1832    #[test]
1833    fn a_denied_egress_is_refused_where_it_would_not_be_enforced() {
1834        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1835
1836        // GCP is asserted at the capability rather than through validation: this sandbox declares
1837        // ceilings GCP cannot enforce, so it is refused for a reason unrelated to egress.
1838        assert!(
1839            SandboxCapabilities::for_platform(Platform::Gcp)
1840                .expect("supported")
1841                .egress_deny
1842        );
1843
1844        for platform in [Platform::Aws, Platform::Kubernetes, Platform::Local] {
1845            sandbox
1846                .validate_for_platform(platform)
1847                .expect("deny is enforced here");
1848        }
1849
1850        // Declares no ceilings, which Azure refuses for its own reason, so this isolates egress.
1851        let egress_only = Sandbox::new("sbx".to_string())
1852            .code(SandboxCode::Image {
1853                image: "alpine".to_string(),
1854            })
1855            .egress(SandboxEgress::Deny)
1856            .lifecycle(SandboxLifecyclePolicy {
1857                max_lifetime_seconds: None,
1858                idle_pause_seconds: None,
1859            })
1860            .build();
1861
1862        egress_only
1863            .validate_for_platform(Platform::Azure)
1864            .expect("Azure creates the sandbox under a Deny policy with full inspection");
1865    }
1866
1867    /// A sandbox naming no ceilings is valid on every platform and still resolves to a concrete
1868    /// set. The rule Azure applies to ceilings that *are* declared is pinned separately, by
1869    /// `azure_sizes_follow_the_rule_the_data_plane_states`.
1870    #[test]
1871    fn a_sandbox_declaring_no_ceilings_takes_the_platforms_own() {
1872        let undeclared = Sandbox::new("sbx".to_string())
1873            .code(SandboxCode::Image {
1874                image: "alpine".to_string(),
1875            })
1876            .egress(SandboxEgress::Deny)
1877            .lifecycle(SandboxLifecyclePolicy {
1878                max_lifetime_seconds: None,
1879                idle_pause_seconds: None,
1880            })
1881            .build();
1882
1883        undeclared
1884            .validate_for_platform(Platform::Azure)
1885            .expect("a sandbox naming no ceilings takes the platform's own");
1886
1887        // A backend still gets a concrete set, so nothing downstream has to invent one.
1888        assert_eq!(undeclared.resolved_limits().cpu, "1");
1889    }
1890
1891    /// Pins Azure's own sizing rule: `250m`, `1500m` and `4000m` are valid, `32000m` and `333m`
1892    /// are refused. Checked at plan time so a bad value is a declaration to fix, not a package
1893    /// that renders without error and dies at the first sandbox.
1894    #[test]
1895    fn azure_sizes_follow_the_rule_the_data_plane_states() {
1896        let sized = |cpu: &str, memory: &str, disk: &str| {
1897            let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1898            let limits = sandbox
1899                .limits
1900                .as_mut()
1901                .expect("the fixture declares limits");
1902            limits.cpu = cpu.to_string();
1903            limits.memory = memory.to_string();
1904            limits.disk = disk.to_string();
1905            sandbox.validate_for_platform(Platform::Azure)
1906        };
1907
1908        sized("250m", "512Mi", "5120Mi").expect("the smallest step the data plane accepts");
1909        sized("4000m", "8192Mi", "40960Mi").expect("cpu, memory and disk are all honoured");
1910        sized("16000m", "32Gi", "320Gi").expect("the top of the range");
1911
1912        // Same off-step case `azure_sandbox_limits` checks the multiple for, not just the range.
1913        let off_step = sized("333m", "512Mi", "5120Mi").expect_err("333m is not a step of 250m");
1914        assert_eq!(off_step.code, "SANDBOX_LIMIT_INVALID", "{off_step}");
1915        assert!(off_step.to_string().contains("cpu"), "{off_step}");
1916
1917        let too_big = sized("32000m", "64Gi", "640Gi").expect_err("32 cores is over the ceiling");
1918        assert_eq!(too_big.code, "SANDBOX_LIMIT_INVALID", "{too_big}");
1919
1920        // Both ceilings are derived from the cpu, so the same memory passes at one size and fails
1921        // at another - which is what makes them worth checking rather than bounding absolutely.
1922        sized("1000m", "2Gi", "20Gi").expect("2Gi is exactly one core's worth");
1923        let over_memory = sized("250m", "2Gi", "5120Mi").expect_err("2Gi needs a full core");
1924        assert_eq!(over_memory.code, "SANDBOX_LIMIT_INVALID", "{over_memory}");
1925        assert!(over_memory.to_string().contains("memory"), "{over_memory}");
1926
1927        let over_disk = sized("250m", "512Mi", "20Gi").expect_err("20Gi needs a full core");
1928        assert!(over_disk.to_string().contains("disk"), "{over_disk}");
1929    }
1930
1931    #[test]
1932    fn preview_ports_require_the_preview_capability() {
1933        let sandbox = sandbox_with(SandboxEgress::Deny, vec![8080]);
1934
1935        sandbox
1936            .validate_for_platform(Platform::Aws)
1937            .expect("AWS mints a port-scoped JWE");
1938
1939        let error = sandbox
1940            .validate_for_platform(Platform::Kubernetes)
1941            .expect_err("Kubernetes preview is deferred");
1942        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1943    }
1944
1945    /// A grant nothing reads is the silent no-op the capability contract exists to prevent, and
1946    /// here it is worse than useless: the reader would take it for a base image that needs
1947    /// authenticating while the platform pulls `code.image` itself.
1948    #[test]
1949    fn a_private_base_image_is_refused_off_aws() {
1950        let mut sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1951        sandbox.code = SandboxCode::Image {
1952            image: "s3://acme-artifacts/agents/bundle.zip".to_string(),
1953        };
1954        sandbox.private_base_image =
1955            Some("123456789012.dkr.ecr.{region}.amazonaws.com/acme:tag".to_string());
1956
1957        sandbox
1958            .validate_for_platform(Platform::Aws)
1959            .expect("AWS builds its image from a bundle, so a base image sits behind code.image");
1960
1961        for platform in [
1962            Platform::Gcp,
1963            Platform::Azure,
1964            Platform::Kubernetes,
1965            Platform::Local,
1966        ] {
1967            let error = sandbox
1968                .validate_for_platform(platform)
1969                .expect_err("a backend that builds no image must refuse a base image for one");
1970            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1971            assert!(
1972                error.to_string().contains("privateBaseImage"),
1973                "the refusal must name the field the user declared: {error}"
1974            );
1975        }
1976    }
1977
1978    #[test]
1979    fn gcp_accepts_a_sandbox_declaring_enforced_limits() {
1980        let sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1981        sandbox
1982            .validate_for_platform(Platform::Gcp)
1983            .expect("Agent Platform enforces declared ceilings, by terminating on breach");
1984    }
1985
1986    #[test]
1987    fn invalid_quantities_are_rejected_with_the_offending_field() {
1988        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1989        sandbox
1990            .limits
1991            .as_mut()
1992            .expect("the fixture declares limits")
1993            .memory = "2Gb".to_string();
1994
1995        let error = sandbox
1996            .validate_for_platform(Platform::Aws)
1997            .expect_err("Gb is not a valid suffix");
1998        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1999        assert!(error.to_string().contains("memory"));
2000
2001        sandbox
2002            .limits
2003            .as_mut()
2004            .expect("the fixture declares limits")
2005            .memory = "2Gi".to_string();
2006        sandbox
2007            .limits
2008            .as_mut()
2009            .expect("the fixture declares limits")
2010            .cpu = "0".to_string();
2011        let error = sandbox
2012            .validate_for_platform(Platform::Aws)
2013            .expect_err("zero cpu is not a ceiling");
2014        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2015    }
2016
2017    #[test]
2018    fn zero_max_processes_is_rejected() {
2019        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2020        sandbox
2021            .limits
2022            .as_mut()
2023            .expect("the fixture declares limits")
2024            .max_processes = Some(0);
2025
2026        let error = sandbox
2027            .validate_for_platform(Platform::Local)
2028            .expect_err("a sandbox must be able to run at least one process");
2029        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2030        assert!(error.to_string().contains("maxProcesses"));
2031    }
2032
2033    /// A process ceiling needs a container runtime. Kubernetes sets one per node rather than per
2034    /// pod, and neither MicroVMs nor Azure sandboxes expose one, so accepting the declaration
2035    /// anywhere else would mean carrying a bound nothing applies.
2036    #[test]
2037    fn a_process_ceiling_is_accepted_only_where_a_runtime_can_apply_it() {
2038        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2039        sandbox
2040            .limits
2041            .as_mut()
2042            .expect("the fixture declares limits")
2043            .max_processes = Some(256);
2044
2045        sandbox
2046            .validate_for_platform(Platform::Local)
2047            .expect("Docker takes a pids limit");
2048
2049        for platform in [Platform::Aws, Platform::Azure, Platform::Kubernetes] {
2050            let error = sandbox
2051                .validate_for_platform(platform)
2052                .expect_err("a process ceiling nothing applies must be refused");
2053            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2054        }
2055    }
2056
2057    /// Lambda rejects a run outside 1–28,800 rather than clamping it, so a value beyond that
2058    /// would pass planning, render into the package, and fail at the first sandbox. Kubernetes
2059    /// takes the same field with no such bound, so the check is AWS's alone.
2060    #[test]
2061    fn a_lifetime_aws_would_reject_is_refused_while_planning() {
2062        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2063
2064        for seconds in [0, 28_801, 100_000] {
2065            sandbox.lifecycle.max_lifetime_seconds = Some(seconds);
2066            let error = sandbox
2067                .validate_for_platform(Platform::Aws)
2068                .expect_err("a lifetime outside what AWS runs is refused");
2069            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "{seconds}s");
2070
2071            // Kubernetes has no such ceiling, so the same declaration is fine there.
2072            sandbox
2073                .validate_for_platform(Platform::Kubernetes)
2074                .expect("the kubelet takes any activeDeadlineSeconds");
2075        }
2076
2077        sandbox.lifecycle.max_lifetime_seconds = Some(28_800);
2078        sandbox
2079            .validate_for_platform(Platform::Aws)
2080            .expect("the ceiling itself is allowed");
2081    }
2082
2083    /// An image Azure can take neither as a catalog name nor as a registry image is refused while
2084    /// planning, not at the first sandbox.
2085    #[test]
2086    fn an_image_azure_cannot_pull_is_refused_while_planning() {
2087        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2088        // Azure enforces no declared ceiling, so a sandbox carrying limits is refused before the
2089        // image is ever read.
2090        sandbox.limits = None;
2091
2092        // A digest too short to be one, blanks, a space and a query string.
2093        for image in ["ubuntu@sha256:abc", "", "   ", "ubuntu latest", "ubuntu?x"] {
2094            sandbox.code = SandboxCode::Image {
2095                image: image.to_string(),
2096            };
2097            let error = sandbox
2098                .validate_for_platform(Platform::Azure)
2099                .expect_err("an image Azure has nowhere to put is refused");
2100            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "image '{image}'");
2101        }
2102
2103        for image in ["ubuntu", "ubuntu-22.04", "debian_slim"] {
2104            sandbox.code = SandboxCode::Image {
2105                image: image.to_string(),
2106            };
2107            sandbox
2108                .validate_for_platform(Platform::Azure)
2109                .unwrap_or_else(|error| panic!("'{image}' is a catalog name: {error}"));
2110        }
2111
2112        for image in [
2113            "ubuntu:24.04",
2114            "ghcr.io/myorg/sandbox:latest",
2115            "docker.io/library/python:3.14-slim",
2116            "localhost:5000/team/agent@sha256:51dafde81dbdb6ebde285137a295cf18a47ca95234fe388a343719cb97305b3d",
2117        ] {
2118            sandbox.code = SandboxCode::Image {
2119                image: image.to_string(),
2120            };
2121            sandbox
2122                .validate_for_platform(Platform::Azure)
2123                .unwrap_or_else(|error| panic!("'{image}' is a registry image: {error}"));
2124        }
2125
2126        // Surrounding space is trimmed rather than carried into the create body.
2127        sandbox.code = SandboxCode::Image {
2128            image: " ubuntu ".to_string(),
2129        };
2130        assert_eq!(
2131            sandbox
2132                .azure_image()
2133                .expect("a padded name is still a name"),
2134            AzureSandboxImage::Catalog("ubuntu")
2135        );
2136    }
2137
2138    /// Every value the catalog allowlist `[A-Za-z0-9._-]+` accepts stays a catalog name, so a
2139    /// declaration that planned under it keeps its meaning. Exhaustive to three characters, then a
2140    /// fixed-seed sample of longer ones, some with surrounding space.
2141    #[test]
2142    fn every_value_the_catalog_allowlist_accepted_is_still_a_catalog_name() {
2143        const ALPHABET: &[u8] =
2144            b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789._-";
2145
2146        let assert_catalog = |value: &str| {
2147            assert_eq!(
2148                classify_azure_sandbox_image(value),
2149                Some(AzureSandboxImage::Catalog(value.trim())),
2150                "'{value}'"
2151            );
2152        };
2153
2154        for a in ALPHABET {
2155            assert_catalog(&String::from_utf8(vec![*a]).unwrap());
2156            for b in ALPHABET {
2157                assert_catalog(&String::from_utf8(vec![*a, *b]).unwrap());
2158                for c in ALPHABET {
2159                    assert_catalog(&String::from_utf8(vec![*a, *b, *c]).unwrap());
2160                }
2161            }
2162        }
2163
2164        let mut state: u64 = 0x9E37_79B9_7F4A_7C15;
2165        let mut next = || {
2166            state ^= state << 13;
2167            state ^= state >> 7;
2168            state ^= state << 17;
2169            state
2170        };
2171        for _ in 0..20_000 {
2172            let len = 4 + (next() % 60) as usize;
2173            let mut value: String = (0..len)
2174                .map(|_| ALPHABET[(next() % ALPHABET.len() as u64) as usize] as char)
2175                .collect();
2176            if next() % 4 == 0 {
2177                value = format!("  {value}\t");
2178            }
2179            assert_catalog(&value);
2180        }
2181    }
2182
2183    /// The label is how the provider finds the image the controller built, so both must derive
2184    /// the same one; it also has to fit a label value, which a raw reference does not.
2185    #[test]
2186    fn the_disk_image_label_is_stable_and_label_safe() {
2187        let label = azure_disk_image_label("docker.io/library/python:3.14-slim");
2188        assert_eq!(
2189            label,
2190            azure_disk_image_label(" docker.io/library/python:3.14-slim ")
2191        );
2192        assert_ne!(
2193            label,
2194            azure_disk_image_label("docker.io/library/python:3.13-slim")
2195        );
2196        assert_eq!(label.len(), 32);
2197        assert!(label.chars().all(|c| c.is_ascii_hexdigit()), "{label}");
2198    }
2199
2200    /// A deadline is accepted only where the platform itself terminates on it — the kubelet's
2201    /// `activeDeadlineSeconds` and Lambda's `maximumDurationInSeconds`. Everywhere else it would
2202    /// need a reaper that does not exist, so it is refused rather than accepted and dropped.
2203    #[test]
2204    fn a_sandbox_deadline_is_accepted_only_where_the_platform_applies_it() {
2205        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2206        sandbox.lifecycle.max_lifetime_seconds = Some(3600);
2207
2208        sandbox
2209            .validate_for_platform(Platform::Kubernetes)
2210            .expect("the kubelet enforces activeDeadlineSeconds");
2211        sandbox
2212            .validate_for_platform(Platform::Aws)
2213            .expect("Lambda terminates the MicroVM at maximumDurationInSeconds");
2214
2215        for platform in [Platform::Azure, Platform::Local] {
2216            let error = sandbox
2217                .validate_for_platform(platform)
2218                .expect_err("a deadline nothing applies must be refused");
2219            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2220        }
2221    }
2222
2223    /// A MicroVM bursts to four times its baseline with no way to opt out, so a ceiling is kept
2224    /// by choosing the size whose *peak* fits inside it. Sizing by baseline would hand back a
2225    /// sandbox that can reach four times what the customer declared.
2226    #[test]
2227    fn an_aws_size_is_chosen_so_its_peak_stays_inside_the_declared_ceiling() {
2228        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2229        let tier = sandbox
2230            .microvm_tier()
2231            .expect("2Gi/1cpu/20Gi is satisfiable");
2232
2233        assert_eq!(
2234            tier.peak_memory_mib, 2048,
2235            "the peak is the declared ceiling"
2236        );
2237        assert_eq!(
2238            tier.baseline_memory_mib, 512,
2239            "which is a quarter of it as the baseline"
2240        );
2241        assert!(tier.max_disk_mib <= 20 * 1024);
2242    }
2243
2244    /// AWS allocates one vCPU per 2GB, so a cpu ceiling below what the memory ceiling implies
2245    /// cannot be honoured together with it. Letting cpu choose the size instead would hand back a
2246    /// machine four times smaller than the memory asked for, with nothing to indicate it.
2247    #[test]
2248    fn a_cpu_ceiling_below_what_the_memory_implies_is_refused_not_quietly_downsized() {
2249        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2250        {
2251            let limits = sandbox
2252                .limits
2253                .as_mut()
2254                .expect("the fixture declares limits");
2255            limits.cpu = "1".to_string();
2256            limits.memory = "8Gi".to_string();
2257        }
2258
2259        let error = sandbox
2260            .microvm_tier()
2261            .expect_err("1 cpu and 8Gi cannot both be ceilings on AWS");
2262        assert!(
2263            error.to_string().contains("4 vCPU"),
2264            "the refusal must say what the memory ceiling implies: {error}"
2265        );
2266
2267        sandbox
2268            .limits
2269            .as_mut()
2270            .expect("the fixture declares limits")
2271            .cpu = "4".to_string();
2272        let tier = sandbox.microvm_tier().expect("4 cpu matches 8Gi");
2273        assert_eq!(tier.peak_memory_mib, 8192);
2274    }
2275
2276    /// Below AWS's smallest peak there is no size that holds the ceiling, and rounding up to the
2277    /// nearest one would silently exceed it.
2278    #[test]
2279    fn an_aws_ceiling_smaller_than_any_size_is_refused_rather_than_rounded() {
2280        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2281        sandbox
2282            .limits
2283            .as_mut()
2284            .expect("the fixture declares limits")
2285            .memory = "1Gi".to_string();
2286
2287        let error = sandbox
2288            .validate_for_platform(Platform::Aws)
2289            .expect_err("no MicroVM size peaks at or below 1Gi");
2290        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2291        assert!(
2292            error.to_string().contains("2Gi"),
2293            "the refusal must say what the smallest holdable ceiling is: {error}"
2294        );
2295    }
2296
2297    /// `alien build` builds an AWS sandbox's base image, so source is a declaration there. On
2298    /// every other platform the image is pulled as declared, and an unbuilt source would schedule
2299    /// a pod that can never run, so the refusal still has to happen at plan time.
2300    #[test]
2301    fn source_code_is_refused_off_aws_rather_than_producing_a_broken_manifest() {
2302        let sandbox = Sandbox::new("agent".to_string())
2303            .code(SandboxCode::Source {
2304                src: "./sandbox".to_string(),
2305                toolchain: ToolchainConfig::Docker {
2306                    dockerfile: None,
2307                    build_args: None,
2308                    target: None,
2309                },
2310            })
2311            .egress(SandboxEgress::Deny)
2312            .lifecycle(SandboxLifecyclePolicy {
2313                max_lifetime_seconds: None,
2314                idle_pause_seconds: None,
2315            })
2316            .build();
2317
2318        sandbox
2319            .validate_for_platform(Platform::Aws)
2320            .expect("an AWS sandbox base image is built by `alien build`");
2321
2322        for platform in [
2323            Platform::Azure,
2324            Platform::Gcp,
2325            Platform::Kubernetes,
2326            Platform::Local,
2327        ] {
2328            let error = sandbox
2329                .validate_for_platform(platform)
2330                .expect_err("no backend builds a sandbox image from source here");
2331            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2332            assert!(
2333                error.to_string().contains("code.image"),
2334                "the refusal must say what to write instead: {error}"
2335            );
2336            assert!(
2337                error.to_string().contains(&platform.to_string()),
2338                "the refusal must name the platform that cannot build it: {error}"
2339            );
2340        }
2341    }
2342
2343    /// `validate_quantity` accepts nine suffixes. Reading only `Gi` and `Mi` would size a
2344    /// declared `4G` as though it were `4Gi`, which for a ceiling means exceeding it.
2345    #[test]
2346    fn every_accepted_unit_converts_rather_than_falling_back() {
2347        assert_eq!(quantity_mib("2Gi"), Some(2048));
2348        assert_eq!(quantity_mib("512Mi"), Some(512));
2349        assert_eq!(quantity_mib("4G"), Some(3814));
2350        assert_eq!(quantity_mib("1Ti"), Some(1024 * 1024));
2351        assert_eq!(millicores("1"), Some(1000));
2352        assert_eq!(millicores("500m"), Some(500));
2353    }
2354
2355    #[test]
2356    fn privileged_supervisor_fixes_identity_and_requires_an_enforceable_backend() {
2357        let mut sandbox = sandbox_with(
2358            SandboxEgress::AllowDomains {
2359                domains: vec!["example.com".to_string()],
2360            },
2361            vec![],
2362        );
2363        sandbox.privileged_supervisor = Some(SandboxPrivilegedSupervisor { command_uid: 60001 });
2364        sandbox
2365            .validate_for_platform(Platform::Aws)
2366            .expect("AWS can enforce the policy in the agent");
2367        assert_eq!(sandbox.cloud_egress(), &SandboxEgress::Allow);
2368        assert_eq!(
2369            sandbox.supervisor_environment()["ALIEN_SANDBOX_EXEC_UID"],
2370            "60001"
2371        );
2372        for platform in [
2373            Platform::Local,
2374            Platform::Kubernetes,
2375            Platform::Azure,
2376            Platform::Gcp,
2377        ] {
2378            let error = sandbox
2379                .validate_for_platform(platform)
2380                .expect_err("cannot grant a privilege boundary the backend lacks");
2381            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2382            assert!(error.message.contains("privilegedSupervisor"));
2383        }
2384        for uid in [0, u32::MAX] {
2385            sandbox.privileged_supervisor.as_mut().unwrap().command_uid = uid;
2386            let error = sandbox
2387                .validate_for_platform(Platform::Aws)
2388                .expect_err("invalid uid must fail at plan time");
2389            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2390        }
2391    }
2392
2393    #[test]
2394    fn unknown_fields_are_rejected() {
2395        let json = r#"{
2396            "id": "sbx",
2397            "code": {"type": "image", "image": "ubuntu:24.04"},
2398            "limits": {"cpu": "1", "memory": "2Gi", "disk": "20Gi"},
2399            "egress": {"mode": "deny"},
2400            "lifecycle": {},
2401            "unexpected": true
2402        }"#;
2403
2404        serde_json::from_str::<Sandbox>(json).expect_err("deny_unknown_fields must reject");
2405    }
2406
2407    #[test]
2408    fn serialization_roundtrips() {
2409        let sandbox = sandbox_with(
2410            SandboxEgress::AllowDomains {
2411                domains: vec!["example.com".to_string()],
2412            },
2413            vec![8080, 9090],
2414        );
2415
2416        let json = serde_json::to_string(&sandbox).expect("serializes");
2417        let restored: Sandbox = serde_json::from_str(&json).expect("deserializes");
2418        assert_eq!(sandbox, restored);
2419    }
2420
2421    #[test]
2422    fn id_is_immutable_across_updates() {
2423        let original = sandbox_with(SandboxEgress::Deny, vec![]);
2424        let renamed = Sandbox::new("other".to_string())
2425            .code(SandboxCode::Image {
2426                image: "ubuntu".to_string(),
2427            })
2428            .limits(
2429                original
2430                    .limits
2431                    .clone()
2432                    .expect("the fixture declares limits"),
2433            )
2434            .egress(SandboxEgress::Deny)
2435            .lifecycle(SandboxLifecyclePolicy {
2436                max_lifetime_seconds: None,
2437                idle_pause_seconds: None,
2438            })
2439            .build();
2440
2441        original
2442            .validate_update(&original.clone())
2443            .expect("an unchanged config is a valid update");
2444        original
2445            .validate_update(&renamed)
2446            .expect_err("renaming a sandbox is not an update");
2447    }
2448
2449    /// Azure declares an idle-pause policy but not a wall-clock ceiling.
2450    ///
2451    /// The two travel together in `SandboxLifecyclePolicy` and are gated separately on purpose:
2452    /// Azure pauses on idle and has no maximum lifetime, so accepting one and refusing the
2453    /// other is the honest split rather than an inconsistency.
2454    #[test]
2455    fn azure_takes_an_idle_policy_and_still_refuses_a_lifetime_ceiling() {
2456        let with_policy = |lifecycle: SandboxLifecyclePolicy| {
2457            Sandbox::new("sbx".to_string())
2458                .code(SandboxCode::Image {
2459                    image: "ubuntu".to_string(),
2460                })
2461                .egress(SandboxEgress::Allow)
2462                .lifecycle(lifecycle)
2463                .build()
2464                .validate_for_platform(Platform::Azure)
2465        };
2466
2467        with_policy(SandboxLifecyclePolicy {
2468            max_lifetime_seconds: None,
2469            idle_pause_seconds: Some(900),
2470        })
2471        .expect("Azure pauses a sandbox on idle");
2472
2473        let error = with_policy(SandboxLifecyclePolicy {
2474            max_lifetime_seconds: Some(3600),
2475            idle_pause_seconds: None,
2476        })
2477        .expect_err("Azure has no wall-clock ceiling to enforce one with");
2478        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2479        assert!(
2480            error.message.contains("sandboxLifetime"),
2481            "names the capability: {}",
2482            error.message
2483        );
2484    }
2485
2486    /// An allowlist naming nothing is a deny-all wearing an allowlist's label.
2487    ///
2488    /// It renders as a `Deny` default with no rules — the shape the Azure provider adds a
2489    /// catch-all to avoid — and a reader scanning the declaration sees "allowDomains" and reads
2490    /// the opposite of what it does.
2491    #[test]
2492    fn an_allowlist_with_no_domains_is_refused() {
2493        let declared = |domains: Vec<String>| {
2494            Sandbox::new("sbx".to_string())
2495                .code(SandboxCode::Image {
2496                    image: "ubuntu".to_string(),
2497                })
2498                .egress(SandboxEgress::AllowDomains { domains })
2499                .lifecycle(SandboxLifecyclePolicy {
2500                    max_lifetime_seconds: None,
2501                    idle_pause_seconds: None,
2502                })
2503                .build()
2504                .validate_for_platform(Platform::Azure)
2505        };
2506
2507        let error = declared(vec![]).expect_err("an empty allowlist must be refused");
2508        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2509
2510        declared(vec!["api.example.com".to_string()])
2511            .expect("a named domain is what an allowlist is for");
2512    }
2513
2514    /// The two expressible modes map to the boolean; a host list maps to nothing so the caller has
2515    /// to refuse rather than silently pick a side.
2516    #[test]
2517    fn internet_access_switch_maps_only_the_two_expressible_modes() {
2518        assert_eq!(SandboxEgress::Allow.internet_access_switch(), Some(true));
2519        assert_eq!(SandboxEgress::Deny.internet_access_switch(), Some(false));
2520        assert_eq!(
2521            SandboxEgress::AllowDomains {
2522                domains: vec!["api.example.com".to_string()]
2523            }
2524            .internet_access_switch(),
2525            None,
2526            "a host list has no boolean and must not be approximated"
2527        );
2528    }
2529
2530    /// A sandbox resource as a stack state recorded it before `session` became `lifecycle` and
2531    /// `idleSuspendSeconds` became `idlePauseSeconds`. Such a state is still what a deployment
2532    /// that old holds, and reading it is the first step of updating or deleting that deployment.
2533    #[test]
2534    fn stack_state_written_before_the_lifecycle_rename_still_reads() {
2535        let stored: crate::StackResourceState = serde_json::from_value(serde_json::json!({
2536            "type": "sandbox",
2537            "config": {
2538                "id": "sandbox",
2539                "code": {
2540                    "type": "image",
2541                    "image": "s3://example-bundles/analysis/v1/bundle.zip"
2542                },
2543                "type": "sandbox",
2544                "egress": { "mode": "allow" },
2545                "session": { "maxLifetimeSeconds": 28800, "idleSuspendSeconds": 600 }
2546            },
2547            "status": "running",
2548            "outputs": {
2549                "type": "sandbox",
2550                "identifier": "arn:aws:lambda:us-east-2:123456789012:microvm-image:example-sandbox",
2551                "parentName": "arn:aws:lambda:us-east-2:123456789012:microvm-image:example-sandbox"
2552            },
2553            "lifecycle": "frozen",
2554            "dependencies": [
2555                { "id": "management", "type": "remote-stack-management" },
2556                { "id": "access", "type": "resource-access" }
2557            ],
2558            "controllerPlatform": "aws"
2559        }))
2560        .expect("a stack state written before the rename must still deserialize");
2561
2562        let expected = Sandbox::new("sandbox".to_string())
2563            .code(SandboxCode::Image {
2564                image: "s3://example-bundles/analysis/v1/bundle.zip".to_string(),
2565            })
2566            .egress(SandboxEgress::Allow)
2567            .lifecycle(SandboxLifecyclePolicy {
2568                max_lifetime_seconds: Some(28_800),
2569                idle_pause_seconds: Some(600),
2570            })
2571            .build();
2572        assert_eq!(stored.config.downcast_ref::<Sandbox>(), Some(&expected));
2573
2574        let rewritten = serde_json::to_value(&stored.config).expect("the sandbox serializes");
2575        assert_eq!(
2576            rewritten["lifecycle"],
2577            serde_json::json!({ "maxLifetimeSeconds": 28800, "idlePauseSeconds": 600 }),
2578            "the next write stores the current names"
2579        );
2580        assert!(rewritten.get("session").is_none());
2581    }
2582}