Skip to main content

alien_core/resources/
sandbox.rs

1//! Sandbox resource for running untrusted code in an isolated environment.
2//!
3//! A Sandbox is a session-oriented resource: the declaration provisions a durable parent, and
4//! the application creates and destroys individual sessions through its binding at runtime.
5//!
6//! The capability set differs per platform and is published rather than assumed. Calling an
7//! unsupported capability is a typed error naming both the platform and the capability, so a
8//! portable application can branch on `SandboxCapabilities` before it calls.
9
10use crate::error::{ErrorData, Result};
11use crate::resource::{ResourceDefinition, ResourceOutputsDefinition, ResourceRef, ResourceType};
12use crate::resources::ToolchainConfig;
13use crate::Platform;
14use alien_error::AlienError;
15use bon::Builder;
16use serde::{Deserialize, Serialize};
17use std::any::Any;
18use std::fmt::Debug;
19
20/// Specifies where the sandbox's root filesystem comes from.
21#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
22#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
23#[serde(rename_all = "camelCase", tag = "type")]
24pub enum SandboxCode {
25    /// A prebuilt container image used as the sandbox root filesystem.
26    #[serde(rename_all = "camelCase")]
27    Image {
28        /// Image reference (e.g. `ubuntu:24.04`, `ghcr.io/myorg/sandbox:latest`).
29        ///
30        /// Two backends narrow it in opposite directions: AWS wants an `s3://` bundle, Azure a
31        /// bare catalog name such as `ubuntu`. Each refuses the other's shape while planning.
32        image: String,
33    },
34    /// Source built into a sandbox image at deploy time.
35    #[serde(rename_all = "camelCase")]
36    Source {
37        /// The source directory to build from
38        src: String,
39        /// Toolchain configuration with type-safe options
40        toolchain: ToolchainConfig,
41    },
42}
43
44/// Hard ceilings enforced on a sandbox session.
45///
46/// These are limits, not scheduling requests. Untrusted code does not respect a hint, so every
47/// field is enforced by the platform and a platform that cannot enforce one is rejected at plan
48/// time rather than silently ignoring it.
49#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
50#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
51#[serde(rename_all = "camelCase", deny_unknown_fields)]
52pub struct SandboxLimits {
53    /// CPU ceiling in cores or millicores (e.g. `"1"`, `"500m"`)
54    pub cpu: String,
55    /// Memory ceiling (e.g. `"2Gi"`, `"512Mi"`)
56    pub memory: String,
57    /// Disk ceiling (e.g. `"20Gi"`)
58    pub disk: String,
59    /// Maximum number of processes, which bounds fork bombs.
60    ///
61    /// Optional because only a container runtime has the primitive: Kubernetes sets a pid ceiling
62    /// per node, not per pod, and neither AWS MicroVMs nor Azure sandboxes expose one. Declaring
63    /// it on a platform that cannot apply it is refused at plan time.
64    #[serde(default, skip_serializing_if = "Option::is_none")]
65    pub max_processes: Option<u32>,
66}
67
68/// One of the five sizes a Lambda MicroVM can be built at.
69///
70/// AWS has no ceiling knob: `minimumMemoryInMiB` sets a *baseline* and a running MicroVM bursts
71/// vertically to four times it with no way to opt out. A declared ceiling is therefore honoured by
72/// picking the tier whose **peak** stays inside it, not the tier whose baseline matches it.
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74pub struct MicrovmTier {
75    /// What `minimumMemoryInMiB` is set to.
76    pub baseline_memory_mib: i64,
77    /// The most memory the MicroVM can reach, in MiB.
78    pub peak_memory_mib: i64,
79    /// The most vCPU the MicroVM can reach.
80    pub peak_vcpu: u32,
81    /// The most disk the MicroVM can use, in MiB.
82    pub max_disk_mib: i64,
83}
84
85/// The published sizes, smallest first. Baseline memory to vCPU is 2 GB per vCPU, peak is four
86/// times baseline, and disk is fixed per tier rather than independently selectable.
87/// Longest life AWS will run a MicroVM for, from `RunMicrovm`'s `maximumDurationInSeconds`.
88const AWS_MAX_SESSION_LIFETIME_SECONDS: u32 = 28_800;
89
90/// Azure's session sizing rule, quoted from the data plane's own refusal of an oversized request:
91/// *CPU must be n×250m for n=1..64 (0.25–16 cores); Memory ≤ cores × 2Gi; Disk ≤ cores × 20Gi*.
92const AZURE_CPU_STEP_MILLICORES: i64 = 250;
93const AZURE_MAX_CPU_MILLICORES: i64 = 16_000;
94const AZURE_MEMORY_MIB_PER_CORE: i64 = 2 * 1024;
95const AZURE_DISK_MIB_PER_CORE: i64 = 20 * 1024;
96
97const MICROVM_TIERS: &[MicrovmTier] = &[
98    MicrovmTier {
99        baseline_memory_mib: 512,
100        peak_memory_mib: 2048,
101        peak_vcpu: 1,
102        max_disk_mib: 8192,
103    },
104    MicrovmTier {
105        baseline_memory_mib: 1024,
106        peak_memory_mib: 4096,
107        peak_vcpu: 2,
108        max_disk_mib: 8192,
109    },
110    MicrovmTier {
111        baseline_memory_mib: 2048,
112        peak_memory_mib: 8192,
113        peak_vcpu: 4,
114        max_disk_mib: 8192,
115    },
116    MicrovmTier {
117        baseline_memory_mib: 4096,
118        peak_memory_mib: 16384,
119        peak_vcpu: 8,
120        max_disk_mib: 16384,
121    },
122    MicrovmTier {
123        baseline_memory_mib: 8192,
124        peak_memory_mib: 32768,
125        peak_vcpu: 16,
126        max_disk_mib: 32768,
127    },
128];
129
130/// Outbound network policy for a sandbox.
131#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
132#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
133#[serde(rename_all = "camelCase", tag = "mode")]
134pub enum SandboxEgress {
135    /// No outbound network access.
136    ///
137    /// Routed traffic only. Link-local is not outbound and no backend's egress control reaches
138    /// it, so this is not a boundary against instance metadata.
139    Deny,
140    /// Unrestricted outbound access to the public internet, and none to private ranges or the
141    /// deployment's own network.
142    ///
143    /// Link-local carries the same exception as `Deny`. AWS and Kubernetes deliver both halves.
144    /// Azure and GCP deliver the first only: one matches host patterns and the other is a single
145    /// switch, so neither can name an address range to exclude.
146    Allow,
147    /// Outbound access only to the listed hostnames.
148    ///
149    /// Azure alone expresses it: its egress proxy matches on host pattern. The others filter by
150    /// CIDR or carry a single switch, and both would approximate the list rather than keep it.
151    #[serde(rename_all = "camelCase")]
152    AllowDomains {
153        /// Hostnames the sandbox may reach
154        domains: Vec<String>,
155    },
156}
157
158impl SandboxEgress {
159    /// The single outbound switch for a backend that has no host matcher, or `None` for a mode a
160    /// boolean cannot carry.
161    ///
162    /// `AllowDomains` needs a host list, so it maps to nothing and each caller refuses it in its
163    /// own error naming the sandbox. One source for what a mode means, so a template and a session
164    /// cannot disagree on it.
165    pub fn internet_access_switch(&self) -> Option<bool> {
166        match self {
167            SandboxEgress::Allow => Some(true),
168            SandboxEgress::Deny => Some(false),
169            SandboxEgress::AllowDomains { .. } => None,
170        }
171    }
172}
173
174/// How long a session may live and when it is suspended.
175#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
176#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
177#[serde(rename_all = "camelCase", deny_unknown_fields)]
178pub struct SandboxSessionPolicy {
179    /// Wall-clock ceiling on a single session, after which the platform terminates it.
180    ///
181    /// Optional because not every backend has the primitive: Kubernetes has
182    /// `activeDeadlineSeconds` and AWS `maximumDurationInSeconds`, while neither Azure nor Local
183    /// expose one, so declaring a ceiling there is refused at plan time rather than accepted and
184    /// never applied. AWS caps it at 8 hours.
185    #[serde(default, skip_serializing_if = "Option::is_none")]
186    pub max_lifetime_seconds: Option<u32>,
187    /// Idle period after which the session is suspended, where the platform supports it
188    #[serde(skip_serializing_if = "Option::is_none")]
189    pub idle_suspend_seconds: Option<u32>,
190}
191
192/// What a platform's sandbox backend can actually do.
193///
194/// Published so portable code can branch before calling rather than discovering a gap through
195/// an error. Every field here corresponds to a capability that at least one platform lacks;
196/// create, exec and terminate are the guaranteed floor and are therefore not listed.
197#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
198#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
199#[serde(rename_all = "camelCase", deny_unknown_fields)]
200pub struct SandboxCapabilities {
201    /// Files can be moved in and out of a session
202    pub files: bool,
203    /// A later call can reach a session created by an earlier one
204    pub reconnect: bool,
205    /// A command can be started, polled and cancelled across separate calls, so it outlives the
206    /// one that started it. False where nothing inside the session owns the process in between.
207    pub jobs: bool,
208    /// An authenticated, port-scoped capability to reach a service inside the sandbox
209    pub preview: bool,
210    /// Session state can be suspended and resumed
211    pub suspend_resume: bool,
212    /// A session's full state can be captured and used to create another
213    pub snapshot: bool,
214    /// Egress can be restricted to a hostname allowlist
215    pub domain_egress_rules: bool,
216    /// Whether a declared `deny` is actually enforced, rather than accepted and dropped
217    pub egress_deny: bool,
218    /// The platform enforces the declared cpu, memory and disk ceilings
219    pub enforced_limits: bool,
220    /// The platform can cap how many processes a session runs
221    pub process_limit: bool,
222    /// The platform terminates a session at a declared wall-clock deadline
223    pub session_lifetime: bool,
224    /// A command runs in its own PID namespace and cannot see or signal the agent's processes.
225    ///
226    /// Only where an agent runs as root. Creating the namespace needs `CAP_SYS_ADMIN`, and the
227    /// Kubernetes sandbox pod drops every capability — which is also what denies `ptrace` by
228    /// construction, so granting it there would remove a lock to add one.
229    pub supervisor_pid_namespace: bool,
230    /// The process supervising a command is a different identity from the command.
231    ///
232    /// False where a command runs as the agent's own user: it can then read the supervisor's
233    /// environment and signal it. Separate from `supervisorPidNamespace`, which is about
234    /// visibility rather than identity — a backend can have one without the other.
235    pub supervisor_isolation: bool,
236}
237
238impl SandboxCapabilities {
239    /// Returns what the given platform's sandbox backend supports.
240    ///
241    /// Errors for platforms with no sandbox backend, rather than returning an all-false set —
242    /// "every capability is missing" and "this platform has no sandboxes" are different
243    /// conditions and an application should not have to tell them apart by inspection.
244    pub fn for_platform(platform: Platform) -> Result<Self> {
245        match platform {
246            Platform::Aws => Ok(Self {
247                files: true,
248                reconnect: true,
249                jobs: true,
250                preview: true,
251                suspend_resume: true,
252                snapshot: false,
253                domain_egress_rules: false,
254                egress_deny: true,
255                enforced_limits: true,
256                // Nothing in the API bounds process count.
257                process_limit: false,
258                // `maximumDurationInSeconds` on `RunMicrovm`, which Lambda enforces by
259                // terminating the MicroVM. Capped at 8 hours by the service.
260                session_lifetime: true,
261                // Measured, not assumed: the agent inside a Lambda MicroVM runs as uid 0 with
262                // `CapEff: 00000000a80425fb`, the standard container default set, which excludes
263                // `CAP_SYS_ADMIN`. It can drop privilege (`CAP_SETUID`/`CAP_SETGID` are held) and
264                // it cannot create a namespace. No backend offers this today.
265                supervisor_pid_namespace: false,
266                // The agent runs as uid 0 and `setuid`s the command to uid 60000, so the command
267                // runs under a different identity than the process supervising it.
268                supervisor_isolation: true,
269            }),
270            Platform::Azure => Ok(Self::azure()),
271            Platform::Gcp => Ok(Self::gcp_agent_platform()),
272            // Preview needs a gateway that validates a session-and-port capability, and that
273            // gateway does not exist yet.
274            Platform::Kubernetes => Ok(Self {
275                files: true,
276                reconnect: true,
277                jobs: true,
278                preview: false,
279                suspend_resume: false,
280                snapshot: false,
281                domain_egress_rules: false,
282                egress_deny: true,
283                enforced_limits: true,
284                // A pid ceiling is a kubelet setting per node, not a pod field.
285                process_limit: false,
286                // `activeDeadlineSeconds` on the pod, which the kubelet enforces.
287                session_lifetime: true,
288                // The pod drops every capability, including the `CAP_SYS_ADMIN` the agent would
289                // need to unshare. That is also what denies `ptrace`, so this stays false rather
290                // than the pod being weakened to make it true.
291                supervisor_pid_namespace: false,
292                // The pod pins one uid (`run_as_user: 65534` on both pod and container) with
293                // `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no
294                // process can setuid to split the command off from a supervisor. No uid split is
295                // possible, so none exists.
296                supervisor_isolation: false,
297            }),
298            Platform::Local => Ok(Self {
299                files: true,
300                reconnect: true,
301                // Nothing runs inside the session: the manager drives Docker from outside it.
302                jobs: false,
303                preview: true,
304                suspend_resume: false,
305                snapshot: false,
306                domain_egress_rules: false,
307                egress_deny: true,
308                enforced_limits: true,
309                // Docker's `--pids-limit`.
310                process_limit: true,
311                session_lifetime: false,
312                // Local has no in-sandbox agent: the manager drives Docker from outside, so
313                // there is no supervisor inside the sandbox to isolate from.
314                supervisor_pid_namespace: false,
315                // The supervisor is the manager on the host, outside the container entirely, and
316                // `docker exec` runs the command as the workload uid — a different identity by
317                // construction.
318                supervisor_isolation: true,
319            }),
320            Platform::Machines | Platform::Test => {
321                Err(AlienError::new(ErrorData::SandboxPlatformUnsupported {
322                    platform: platform.to_string(),
323                }))
324            }
325        }
326    }
327
328    /// What the Azure sandbox backend supports; the body of the `Platform::Azure` arm.
329    pub fn azure() -> Self {
330        Self {
331            files: true,
332            reconnect: true,
333            // No Alien process runs inside the session to own a command between two calls.
334            jobs: false,
335            // A sandbox port carries a URL and an auth config, and the auth config offers two
336            // things: anonymous, or Entra ID with an allowlist of human email addresses.
337            // Neither is a credential scoped to a port for a fixed time, which is what a
338            // preview capability is. Returning the anonymous URL would publish the port.
339            preview: false,
340            suspend_resume: true,
341            // False for a client reason, not a cloud one: this client has no snapshot call, and
342            // `CreateSessionRequest` has no field to consume the id it would return. Also
343            // unclaimed: Microsoft does not garbage-collect snapshots, so an id is a bill that grows.
344            snapshot: false,
345            domain_egress_rules: true,
346            egress_deny: true,
347            // Enforced inside the session, not at create: an over-allocation raises `MemoryError`
348            // while the sandbox keeps running. `azure_session_limits` checks the continuous
349            // sizing rule at plan time instead of matching a tier.
350            enforced_limits: true,
351            process_limit: false,
352            // Auto-suspend and auto-delete exist; a wall-clock ceiling does not. Accepting
353            // `maxLifetimeSeconds` here would be the silent no-op the capability set exists
354            // to prevent, so this is a decision rather than a gap.
355            session_lifetime: false,
356            // No Alien process inside an Azure sandbox, so there is no supervisor to isolate.
357            supervisor_pid_namespace: false,
358            // No Alien process runs the command at all — the platform's own data plane does,
359            // so there is no separate supervisor identity to speak of.
360            supervisor_isolation: false,
361        }
362    }
363
364    /// What the GCP Agent Platform sandbox backend supports; the body of the `Platform::Gcp` arm.
365    pub fn gcp_agent_platform() -> Self {
366        Self {
367            // Agent file operations move over the session envelope.
368            files: true,
369            // Reaching a session across processes is safe because `generation` is derived from the
370            // container boot id read through the agent's health op, so a caller detects a container
371            // replaced under a stable session name rather than reconnecting to a blank one.
372            reconnect: true,
373            jobs: true,
374            // No method mints a port-scoped ingress capability; the only ingress is `:execute`.
375            preview: false,
376            // `:pause` and `:resume` preserve the running container.
377            suspend_resume: true,
378            // The create path never sends `sandbox_environment_snapshot`, so no session state is
379            // reachable through the trait; declared false until the client carries it.
380            snapshot: false,
381            // Egress is shaped by VPC and DNS peering, which is not a hostname allowlist.
382            domain_egress_rules: false,
383            // A declared `deny` blocks both routed egress and DNS.
384            egress_deny: true,
385            // The declared ceilings are enforced, but by terminating the session on breach rather
386            // than by refusing the allocation — a caller reading `true` should expect the session
387            // to die, not a clean error at the point of the request.
388            enforced_limits: true,
389            // No ceiling on process count is observed.
390            process_limit: false,
391            // `ttl` maps to a session `expireTime` the platform terminates at.
392            session_lifetime: true,
393            // No PID-namespace isolation between the command and anything supervising it.
394            supervisor_pid_namespace: false,
395            // No separate supervisor identity: the command is not run under a different identity
396            // than the process supervising it.
397            supervisor_isolation: false,
398        }
399    }
400
401    /// Returns a typed error if the named capability is absent on this platform.
402    pub fn require(&self, capability: SandboxCapability, platform: Platform) -> Result<()> {
403        let available = match capability {
404            SandboxCapability::Files => self.files,
405            SandboxCapability::Reconnect => self.reconnect,
406            SandboxCapability::Jobs => self.jobs,
407            SandboxCapability::Preview => self.preview,
408            SandboxCapability::SuspendResume => self.suspend_resume,
409            SandboxCapability::Snapshot => self.snapshot,
410            SandboxCapability::DomainEgressRules => self.domain_egress_rules,
411            SandboxCapability::EgressDeny => self.egress_deny,
412            SandboxCapability::EnforcedLimits => self.enforced_limits,
413            SandboxCapability::ProcessLimit => self.process_limit,
414            SandboxCapability::SessionLifetime => self.session_lifetime,
415            SandboxCapability::SupervisorPidNamespace => self.supervisor_pid_namespace,
416            SandboxCapability::SupervisorIsolation => self.supervisor_isolation,
417        };
418
419        if available {
420            return Ok(());
421        }
422
423        Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
424            capability: capability.as_str().to_string(),
425            platform: platform.to_string(),
426        }))
427    }
428}
429
430/// Names a single sandbox capability, so an unsupported call can report which one it needed.
431#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
432#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
433#[serde(rename_all = "camelCase")]
434pub enum SandboxCapability {
435    /// Moving files in and out of a session
436    Files,
437    /// Reaching a session created by an earlier call
438    Reconnect,
439    /// Starting, polling and cancelling a command across separate calls
440    Jobs,
441    /// An authenticated, port-scoped ingress capability
442    Preview,
443    /// Suspending and resuming session state
444    SuspendResume,
445    /// Capturing full session state
446    Snapshot,
447    /// Restricting egress to a hostname allowlist
448    DomainEgressRules,
449    /// Refusing outbound access when a sandbox declares none
450    EgressDeny,
451    /// Platform-enforced resource ceilings
452    EnforcedLimits,
453    /// A ceiling on the number of processes a session may run
454    ProcessLimit,
455    /// A wall-clock ceiling on a session, applied by the platform rather than by a caller
456    SessionLifetime,
457    /// A command runs in its own PID namespace, isolated from the agent supervising it
458    SupervisorPidNamespace,
459    /// A command runs under a different identity than the process supervising it
460    SupervisorIsolation,
461}
462
463impl SandboxCapability {
464    /// Returns the stable identifier used in errors and capability queries.
465    pub fn as_str(&self) -> &'static str {
466        match self {
467            Self::Files => "files",
468            Self::Reconnect => "reconnect",
469            Self::Jobs => "jobs",
470            Self::Preview => "preview",
471            Self::SuspendResume => "suspendResume",
472            Self::Snapshot => "snapshot",
473            Self::DomainEgressRules => "domainEgressRules",
474            Self::EgressDeny => "egressDeny",
475            Self::EnforcedLimits => "enforcedLimits",
476            Self::ProcessLimit => "processLimit",
477            Self::SessionLifetime => "sessionLifetime",
478            Self::SupervisorPidNamespace => "supervisorPidNamespace",
479            Self::SupervisorIsolation => "supervisorIsolation",
480        }
481    }
482}
483
484/// An isolated environment for running untrusted code, created per session at runtime.
485#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Builder)]
486#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
487#[serde(rename_all = "camelCase", deny_unknown_fields)]
488#[builder(start_fn = new)]
489pub struct Sandbox {
490    /// Identifier for the sandbox. Must contain only alphanumeric characters, hyphens, and
491    /// underscores ([A-Za-z0-9-_]). Maximum 64 characters.
492    #[builder(start_fn)]
493    pub id: String,
494    /// Where the sandbox's root filesystem comes from
495    pub code: SandboxCode,
496    /// Enforced resource ceilings.
497    ///
498    /// Optional because not every platform can enforce them, and a declaration that names none
499    /// takes the platform's own defaults. Naming them on a platform that cannot enforce them is
500    /// rejected at plan time rather than silently ignored.
501    #[serde(skip_serializing_if = "Option::is_none")]
502    pub limits: Option<SandboxLimits>,
503    /// Outbound network policy
504    pub egress: SandboxEgress,
505    /// Session lifetime and idle behaviour
506    pub session: SandboxSessionPolicy,
507    /// Ports eligible for a preview capability. An application reaches its sandbox through the
508    /// provider, so it cannot widen its own ingress at runtime; a holder of a remote binding's
509    /// credentials is bounded by no port condition, which is why a remote sandbox declares none.
510    #[builder(default)]
511    #[serde(default, skip_serializing_if = "Vec::is_empty")]
512    pub preview_ports: Vec<u16>,
513}
514
515/// Whether the artifact being rendered restricts which network modes it accepts.
516///
517/// Cloud setup needs explicit subnets for private databases and restricted sandbox connectors.
518/// Kubernetes targets do not emit these cloud backends.
519pub fn restricts_network_mode(stack: &crate::Stack, targets_kubernetes: bool) -> bool {
520    !targets_kubernetes && stack_needs_named_subnets_at_setup(stack)
521}
522
523/// Whether any setup-owned resource forces setup to name subnets.
524///
525/// Private databases and restricted sandbox connectors require subnet IDs. Neither generator can
526/// enumerate the account default VPC's subnets. Callers rendering an artifact want
527/// [`restricts_network_mode`] instead:
528/// this one answers for the declaration, which on a Kubernetes target is not what gets emitted.
529pub fn stack_needs_named_subnets_at_setup(stack: &crate::Stack) -> bool {
530    stack.resources().any(|(_resource_id, resource)| {
531        if resource.lifecycle == crate::ResourceLifecycle::Frozen
532            && resource.config.downcast_ref::<crate::Postgres>().is_some()
533        {
534            return true;
535        }
536        resource
537            .config
538            .downcast_ref::<Sandbox>()
539            .is_some_and(|sandbox| !matches!(sandbox.egress, SandboxEgress::Allow))
540    })
541}
542
543impl Sandbox {
544    /// The resource type identifier for Sandbox
545    pub const RESOURCE_TYPE: ResourceType = ResourceType::from_static("sandbox");
546
547    /// Returns the sandbox's unique identifier.
548    pub fn id(&self) -> &str {
549        &self.id
550    }
551
552    /// The declared ceilings, or the defaults a platform applies when none were named.
553    ///
554    /// Backends want a concrete set: a sandbox with no declared ceilings still runs inside
555    /// whatever the platform gives it, and a backend that had to branch on `None` would end up
556    /// inventing its own default anyway.
557    pub fn resolved_limits(&self) -> SandboxLimits {
558        self.limits.clone().unwrap_or_else(default_limits)
559    }
560
561    /// Validates the declaration against what the target platform can enforce.
562    ///
563    /// Runs at plan time so an unenforceable limit or an unsupported egress mode fails before
564    /// anything is provisioned, rather than at the first exec.
565    pub fn validate_for_platform(&self, platform: Platform) -> Result<()> {
566        let capabilities = SandboxCapabilities::for_platform(platform)?;
567
568        // No backend builds a sandbox image from source: an empty image string schedules a pod
569        // that can never run, the silent no-op the capability contract forbids — the failure
570        // has to land here instead.
571        if let SandboxCode::Source { .. } = &self.code {
572            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
573                resource_id: self.id.clone(),
574                field: "code".to_string(),
575                value: "source".to_string(),
576                reason: "no sandbox backend builds an image from source yet; give code.image a \
577                         prebuilt reference"
578                    .to_string(),
579            }));
580        }
581
582        // Read before the limits, because the image is declared whether or not any are.
583        if platform == Platform::Azure {
584            self.azure_catalog_image()?;
585        }
586
587        let Some(limits) = self.limits.as_ref() else {
588            // Nothing declared, so nothing to enforce and nothing to reject.
589            return self.validate_capabilities(&capabilities, platform);
590        };
591
592        validate_quantity(&self.id, "cpu", &limits.cpu)?;
593        validate_quantity(&self.id, "memory", &limits.memory)?;
594        validate_quantity(&self.id, "disk", &limits.disk)?;
595
596        if let Some(max_processes) = limits.max_processes {
597            if max_processes == 0 {
598                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
599                    resource_id: self.id.clone(),
600                    field: "maxProcesses".to_string(),
601                    value: "0".to_string(),
602                    reason: "a sandbox that may run no processes cannot run code".to_string(),
603                }));
604            }
605            capabilities.require(SandboxCapability::ProcessLimit, platform)?;
606        }
607
608        // Declaring limits a platform ignores is worse than not declaring them: the stack reads
609        // as bounded while the sandbox is not.
610        capabilities.require(SandboxCapability::EnforcedLimits, platform)?;
611
612        if platform == Platform::Azure {
613            self.azure_session_limits()?;
614        }
615
616        if platform == Platform::Aws {
617            // Refused here rather than at emit so a customer sees it while planning, and so both
618            // package formats inherit the same answer.
619            self.microvm_tier()?;
620
621            // The ceiling is Lambda's, and it rejects the run rather than clamping — so a value
622            // outside it would pass planning, render into the package, and fail at the first
623            // session. Kubernetes takes the same field with no such bound, which is why this
624            // sits under the AWS gate rather than on the type.
625            if let Some(seconds) = self.session.max_lifetime_seconds {
626                if !(1..=AWS_MAX_SESSION_LIFETIME_SECONDS).contains(&seconds) {
627                    return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
628                        resource_id: self.id.clone(),
629                        field: "maxLifetimeSeconds".to_string(),
630                        value: seconds.to_string(),
631                        reason: format!(
632                            "AWS runs a MicroVM for between 1 and \
633                             {AWS_MAX_SESSION_LIFETIME_SECONDS} seconds"
634                        ),
635                    }));
636                }
637            }
638        }
639
640        self.validate_capabilities(&capabilities, platform)
641    }
642
643    /// The catalog disk image Azure creates a session from.
644    ///
645    /// Azure names a public catalog entry rather than pulling a reference, so a registry path,
646    /// tag or digest has nowhere to go. An allowlist, because the answer to "what else could be
647    /// in there" is a name the data plane rejects at the first session, long after the apply.
648    pub fn azure_catalog_image(&self) -> Result<&str> {
649        let refused = |value: &str, reason: &str| {
650            AlienError::new(ErrorData::SandboxLimitInvalid {
651                resource_id: self.id.clone(),
652                field: "code.image".to_string(),
653                value: value.to_string(),
654                reason: reason.to_string(),
655            })
656        };
657
658        let SandboxCode::Image { image } = &self.code else {
659            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
660                resource_id: self.id.clone(),
661                field: "code".to_string(),
662                value: "source".to_string(),
663                reason: "no sandbox backend builds an image from source yet".to_string(),
664            }));
665        };
666
667        let image = image.trim();
668        if image.is_empty() {
669            return Err(refused(image, "a sandbox has to name an image"));
670        }
671        if !image
672            .chars()
673            .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '-'))
674        {
675            return Err(refused(
676                image,
677                "Azure creates a session from a public catalog disk image, so code.image must be \
678                 a bare catalog name such as 'ubuntu'",
679            ));
680        }
681        Ok(image)
682    }
683
684    /// Checks the declared ceilings against Azure's sizing rule (the `AZURE_*` constants above).
685    /// Refused at plan time, like [`Self::microvm_tier`], so a bad value is a declaration to fix
686    /// rather than a runtime fault at create.
687    pub fn azure_session_limits(&self) -> Result<()> {
688        let Some(limits) = self.limits.as_ref() else {
689            // Nothing declared means the binding substitutes Alien's own default sizing, which is
690            // inside the rule — asserted where those constants live, since this cannot see them.
691            return Ok(());
692        };
693
694        let refused = |field: &str, value: &str, reason: &str| {
695            AlienError::new(ErrorData::SandboxLimitInvalid {
696                resource_id: self.id.clone(),
697                field: field.to_string(),
698                value: value.to_string(),
699                reason: reason.to_string(),
700            })
701        };
702
703        let cpu_millicores = millicores(&limits.cpu)
704            .ok_or_else(|| refused("cpu", &limits.cpu, "expected cores or millicores"))?;
705
706        // The multiple is checked, not just the range: `333m` sits inside 0.25–16 cores and is
707        // still refused on the wire, so a bounds-only check would pass a declaration that fails
708        // at create.
709        if cpu_millicores % AZURE_CPU_STEP_MILLICORES != 0
710            || !(AZURE_CPU_STEP_MILLICORES..=AZURE_MAX_CPU_MILLICORES).contains(&cpu_millicores)
711        {
712            return Err(refused(
713                "cpu",
714                &limits.cpu,
715                "Azure allocates cpu in steps of 250m from 250m to 16000m",
716            ));
717        }
718
719        // Both ceilings are derived from the cpu, so they cannot be checked before it is known.
720        let memory_ceiling_mib = cpu_millicores * AZURE_MEMORY_MIB_PER_CORE / 1000;
721        let disk_ceiling_mib = cpu_millicores * AZURE_DISK_MIB_PER_CORE / 1000;
722
723        let memory_mib = quantity_mib(&limits.memory)
724            .ok_or_else(|| refused("memory", &limits.memory, "Azure sizes memory in whole MiB"))?;
725        if memory_mib > memory_ceiling_mib {
726            return Err(refused(
727                "memory",
728                &limits.memory,
729                &format!(
730                    "Azure allows at most 2Gi of memory per core, or {memory_ceiling_mib}Mi \
731                          at the declared cpu"
732                ),
733            ));
734        }
735
736        let disk_mib = quantity_mib(&limits.disk)
737            .ok_or_else(|| refused("disk", &limits.disk, "Azure sizes disk in whole MiB"))?;
738        if disk_mib > disk_ceiling_mib {
739            return Err(refused(
740                "disk",
741                &limits.disk,
742                &format!(
743                    "Azure allows at most 20Gi of disk per core, or {disk_ceiling_mib}Mi at \
744                          the declared cpu"
745                ),
746            ));
747        }
748
749        Ok(())
750    }
751
752    /// The MicroVM size that keeps every declared ceiling, or why none does.
753    ///
754    /// AWS sizes are discrete and a running MicroVM bursts to four times its baseline, so the
755    /// only tier that honours a ceiling is one whose peak fits inside it. A declaration no tier
756    /// satisfies is refused: shipping the nearest size would give the customer a sandbox that
757    /// exceeds the bound they wrote down.
758    pub fn microvm_tier(&self) -> Result<MicrovmTier> {
759        let Some(limits) = self.limits.as_ref() else {
760            // Nothing declared: AWS's own default baseline, which is also `default_limits`.
761            return Ok(MICROVM_TIERS[2]);
762        };
763
764        let memory_mib = quantity_mib(&limits.memory).ok_or_else(|| {
765            AlienError::new(ErrorData::SandboxLimitInvalid {
766                resource_id: self.id.clone(),
767                field: "memory".to_string(),
768                value: limits.memory.clone(),
769                reason: "AWS sizes a MicroVM in whole MiB".to_string(),
770            })
771        })?;
772        let disk_mib = quantity_mib(&limits.disk).ok_or_else(|| {
773            AlienError::new(ErrorData::SandboxLimitInvalid {
774                resource_id: self.id.clone(),
775                field: "disk".to_string(),
776                value: limits.disk.clone(),
777                reason: "AWS sizes a MicroVM's disk in whole MiB".to_string(),
778            })
779        })?;
780        let cpu_millicores = millicores(&limits.cpu).ok_or_else(|| {
781            AlienError::new(ErrorData::SandboxLimitInvalid {
782                resource_id: self.id.clone(),
783                field: "cpu".to_string(),
784                value: limits.cpu.clone(),
785                reason: "expected cores or millicores".to_string(),
786            })
787        })?;
788
789        // Memory and disk choose the size; cpu is then checked rather than used to choose.
790        // AWS couples cpu to memory at 2 GB per vCPU, so letting a low cpu ceiling select the
791        // size too would quietly hand back a machine four times smaller than the memory ceiling
792        // asked for, with nothing to indicate it.
793        let sized = |tier: &&MicrovmTier| {
794            tier.peak_memory_mib <= memory_mib && tier.max_disk_mib <= disk_mib
795        };
796
797        let tier = MICROVM_TIERS
798            .iter()
799            .rev()
800            .find(sized)
801            .copied()
802            .ok_or_else(|| {
803                AlienError::new(ErrorData::SandboxLimitInvalid {
804                    resource_id: self.id.clone(),
805                    field: "memory".to_string(),
806                    value: limits.memory.clone(),
807                    reason: format!(
808                        "a Lambda MicroVM bursts to four times its baseline, so the smallest \
809                         ceiling AWS can hold is 2Gi memory with 8Gi disk; '{}' memory and '{}' \
810                         disk fit no size",
811                        limits.memory, limits.disk
812                    ),
813                })
814            })?;
815
816        let required_millicores = i64::from(tier.peak_vcpu) * 1000;
817        if cpu_millicores < required_millicores {
818            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
819                resource_id: self.id.clone(),
820                field: "cpu".to_string(),
821                value: limits.cpu.clone(),
822                reason: format!(
823                    "AWS allocates one vCPU per 2GB, so a MicroVM sized to a '{}' memory ceiling \
824                     reaches {} vCPU; declare cpu '{}' or lower the memory ceiling",
825                    limits.memory, tier.peak_vcpu, tier.peak_vcpu
826                ),
827            }));
828        }
829
830        Ok(tier)
831    }
832
833    /// The capability checks that do not depend on declared limits.
834    fn validate_capabilities(
835        &self,
836        capabilities: &SandboxCapabilities,
837        platform: Platform,
838    ) -> Result<()> {
839        if matches!(self.egress, SandboxEgress::AllowDomains { .. }) {
840            capabilities.require(SandboxCapability::DomainEgressRules, platform)?;
841        }
842
843        // `allow` asks for no restriction, so a backend that ignores it fails loudly on the first
844        // blocked connection. `deny` asks for one, and a backend that ignores it puts untrusted
845        // code on the internet with nothing to notice — so only this direction is gated.
846        // An empty list is not a restriction anyone wrote down: it renders as a deny-all wearing
847        // an allowlist's label, which reads at a glance as the opposite of what it does.
848        if let SandboxEgress::AllowDomains { domains } = &self.egress {
849            if domains.is_empty() {
850                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
851                    resource_id: self.id.clone(),
852                    field: "egress.domains".to_string(),
853                    value: "[]".to_string(),
854                    reason: "an allowlist naming no domain denies everything; declare \
855                             egress: deny if that is what was meant"
856                        .to_string(),
857                }));
858            }
859        }
860
861        if matches!(self.egress, SandboxEgress::Deny) {
862            capabilities.require(SandboxCapability::EgressDeny, platform)?;
863        }
864
865        if !self.preview_ports.is_empty() {
866            capabilities.require(SandboxCapability::Preview, platform)?;
867        }
868
869        if self.session.idle_suspend_seconds.is_some() {
870            capabilities.require(SandboxCapability::SuspendResume, platform)?;
871        }
872
873        if self.session.max_lifetime_seconds.is_some() {
874            capabilities.require(SandboxCapability::SessionLifetime, platform)?;
875        }
876
877        Ok(())
878    }
879}
880
881/// Ceilings applied when a declaration names none.
882///
883/// Modest on purpose: an undeclared sandbox is one whose author did not think about sizing, and
884/// the safe reading of that is a small box rather than a generous one.
885fn default_limits() -> SandboxLimits {
886    SandboxLimits {
887        cpu: "1".to_string(),
888        memory: "2Gi".to_string(),
889        disk: "8Gi".to_string(),
890        max_processes: None,
891    }
892}
893
894/// Validates a Kubernetes-style resource quantity such as `500m`, `2Gi` or `1`.
895fn validate_quantity(resource_id: &str, field: &str, value: &str) -> Result<()> {
896    let invalid = |reason: &str| {
897        AlienError::new(ErrorData::SandboxLimitInvalid {
898            resource_id: resource_id.to_string(),
899            field: field.to_string(),
900            value: value.to_string(),
901            reason: reason.to_string(),
902        })
903    };
904
905    let digits_end = value
906        .find(|c: char| !c.is_ascii_digit() && c != '.')
907        .unwrap_or(value.len());
908    let (number, suffix) = value.split_at(digits_end);
909
910    let parsed: f64 = number
911        .parse()
912        .map_err(|_| invalid("expected a number, optionally followed by a unit suffix"))?;
913
914    if parsed <= 0.0 {
915        return Err(invalid("must be greater than zero"));
916    }
917
918    const SUFFIXES: &[&str] = &["", "m", "k", "M", "G", "T", "Ki", "Mi", "Gi", "Ti"];
919    if !SUFFIXES.contains(&suffix) {
920        return Err(invalid(
921            "unit must be one of m, k, M, G, T, Ki, Mi, Gi, Ti, or absent",
922        ));
923    }
924
925    Ok(())
926}
927
928/// Splits a quantity into its number and unit suffix.
929fn split_quantity(value: &str) -> Option<(f64, &str)> {
930    let trimmed = value.trim();
931    let digits_end = trimmed
932        .find(|c: char| !c.is_ascii_digit() && c != '.')
933        .unwrap_or(trimmed.len());
934    let (number, suffix) = trimmed.split_at(digits_end);
935    number.parse().ok().map(|number| (number, suffix))
936}
937
938/// A memory or disk quantity in whole MiB, rounded down.
939///
940/// Every suffix `validate_quantity` accepts is handled here. Reading only `Gi` and `Mi` and
941/// falling back for the rest would turn a declared `4G` into a different size than the customer
942/// asked for, which for a ceiling means a sandbox larger than its bound.
943pub fn quantity_mib(value: &str) -> Option<i64> {
944    let (number, suffix) = split_quantity(value)?;
945    let bytes = match suffix {
946        "" => number,
947        "k" => number * 1e3,
948        "M" => number * 1e6,
949        "G" => number * 1e9,
950        "T" => number * 1e12,
951        "Ki" => number * 1024.0,
952        "Mi" => number * 1024.0 * 1024.0,
953        "Gi" => number * 1024.0 * 1024.0 * 1024.0,
954        "Ti" => number * 1024.0 * 1024.0 * 1024.0 * 1024.0,
955        // `m` is a millicore suffix; memory has no use for it.
956        _ => return None,
957    };
958    Some((bytes / (1024.0 * 1024.0)) as i64)
959}
960
961/// A CPU quantity in millicores.
962pub fn millicores(value: &str) -> Option<i64> {
963    let (number, suffix) = split_quantity(value)?;
964    match suffix {
965        "" => Some((number * 1000.0) as i64),
966        "m" => Some(number as i64),
967        _ => None,
968    }
969}
970
971/// Outputs generated by a successfully provisioned Sandbox parent.
972#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
973#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
974#[serde(rename_all = "camelCase")]
975pub struct SandboxOutputs {
976    /// Name of the durable parent that sessions are created inside
977    pub parent_name: String,
978    /// Platform-specific identifier for the parent (image ARN, sandbox group id, namespace)
979    #[serde(skip_serializing_if = "Option::is_none")]
980    pub identifier: Option<String>,
981    /// Data-plane endpoint sessions are created through, where the platform has one
982    #[serde(skip_serializing_if = "Option::is_none")]
983    pub endpoint: Option<String>,
984}
985
986impl ResourceOutputsDefinition for SandboxOutputs {
987    fn get_resource_type(&self) -> ResourceType {
988        Sandbox::RESOURCE_TYPE
989    }
990
991    fn as_any(&self) -> &dyn Any {
992        self
993    }
994
995    fn box_clone(&self) -> Box<dyn ResourceOutputsDefinition> {
996        Box::new(self.clone())
997    }
998
999    fn outputs_eq(&self, other: &dyn ResourceOutputsDefinition) -> bool {
1000        other.as_any().downcast_ref::<SandboxOutputs>() == Some(self)
1001    }
1002
1003    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1004        serde_json::to_value(self)
1005    }
1006}
1007
1008impl ResourceDefinition for Sandbox {
1009    fn get_resource_type(&self) -> ResourceType {
1010        Self::RESOURCE_TYPE
1011    }
1012
1013    fn id(&self) -> &str {
1014        &self.id
1015    }
1016
1017    fn get_dependencies(&self) -> Vec<ResourceRef> {
1018        Vec::new()
1019    }
1020
1021    fn validate_update(&self, new_config: &dyn ResourceDefinition) -> Result<()> {
1022        let new_sandbox = new_config
1023            .as_any()
1024            .downcast_ref::<Sandbox>()
1025            .ok_or_else(|| {
1026                AlienError::new(ErrorData::UnexpectedResourceType {
1027                    resource_id: self.id.clone(),
1028                    expected: Self::RESOURCE_TYPE,
1029                    actual: new_config.get_resource_type(),
1030                })
1031            })?;
1032
1033        if self.id != new_sandbox.id {
1034            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
1035                resource_id: self.id.clone(),
1036                reason: "the 'id' field is immutable".to_string(),
1037            }));
1038        }
1039
1040        Ok(())
1041    }
1042
1043    fn as_any(&self) -> &dyn Any {
1044        self
1045    }
1046
1047    fn as_any_mut(&mut self) -> &mut dyn Any {
1048        self
1049    }
1050
1051    fn box_clone(&self) -> Box<dyn ResourceDefinition> {
1052        Box::new(self.clone())
1053    }
1054
1055    fn resource_eq(&self, other: &dyn ResourceDefinition) -> bool {
1056        other.as_any().downcast_ref::<Sandbox>() == Some(self)
1057    }
1058
1059    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1060        serde_json::to_value(self)
1061    }
1062}
1063
1064/// The one token a sandbox bundle URI may carry, replaced with the deploying region.
1065///
1066/// AWS builds a MicroVM image only from a bucket in the image's own region, so a vendor
1067/// publishing to every supported region needs one stored URI that resolves per region.
1068pub const BUNDLE_REGION_TOKEN: &str = "{region}";
1069
1070/// A bundle URI split around its region token, or carried whole when it has none.
1071#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1072pub enum BundleUri<'a> {
1073    /// No token: emitted exactly as it is today.
1074    Literal(&'a str),
1075    /// The text either side of the token, for an emitter to rejoin around its own region
1076    /// expression.
1077    Regional { before: &'a str, after: &'a str },
1078}
1079
1080/// The prefix a rebuild's new key still sits under: everything above the file name and the
1081/// version segment beneath it. `None` when nothing sits there, meaning no prefix can be granted
1082/// without also granting objects a rebuild never reads. Shared so both emitters agree on it.
1083pub fn stable_bundle_key_prefix(key: &str) -> Option<&str> {
1084    let (above_file, _) = key.rsplit_once('/')?;
1085    let (above_version, _) = above_file.rsplit_once('/')?;
1086    Some(above_version)
1087}
1088
1089/// Reads a sandbox bundle URI, refusing anything an image build would only reject later.
1090///
1091/// The token is accepted in the bucket alone. A key-position token would name an object that does
1092/// not exist, and any other brace is a typo that would otherwise reach S3 verbatim and fail ~160s
1093/// into the build — which is the failure this whole check exists to move to plan time.
1094pub fn parse_bundle_uri(uri: &str) -> std::result::Result<BundleUri<'_>, String> {
1095    let path = uri
1096        .strip_prefix("s3://")
1097        .ok_or_else(|| format!("'{uri}' is not an s3:// URI"))?;
1098    let (bucket, key) = path
1099        .split_once('/')
1100        .ok_or_else(|| format!("'{uri}' names a bucket with no object key"))?;
1101
1102    // Both emitters interpolate this path into the build role's resource ARN, where `*` and `?`
1103    // are IAM wildcards rather than literal characters. S3 accepts them in a key, so a bundle
1104    // published under one would silently widen the grant past the bundle it names.
1105    if path.contains('*') || path.contains('?') {
1106        return Err(format!(
1107            "'{uri}' carries an IAM wildcard; the bundle's path is interpolated into the build \
1108             role's grant, so '*' and '?' would widen it past the bundle"
1109        ));
1110    }
1111
1112    if key.contains('{') || key.contains('}') {
1113        return Err(format!(
1114            "'{uri}' places a token in the object key; {BUNDLE_REGION_TOKEN} is accepted in the \
1115             bucket name alone"
1116        ));
1117    }
1118
1119    let Some((before, after)) = bucket.split_once(BUNDLE_REGION_TOKEN) else {
1120        if bucket.contains('{') || bucket.contains('}') {
1121            return Err(format!(
1122                "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the \
1123                 only one"
1124            ));
1125        }
1126        return Ok(BundleUri::Literal(uri));
1127    };
1128
1129    if after.contains(BUNDLE_REGION_TOKEN) {
1130        return Err(format!("'{uri}' repeats {BUNDLE_REGION_TOKEN}"));
1131    }
1132    if before.contains('{') || before.contains('}') || after.contains('{') || after.contains('}') {
1133        return Err(format!(
1134            "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the only one"
1135        ));
1136    }
1137
1138    Ok(BundleUri::Regional {
1139        before: &uri[.."s3://".len() + before.len()],
1140        after: &uri["s3://".len() + before.len() + BUNDLE_REGION_TOKEN.len()..],
1141    })
1142}
1143
1144#[cfg(test)]
1145mod tests {
1146    use super::*;
1147
1148    #[test]
1149    fn private_database_setup_requires_named_subnets() {
1150        for lifecycle in [
1151            crate::ResourceLifecycle::Frozen,
1152            crate::ResourceLifecycle::Live,
1153        ] {
1154            let stack = crate::Stack::new("database".to_string())
1155                .add(
1156                    crate::Postgres::new("metadata".to_string()).build(),
1157                    lifecycle,
1158                )
1159                .build();
1160            assert_eq!(
1161                restricts_network_mode(&stack, false),
1162                lifecycle == crate::ResourceLifecycle::Frozen,
1163            );
1164            assert!(!restricts_network_mode(&stack, true));
1165        }
1166        assert!(!restricts_network_mode(
1167            &crate::Stack::new("empty".to_string()).build(),
1168            false,
1169        ));
1170    }
1171
1172    /// A wildcard reaching the grant would widen it past the bundle, and it widens the Frozen
1173    /// object grant as readily as the Live prefix — both interpolate the path into the ARN.
1174    #[test]
1175    fn a_uri_carrying_an_iam_wildcard_is_refused() {
1176        for uri in [
1177            "s3://acme/team-*/v1/bundle.zip",
1178            "s3://acme/sandbox-bundle/f00d/bundle?.zip",
1179            "s3://acme-*/sandbox-bundle/f00d/bundle.zip",
1180        ] {
1181            let error = parse_bundle_uri(uri).expect_err("a wildcard must be refused");
1182            assert!(error.contains("IAM wildcard"), "for {uri}: {error}");
1183        }
1184
1185        parse_bundle_uri("s3://acme/sandbox-bundle/f00d/bundle.zip")
1186            .expect("an ordinary key still parses");
1187    }
1188
1189    /// The rule both package formats grant by, pinned here rather than in either. The
1190    /// near-misses the cases separate: the key's first segment grants objects a rebuild never
1191    /// reads, and the object's own directory pins the version segment that moves.
1192    #[test]
1193    fn a_grantable_prefix_stops_above_the_segment_that_moves() {
1194        assert_eq!(
1195            stable_bundle_key_prefix("sandbox-bundle/f00dcafe/bundle.zip"),
1196            Some("sandbox-bundle")
1197        );
1198        assert_eq!(
1199            stable_bundle_key_prefix("artifacts/team-a/sandbox/f00dcafe/bundle.zip"),
1200            Some("artifacts/team-a/sandbox"),
1201            "a deeper key narrows the prefix, it never widens to the first segment"
1202        );
1203
1204        // Nothing sits above the version segment, so no prefix a moved bundle stays inside
1205        // exists. Emitting the object grant instead installs a role that denies the next rebuild.
1206        assert_eq!(stable_bundle_key_prefix("agents/bundle.zip"), None);
1207        assert_eq!(stable_bundle_key_prefix("bundle.zip"), None);
1208    }
1209
1210    fn sandbox_with(egress: SandboxEgress, preview_ports: Vec<u16>) -> Sandbox {
1211        Sandbox::new("agent-sbx".to_string())
1212            .code(SandboxCode::Image {
1213                image: "ubuntu".to_string(),
1214            })
1215            .limits(SandboxLimits {
1216                cpu: "1".to_string(),
1217                memory: "2Gi".to_string(),
1218                disk: "20Gi".to_string(),
1219                max_processes: None,
1220            })
1221            .egress(egress)
1222            .session(SandboxSessionPolicy {
1223                max_lifetime_seconds: None,
1224                idle_suspend_seconds: None,
1225            })
1226            .preview_ports(preview_ports)
1227            .build()
1228    }
1229
1230    /// A URI with no token must come back whole, because every bundle configured today has none
1231    /// and emitting one differently would change every existing customer's template.
1232    #[test]
1233    fn a_uri_without_a_token_is_carried_whole() {
1234        assert_eq!(
1235            parse_bundle_uri("s3://acme-artifacts-us-east-2/agents/bundle.zip"),
1236            Ok(BundleUri::Literal(
1237                "s3://acme-artifacts-us-east-2/agents/bundle.zip"
1238            ))
1239        );
1240    }
1241
1242    /// The split has to rejoin to the original with the region in place, or an emitter builds a
1243    /// URI that is subtly not the one the vendor configured.
1244    #[test]
1245    fn a_regional_uri_splits_either_side_of_the_token() {
1246        let BundleUri::Regional { before, after } =
1247            parse_bundle_uri("s3://acme-artifacts-{region}/agents/bundle.zip")
1248                .expect("the token is accepted in the bucket")
1249        else {
1250            panic!("a bucket-position token must split");
1251        };
1252
1253        assert_eq!(before, "s3://acme-artifacts-");
1254        assert_eq!(after, "/agents/bundle.zip");
1255        assert_eq!(
1256            format!("{before}us-east-2{after}"),
1257            "s3://acme-artifacts-us-east-2/agents/bundle.zip",
1258            "the halves must rejoin to the URI the vendor meant"
1259        );
1260    }
1261
1262    /// Each of these reaches S3 verbatim and dies ~160s into an image build if it is not refused
1263    /// here, which is the whole reason this runs at plan time.
1264    #[test]
1265    fn a_token_this_build_cannot_resolve_is_refused() {
1266        for uri in [
1267            "s3://acme-artifacts-{regio}/bundle.zip",
1268            "s3://acme-artifacts/{region}/bundle.zip",
1269            "s3://acme-artifacts-{region}-{region}/bundle.zip",
1270            "s3://acme-artifacts/bundle-{version}.zip",
1271            "s3://acme}-artifacts-{region}/bundle.zip",
1272            "s3://acme{-artifacts-{region}/bundle.zip",
1273        ] {
1274            assert!(
1275                parse_bundle_uri(uri).is_err(),
1276                "'{uri}' must be refused before it can reach an image build"
1277            );
1278        }
1279    }
1280
1281    #[test]
1282    fn resource_type_is_stable() {
1283        assert_eq!(Sandbox::RESOURCE_TYPE.as_ref(), "sandbox");
1284    }
1285
1286    #[test]
1287    fn capability_sets_are_per_platform() {
1288        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1289        assert!(
1290            gcp.reconnect,
1291            "generation from the container boot id makes a session reachable across processes"
1292        );
1293        assert!(!gcp.preview);
1294        assert!(gcp.enforced_limits);
1295
1296        let azure = SandboxCapabilities::for_platform(Platform::Azure).expect("azure is supported");
1297        assert!(azure.files, "every backend moves files");
1298        assert!(gcp.files);
1299        // Azure is the only backend whose egress policy matches on host pattern, and the only
1300        // one where `deny` and a hostname list are the same object.
1301        assert!(azure.domain_egress_rules);
1302        assert!(azure.egress_deny);
1303        // The data plane honours a continuous cpu/memory/disk surface and refuses anything
1304        // outside `n×250m`, with the rule in the message.
1305        assert!(azure.enforced_limits);
1306        assert!(azure.suspend_resume);
1307        // Both stay false for reasons that are not "unbuilt": a snapshot id has nothing to
1308        // consume it on any backend, and an Azure port's auth is anonymous or a human allowlist,
1309        // neither of which is a port-scoped credential.
1310        assert!(!azure.snapshot);
1311        assert!(!azure.preview);
1312
1313        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1314        assert!(!aws.snapshot, "AWS has no user-callable session snapshot");
1315        assert!(aws.suspend_resume);
1316
1317        let k8s =
1318            SandboxCapabilities::for_platform(Platform::Kubernetes).expect("k8s is supported");
1319        assert!(
1320            !k8s.preview,
1321            "the session-scoped ingress gateway does not exist yet"
1322        );
1323    }
1324
1325    /// Whether the process supervising a command is a separate identity from the command.
1326    ///
1327    /// Values are measured, not inferred. AWS: the agent runs as uid 0 with
1328    /// `CapEff: 00000000a80425fb` and `setuid`s the command to uid 60000, so the two differ.
1329    /// Kubernetes: the sandbox pod pins `run_as_user: 65534` on both pod and container with
1330    /// `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no uid split is
1331    /// possible (`kubernetes_spec.rs`). Local: `docker exec` runs as the workload uid while the
1332    /// manager supervises from the host. Azure and Agent Platform have no in-sandbox supervisor.
1333    #[test]
1334    fn supervisor_isolation_is_per_platform() {
1335        let value = |platform| {
1336            SandboxCapabilities::for_platform(platform)
1337                .expect("supported")
1338                .supervisor_isolation
1339        };
1340
1341        assert!(
1342            value(Platform::Aws),
1343            "root agent setuids the command to 60000"
1344        );
1345        assert!(
1346            value(Platform::Local),
1347            "the supervisor is on the host, outside the container"
1348        );
1349        assert!(
1350            !value(Platform::Kubernetes),
1351            "a single pinned uid cannot be split"
1352        );
1353        assert!(!value(Platform::Azure), "no Alien process runs the command");
1354        assert!(
1355            !value(Platform::Gcp),
1356            "no separate supervisor identity runs the command"
1357        );
1358    }
1359
1360    /// The point of the field: AWS and GCP report the *same* `supervisor_pid_namespace` (neither
1361    /// has `CAP_SYS_ADMIN`), so that axis alone reads them as equivalent. They are not — AWS
1362    /// separates the command's identity from the supervisor's and Agent Platform does not.
1363    #[test]
1364    fn supervisor_isolation_separates_aws_from_a_subprocess_backend() {
1365        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1366        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1367
1368        assert_eq!(
1369            aws.supervisor_pid_namespace, gcp.supervisor_pid_namespace,
1370            "the older axis cannot tell them apart"
1371        );
1372        assert!(
1373            aws.supervisor_isolation,
1374            "AWS setuids the command off the supervisor"
1375        );
1376        assert!(
1377            !gcp.supervisor_isolation,
1378            "the command runs under no separate supervisor identity"
1379        );
1380    }
1381
1382    /// The Agent Platform row, each value against the behaviour it was measured from. `reconnect`
1383    /// is the tripwire: it is `true` only because `generation` is derived from the container boot
1384    /// id read through the agent's health op, so a caller detects a replaced container instead of
1385    /// reconnecting to a blank one. It is also the body of the `Platform::Gcp` arm, asserted below.
1386    #[test]
1387    fn gcp_agent_platform_row_matches_measured_backend() {
1388        let row = SandboxCapabilities::gcp_agent_platform();
1389
1390        assert!(row.files, "agent file ops move over the session envelope");
1391        assert!(
1392            row.reconnect,
1393            "generation is derived from the container boot id, so a session is reachable across \
1394             processes"
1395        );
1396        assert!(
1397            !row.preview,
1398            "the only ingress is :execute; no port-scoped capability"
1399        );
1400        assert!(
1401            row.suspend_resume,
1402            ":pause and :resume preserve the container"
1403        );
1404        assert!(
1405            !row.snapshot,
1406            "the create path never sends a snapshot, so none is reachable through the trait"
1407        );
1408        assert!(
1409            !row.domain_egress_rules,
1410            "VPC and DNS peering is not a hostname allowlist"
1411        );
1412        assert!(
1413            row.egress_deny,
1414            "a declared deny blocks both egress and DNS"
1415        );
1416        assert!(
1417            row.enforced_limits,
1418            "ceilings are enforced, by terminating the session on breach"
1419        );
1420        assert!(!row.process_limit, "no process-count ceiling is observed");
1421        assert!(row.session_lifetime, "ttl maps to a session expireTime");
1422        assert!(!row.supervisor_pid_namespace, "no PID-namespace isolation");
1423        assert!(
1424            !row.supervisor_isolation,
1425            "the command is not run under a separate supervisor identity"
1426        );
1427
1428        // Agent Platform is the registered GCP backend, so the arm returns exactly this row.
1429        let live = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1430        assert_eq!(
1431            live, row,
1432            "the Platform::Gcp arm is the Agent Platform capability row"
1433        );
1434    }
1435
1436    #[test]
1437    fn platforms_without_a_backend_are_an_error_not_an_empty_set() {
1438        let error = SandboxCapabilities::for_platform(Platform::Machines)
1439            .expect_err("Machines has no sandbox backend");
1440        assert_eq!(error.code, "SANDBOX_PLATFORM_UNSUPPORTED");
1441    }
1442
1443    #[test]
1444    fn unsupported_capability_names_platform_and_capability() {
1445        let capabilities = SandboxCapabilities::for_platform(Platform::Gcp).expect("supported");
1446        let error = capabilities
1447            .require(SandboxCapability::Preview, Platform::Gcp)
1448            .expect_err("GCP has no preview");
1449
1450        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1451        let rendered = error.to_string();
1452        assert!(
1453            rendered.contains("preview"),
1454            "names the capability: {rendered}"
1455        );
1456        assert!(rendered.contains("gcp"), "names the platform: {rendered}");
1457    }
1458
1459    /// Azure matches on hostname; AWS and Kubernetes match CIDRs, and Local and GCP have a
1460    /// switch rather than a filter. Accepting a hostname list on those four would leave a stack
1461    /// reading as restricted while the sandbox reaches the whole internet.
1462    #[test]
1463    fn a_hostname_allowlist_is_refused_everywhere_it_would_be_approximated() {
1464        let sandbox = sandbox_with(
1465            SandboxEgress::AllowDomains {
1466                domains: vec!["example.com".to_string()],
1467            },
1468            vec![],
1469        );
1470
1471        for platform in [
1472            Platform::Aws,
1473            Platform::Gcp,
1474            Platform::Kubernetes,
1475            Platform::Local,
1476        ] {
1477            let error = sandbox
1478                .validate_for_platform(platform)
1479                .expect_err("only Azure expresses a hostname allowlist");
1480            assert_eq!(
1481                error.code, "SANDBOX_CAPABILITY_UNSUPPORTED",
1482                "on {platform:?}"
1483            );
1484        }
1485
1486        assert!(
1487            SandboxCapabilities::for_platform(Platform::Azure)
1488                .expect("supported")
1489                .domain_egress_rules,
1490            "Azure's egress policy matches on host pattern"
1491        );
1492    }
1493
1494    /// `deny` is the declaration that carries a security promise, so a backend that cannot keep
1495    /// it has to refuse rather than accept it and run the code with open egress.
1496    #[test]
1497    fn a_denied_egress_is_refused_where_it_would_not_be_enforced() {
1498        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1499
1500        // GCP is asserted at the capability rather than through validation: this sandbox declares
1501        // ceilings GCP cannot enforce, so it is refused for a reason unrelated to egress.
1502        assert!(
1503            SandboxCapabilities::for_platform(Platform::Gcp)
1504                .expect("supported")
1505                .egress_deny
1506        );
1507
1508        for platform in [Platform::Aws, Platform::Kubernetes, Platform::Local] {
1509            sandbox
1510                .validate_for_platform(platform)
1511                .expect("deny is enforced here");
1512        }
1513
1514        // Declares no ceilings, which Azure refuses for its own reason, so this isolates egress.
1515        let egress_only = Sandbox::new("sbx".to_string())
1516            .code(SandboxCode::Image {
1517                image: "alpine".to_string(),
1518            })
1519            .egress(SandboxEgress::Deny)
1520            .session(SandboxSessionPolicy {
1521                max_lifetime_seconds: None,
1522                idle_suspend_seconds: None,
1523            })
1524            .build();
1525
1526        egress_only
1527            .validate_for_platform(Platform::Azure)
1528            .expect("Azure creates the sandbox under a Deny policy with full inspection");
1529    }
1530
1531    /// A sandbox naming no ceilings is valid on every platform and still resolves to a concrete
1532    /// set. The rule Azure applies to ceilings that *are* declared is pinned separately, by
1533    /// `azure_sizes_follow_the_rule_the_data_plane_states`.
1534    #[test]
1535    fn a_sandbox_declaring_no_ceilings_takes_the_platforms_own() {
1536        let undeclared = Sandbox::new("sbx".to_string())
1537            .code(SandboxCode::Image {
1538                image: "alpine".to_string(),
1539            })
1540            .egress(SandboxEgress::Deny)
1541            .session(SandboxSessionPolicy {
1542                max_lifetime_seconds: None,
1543                idle_suspend_seconds: None,
1544            })
1545            .build();
1546
1547        undeclared
1548            .validate_for_platform(Platform::Azure)
1549            .expect("a sandbox naming no ceilings takes the platform's own");
1550
1551        // A backend still gets a concrete set, so nothing downstream has to invent one.
1552        assert_eq!(undeclared.resolved_limits().cpu, "1");
1553    }
1554
1555    /// Pins Azure's own sizing rule: `250m`, `1500m` and `4000m` are valid, `32000m` and `333m`
1556    /// are refused. Checked at plan time so a bad value is a declaration to fix, not a package
1557    /// that renders without error and dies at the first session.
1558    #[test]
1559    fn azure_sizes_follow_the_rule_the_data_plane_states() {
1560        let sized = |cpu: &str, memory: &str, disk: &str| {
1561            let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1562            let limits = sandbox
1563                .limits
1564                .as_mut()
1565                .expect("the fixture declares limits");
1566            limits.cpu = cpu.to_string();
1567            limits.memory = memory.to_string();
1568            limits.disk = disk.to_string();
1569            sandbox.validate_for_platform(Platform::Azure)
1570        };
1571
1572        sized("250m", "512Mi", "5120Mi").expect("the smallest step the data plane accepts");
1573        sized("4000m", "8192Mi", "40960Mi").expect("cpu, memory and disk are all honoured");
1574        sized("16000m", "32Gi", "320Gi").expect("the top of the range");
1575
1576        // Same off-step case `azure_session_limits` checks the multiple for, not just the range.
1577        let off_step = sized("333m", "512Mi", "5120Mi").expect_err("333m is not a step of 250m");
1578        assert_eq!(off_step.code, "SANDBOX_LIMIT_INVALID", "{off_step}");
1579        assert!(off_step.to_string().contains("cpu"), "{off_step}");
1580
1581        let too_big = sized("32000m", "64Gi", "640Gi").expect_err("32 cores is over the ceiling");
1582        assert_eq!(too_big.code, "SANDBOX_LIMIT_INVALID", "{too_big}");
1583
1584        // Both ceilings are derived from the cpu, so the same memory passes at one size and fails
1585        // at another - which is what makes them worth checking rather than bounding absolutely.
1586        sized("1000m", "2Gi", "20Gi").expect("2Gi is exactly one core's worth");
1587        let over_memory = sized("250m", "2Gi", "5120Mi").expect_err("2Gi needs a full core");
1588        assert_eq!(over_memory.code, "SANDBOX_LIMIT_INVALID", "{over_memory}");
1589        assert!(over_memory.to_string().contains("memory"), "{over_memory}");
1590
1591        let over_disk = sized("250m", "512Mi", "20Gi").expect_err("20Gi needs a full core");
1592        assert!(over_disk.to_string().contains("disk"), "{over_disk}");
1593    }
1594
1595    #[test]
1596    fn preview_ports_require_the_preview_capability() {
1597        let sandbox = sandbox_with(SandboxEgress::Deny, vec![8080]);
1598
1599        sandbox
1600            .validate_for_platform(Platform::Aws)
1601            .expect("AWS mints a port-scoped JWE");
1602
1603        let error = sandbox
1604            .validate_for_platform(Platform::Kubernetes)
1605            .expect_err("Kubernetes preview is deferred");
1606        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1607    }
1608
1609    #[test]
1610    fn gcp_accepts_a_sandbox_declaring_enforced_limits() {
1611        let sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1612        sandbox
1613            .validate_for_platform(Platform::Gcp)
1614            .expect("Agent Platform enforces declared ceilings, by terminating on breach");
1615    }
1616
1617    #[test]
1618    fn invalid_quantities_are_rejected_with_the_offending_field() {
1619        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1620        sandbox
1621            .limits
1622            .as_mut()
1623            .expect("the fixture declares limits")
1624            .memory = "2Gb".to_string();
1625
1626        let error = sandbox
1627            .validate_for_platform(Platform::Aws)
1628            .expect_err("Gb is not a valid suffix");
1629        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1630        assert!(error.to_string().contains("memory"));
1631
1632        sandbox
1633            .limits
1634            .as_mut()
1635            .expect("the fixture declares limits")
1636            .memory = "2Gi".to_string();
1637        sandbox
1638            .limits
1639            .as_mut()
1640            .expect("the fixture declares limits")
1641            .cpu = "0".to_string();
1642        let error = sandbox
1643            .validate_for_platform(Platform::Aws)
1644            .expect_err("zero cpu is not a ceiling");
1645        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1646    }
1647
1648    #[test]
1649    fn zero_max_processes_is_rejected() {
1650        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1651        sandbox
1652            .limits
1653            .as_mut()
1654            .expect("the fixture declares limits")
1655            .max_processes = Some(0);
1656
1657        let error = sandbox
1658            .validate_for_platform(Platform::Local)
1659            .expect_err("a sandbox must be able to run at least one process");
1660        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1661        assert!(error.to_string().contains("maxProcesses"));
1662    }
1663
1664    /// A process ceiling needs a container runtime. Kubernetes sets one per node rather than per
1665    /// pod, and neither MicroVMs nor Azure sandboxes expose one, so accepting the declaration
1666    /// anywhere else would mean carrying a bound nothing applies.
1667    #[test]
1668    fn a_process_ceiling_is_accepted_only_where_a_runtime_can_apply_it() {
1669        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1670        sandbox
1671            .limits
1672            .as_mut()
1673            .expect("the fixture declares limits")
1674            .max_processes = Some(256);
1675
1676        sandbox
1677            .validate_for_platform(Platform::Local)
1678            .expect("Docker takes a pids limit");
1679
1680        for platform in [Platform::Aws, Platform::Azure, Platform::Kubernetes] {
1681            let error = sandbox
1682                .validate_for_platform(platform)
1683                .expect_err("a process ceiling nothing applies must be refused");
1684            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1685        }
1686    }
1687
1688    /// Lambda rejects a run outside 1–28,800 rather than clamping it, so a value beyond that
1689    /// would pass planning, render into the package, and fail at the first session. Kubernetes
1690    /// takes the same field with no such bound, so the check is AWS's alone.
1691    #[test]
1692    fn a_lifetime_aws_would_reject_is_refused_while_planning() {
1693        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1694
1695        for seconds in [0, 28_801, 100_000] {
1696            sandbox.session.max_lifetime_seconds = Some(seconds);
1697            let error = sandbox
1698                .validate_for_platform(Platform::Aws)
1699                .expect_err("a lifetime outside what AWS runs is refused");
1700            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "{seconds}s");
1701
1702            // Kubernetes has no such ceiling, so the same declaration is fine there.
1703            sandbox
1704                .validate_for_platform(Platform::Kubernetes)
1705                .expect("the kubelet takes any activeDeadlineSeconds");
1706        }
1707
1708        sandbox.session.max_lifetime_seconds = Some(28_800);
1709        sandbox
1710            .validate_for_platform(Platform::Aws)
1711            .expect("the ceiling itself is allowed");
1712    }
1713
1714    /// An image reference Azure cannot honour is refused while planning, not at the first session.
1715    ///
1716    /// `code.image`'s own documentation gives a tag and a registry path as examples — exactly
1717    /// what Azure cannot take, so this is the shape a customer is most likely to declare.
1718    #[test]
1719    fn an_image_azure_cannot_pull_is_refused_while_planning() {
1720        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1721        // Azure enforces no declared ceiling, so a sandbox carrying limits is refused before the
1722        // image is ever read.
1723        sandbox.limits = None;
1724
1725        for image in [
1726            "ubuntu:24.04",
1727            "ghcr.io/myorg/sandbox:latest",
1728            "ubuntu@sha256:abc",
1729            "",
1730            "   ",
1731            "ubuntu latest",
1732            "ubuntu?x",
1733        ] {
1734            sandbox.code = SandboxCode::Image {
1735                image: image.to_string(),
1736            };
1737            let error = sandbox
1738                .validate_for_platform(Platform::Azure)
1739                .expect_err("an image Azure has nowhere to put is refused");
1740            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "image '{image}'");
1741
1742            // The same declaration is ordinary everywhere that pulls a reference.
1743            sandbox
1744                .validate_for_platform(Platform::Kubernetes)
1745                .expect("a registry reference is what every other backend takes");
1746        }
1747
1748        for image in ["ubuntu", "ubuntu-22.04", "debian_slim"] {
1749            sandbox.code = SandboxCode::Image {
1750                image: image.to_string(),
1751            };
1752            sandbox
1753                .validate_for_platform(Platform::Azure)
1754                .unwrap_or_else(|error| panic!("'{image}' is a catalog name: {error}"));
1755        }
1756
1757        // Surrounding space is trimmed rather than carried into the create body.
1758        sandbox.code = SandboxCode::Image {
1759            image: " ubuntu ".to_string(),
1760        };
1761        assert_eq!(
1762            sandbox
1763                .azure_catalog_image()
1764                .expect("a padded name is still a name"),
1765            "ubuntu"
1766        );
1767    }
1768
1769    /// A deadline is accepted only where the platform itself terminates on it — the kubelet's
1770    /// `activeDeadlineSeconds` and Lambda's `maximumDurationInSeconds`. Everywhere else it would
1771    /// need a reaper that does not exist, so it is refused rather than accepted and dropped.
1772    #[test]
1773    fn a_session_deadline_is_accepted_only_where_the_platform_applies_it() {
1774        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1775        sandbox.session.max_lifetime_seconds = Some(3600);
1776
1777        sandbox
1778            .validate_for_platform(Platform::Kubernetes)
1779            .expect("the kubelet enforces activeDeadlineSeconds");
1780        sandbox
1781            .validate_for_platform(Platform::Aws)
1782            .expect("Lambda terminates the MicroVM at maximumDurationInSeconds");
1783
1784        for platform in [Platform::Azure, Platform::Local] {
1785            let error = sandbox
1786                .validate_for_platform(platform)
1787                .expect_err("a deadline nothing applies must be refused");
1788            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1789        }
1790    }
1791
1792    /// A MicroVM bursts to four times its baseline with no way to opt out, so a ceiling is kept
1793    /// by choosing the size whose *peak* fits inside it. Sizing by baseline would hand back a
1794    /// sandbox that can reach four times what the customer declared.
1795    #[test]
1796    fn an_aws_size_is_chosen_so_its_peak_stays_inside_the_declared_ceiling() {
1797        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1798        let tier = sandbox
1799            .microvm_tier()
1800            .expect("2Gi/1cpu/20Gi is satisfiable");
1801
1802        assert_eq!(
1803            tier.peak_memory_mib, 2048,
1804            "the peak is the declared ceiling"
1805        );
1806        assert_eq!(
1807            tier.baseline_memory_mib, 512,
1808            "which is a quarter of it as the baseline"
1809        );
1810        assert!(tier.max_disk_mib <= 20 * 1024);
1811    }
1812
1813    /// AWS allocates one vCPU per 2GB, so a cpu ceiling below what the memory ceiling implies
1814    /// cannot be honoured together with it. Letting cpu choose the size instead would hand back a
1815    /// machine four times smaller than the memory asked for, with nothing to indicate it.
1816    #[test]
1817    fn a_cpu_ceiling_below_what_the_memory_implies_is_refused_not_quietly_downsized() {
1818        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1819        {
1820            let limits = sandbox
1821                .limits
1822                .as_mut()
1823                .expect("the fixture declares limits");
1824            limits.cpu = "1".to_string();
1825            limits.memory = "8Gi".to_string();
1826        }
1827
1828        let error = sandbox
1829            .microvm_tier()
1830            .expect_err("1 cpu and 8Gi cannot both be ceilings on AWS");
1831        assert!(
1832            error.to_string().contains("4 vCPU"),
1833            "the refusal must say what the memory ceiling implies: {error}"
1834        );
1835
1836        sandbox
1837            .limits
1838            .as_mut()
1839            .expect("the fixture declares limits")
1840            .cpu = "4".to_string();
1841        let tier = sandbox.microvm_tier().expect("4 cpu matches 8Gi");
1842        assert_eq!(tier.peak_memory_mib, 8192);
1843    }
1844
1845    /// Below AWS's smallest peak there is no size that holds the ceiling, and rounding up to the
1846    /// nearest one would silently exceed it.
1847    #[test]
1848    fn an_aws_ceiling_smaller_than_any_size_is_refused_rather_than_rounded() {
1849        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1850        sandbox
1851            .limits
1852            .as_mut()
1853            .expect("the fixture declares limits")
1854            .memory = "1Gi".to_string();
1855
1856        let error = sandbox
1857            .validate_for_platform(Platform::Aws)
1858            .expect_err("no MicroVM size peaks at or below 1Gi");
1859        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1860        assert!(
1861            error.to_string().contains("2Gi"),
1862            "the refusal must say what the smallest holdable ceiling is: {error}"
1863        );
1864    }
1865
1866    /// `Source` is a public part of the type that no backend builds: an empty image string
1867    /// schedules a pod that can never run, so the refusal has to happen at plan time and on
1868    /// every platform, not in one emitter.
1869    #[test]
1870    fn source_code_is_refused_everywhere_rather_than_producing_a_broken_manifest() {
1871        let sandbox = Sandbox::new("agent".to_string())
1872            .code(SandboxCode::Source {
1873                src: "./sandbox".to_string(),
1874                toolchain: ToolchainConfig::Docker {
1875                    dockerfile: None,
1876                    build_args: None,
1877                    target: None,
1878                },
1879            })
1880            .egress(SandboxEgress::Deny)
1881            .session(SandboxSessionPolicy {
1882                max_lifetime_seconds: None,
1883                idle_suspend_seconds: None,
1884            })
1885            .build();
1886
1887        for platform in [
1888            Platform::Aws,
1889            Platform::Azure,
1890            Platform::Gcp,
1891            Platform::Kubernetes,
1892            Platform::Local,
1893        ] {
1894            let error = sandbox
1895                .validate_for_platform(platform)
1896                .expect_err("no backend builds a sandbox image from source");
1897            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1898            assert!(
1899                error.to_string().contains("code.image"),
1900                "the refusal must say what to write instead: {error}"
1901            );
1902        }
1903    }
1904
1905    /// `validate_quantity` accepts nine suffixes. Reading only `Gi` and `Mi` would size a
1906    /// declared `4G` as though it were `4Gi`, which for a ceiling means exceeding it.
1907    #[test]
1908    fn every_accepted_unit_converts_rather_than_falling_back() {
1909        assert_eq!(quantity_mib("2Gi"), Some(2048));
1910        assert_eq!(quantity_mib("512Mi"), Some(512));
1911        assert_eq!(quantity_mib("4G"), Some(3814));
1912        assert_eq!(quantity_mib("1Ti"), Some(1024 * 1024));
1913        assert_eq!(millicores("1"), Some(1000));
1914        assert_eq!(millicores("500m"), Some(500));
1915    }
1916
1917    #[test]
1918    fn unknown_fields_are_rejected() {
1919        let json = r#"{
1920            "id": "sbx",
1921            "code": {"type": "image", "image": "ubuntu:24.04"},
1922            "limits": {"cpu": "1", "memory": "2Gi", "disk": "20Gi"},
1923            "egress": {"mode": "deny"},
1924            "session": {},
1925            "unexpected": true
1926        }"#;
1927
1928        serde_json::from_str::<Sandbox>(json).expect_err("deny_unknown_fields must reject");
1929    }
1930
1931    #[test]
1932    fn serialization_roundtrips() {
1933        let sandbox = sandbox_with(
1934            SandboxEgress::AllowDomains {
1935                domains: vec!["example.com".to_string()],
1936            },
1937            vec![8080, 9090],
1938        );
1939
1940        let json = serde_json::to_string(&sandbox).expect("serializes");
1941        let restored: Sandbox = serde_json::from_str(&json).expect("deserializes");
1942        assert_eq!(sandbox, restored);
1943    }
1944
1945    #[test]
1946    fn id_is_immutable_across_updates() {
1947        let original = sandbox_with(SandboxEgress::Deny, vec![]);
1948        let renamed = Sandbox::new("other".to_string())
1949            .code(SandboxCode::Image {
1950                image: "ubuntu".to_string(),
1951            })
1952            .limits(
1953                original
1954                    .limits
1955                    .clone()
1956                    .expect("the fixture declares limits"),
1957            )
1958            .egress(SandboxEgress::Deny)
1959            .session(SandboxSessionPolicy {
1960                max_lifetime_seconds: None,
1961                idle_suspend_seconds: None,
1962            })
1963            .build();
1964
1965        original
1966            .validate_update(&original.clone())
1967            .expect("an unchanged config is a valid update");
1968        original
1969            .validate_update(&renamed)
1970            .expect_err("renaming a sandbox is not an update");
1971    }
1972
1973    /// Azure declares an idle-suspend policy but not a wall-clock ceiling.
1974    ///
1975    /// The two travel together in `SandboxSessionPolicy` and are gated separately on purpose:
1976    /// Azure suspends on idle and has no maximum lifetime, so accepting one and refusing the
1977    /// other is the honest split rather than an inconsistency.
1978    #[test]
1979    fn azure_takes_an_idle_policy_and_still_refuses_a_lifetime_ceiling() {
1980        let with_policy = |session: SandboxSessionPolicy| {
1981            Sandbox::new("sbx".to_string())
1982                .code(SandboxCode::Image {
1983                    image: "ubuntu".to_string(),
1984                })
1985                .egress(SandboxEgress::Allow)
1986                .session(session)
1987                .build()
1988                .validate_for_platform(Platform::Azure)
1989        };
1990
1991        with_policy(SandboxSessionPolicy {
1992            max_lifetime_seconds: None,
1993            idle_suspend_seconds: Some(900),
1994        })
1995        .expect("Azure suspends a session on idle");
1996
1997        let error = with_policy(SandboxSessionPolicy {
1998            max_lifetime_seconds: Some(3600),
1999            idle_suspend_seconds: None,
2000        })
2001        .expect_err("Azure has no wall-clock ceiling to enforce one with");
2002        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2003        assert!(
2004            error.message.contains("sessionLifetime"),
2005            "names the capability: {}",
2006            error.message
2007        );
2008    }
2009
2010    /// An allowlist naming nothing is a deny-all wearing an allowlist's label.
2011    ///
2012    /// It renders as a `Deny` default with no rules — the shape the Azure provider adds a
2013    /// catch-all to avoid — and a reader scanning the declaration sees "allowDomains" and reads
2014    /// the opposite of what it does.
2015    #[test]
2016    fn an_allowlist_with_no_domains_is_refused() {
2017        let declared = |domains: Vec<String>| {
2018            Sandbox::new("sbx".to_string())
2019                .code(SandboxCode::Image {
2020                    image: "ubuntu".to_string(),
2021                })
2022                .egress(SandboxEgress::AllowDomains { domains })
2023                .session(SandboxSessionPolicy {
2024                    max_lifetime_seconds: None,
2025                    idle_suspend_seconds: None,
2026                })
2027                .build()
2028                .validate_for_platform(Platform::Azure)
2029        };
2030
2031        let error = declared(vec![]).expect_err("an empty allowlist must be refused");
2032        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2033
2034        declared(vec!["api.example.com".to_string()])
2035            .expect("a named domain is what an allowlist is for");
2036    }
2037
2038    /// The two expressible modes map to the boolean; a host list maps to nothing so the caller has
2039    /// to refuse rather than silently pick a side.
2040    #[test]
2041    fn internet_access_switch_maps_only_the_two_expressible_modes() {
2042        assert_eq!(SandboxEgress::Allow.internet_access_switch(), Some(true));
2043        assert_eq!(SandboxEgress::Deny.internet_access_switch(), Some(false));
2044        assert_eq!(
2045            SandboxEgress::AllowDomains {
2046                domains: vec!["api.example.com".to_string()]
2047            }
2048            .internet_access_switch(),
2049            None,
2050            "a host list has no boolean and must not be approximated"
2051        );
2052    }
2053}