Skip to main content

alien_core/resources/
sandbox.rs

1//! Sandbox resource for running untrusted code in an isolated environment.
2//!
3//! A Sandbox is a session-oriented resource: the declaration provisions a durable parent, and
4//! the application creates and destroys individual sessions through its binding at runtime.
5//!
6//! The capability set differs per platform and is published rather than assumed. Calling an
7//! unsupported capability is a typed error naming both the platform and the capability, so a
8//! portable application can branch on `SandboxCapabilities` before it calls.
9
10use crate::error::{ErrorData, Result};
11use crate::resource::{ResourceDefinition, ResourceOutputsDefinition, ResourceRef, ResourceType};
12use crate::resources::ToolchainConfig;
13use crate::Platform;
14use alien_error::AlienError;
15use bon::Builder;
16use serde::{Deserialize, Serialize};
17use std::any::Any;
18use std::fmt::Debug;
19
20/// Specifies where the sandbox's root filesystem comes from.
21#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
22#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
23#[serde(rename_all = "camelCase", tag = "type")]
24pub enum SandboxCode {
25    /// A prebuilt container image used as the sandbox root filesystem.
26    #[serde(rename_all = "camelCase")]
27    Image {
28        /// Image reference (e.g. `ubuntu:24.04`, `ghcr.io/myorg/sandbox:latest`).
29        ///
30        /// Two backends narrow it in opposite directions: AWS wants an `s3://` bundle, Azure a
31        /// bare catalog name such as `ubuntu`. Each refuses the other's shape while planning.
32        image: String,
33    },
34    /// Source built into a sandbox image at deploy time.
35    #[serde(rename_all = "camelCase")]
36    Source {
37        /// The source directory to build from
38        src: String,
39        /// Toolchain configuration with type-safe options
40        toolchain: ToolchainConfig,
41    },
42}
43
44/// Hard ceilings enforced on a sandbox session.
45///
46/// These are limits, not scheduling requests. Untrusted code does not respect a hint, so every
47/// field is enforced by the platform and a platform that cannot enforce one is rejected at plan
48/// time rather than silently ignoring it.
49#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
50#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
51#[serde(rename_all = "camelCase", deny_unknown_fields)]
52pub struct SandboxLimits {
53    /// CPU ceiling in cores or millicores (e.g. `"1"`, `"500m"`)
54    pub cpu: String,
55    /// Memory ceiling (e.g. `"2Gi"`, `"512Mi"`)
56    pub memory: String,
57    /// Disk ceiling (e.g. `"20Gi"`)
58    pub disk: String,
59    /// Maximum number of processes, which bounds fork bombs.
60    ///
61    /// Optional because only a container runtime has the primitive: Kubernetes sets a pid ceiling
62    /// per node, not per pod, and neither AWS MicroVMs nor Azure sandboxes expose one. Declaring
63    /// it on a platform that cannot apply it is refused at plan time.
64    #[serde(default, skip_serializing_if = "Option::is_none")]
65    pub max_processes: Option<u32>,
66}
67
68/// One of the five sizes a Lambda MicroVM can be built at.
69///
70/// AWS has no ceiling knob: `minimumMemoryInMiB` sets a *baseline* and a running MicroVM bursts
71/// vertically to four times it with no way to opt out. A declared ceiling is therefore honoured by
72/// picking the tier whose **peak** stays inside it, not the tier whose baseline matches it.
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74pub struct MicrovmTier {
75    /// What `minimumMemoryInMiB` is set to.
76    pub baseline_memory_mib: i64,
77    /// The most memory the MicroVM can reach, in MiB.
78    pub peak_memory_mib: i64,
79    /// The most vCPU the MicroVM can reach.
80    pub peak_vcpu: u32,
81    /// The most disk the MicroVM can use, in MiB.
82    pub max_disk_mib: i64,
83}
84
85/// The published sizes, smallest first. Baseline memory to vCPU is 2 GB per vCPU, peak is four
86/// times baseline, and disk is fixed per tier rather than independently selectable.
87/// Longest life AWS will run a MicroVM for, from `RunMicrovm`'s `maximumDurationInSeconds`.
88const AWS_MAX_SESSION_LIFETIME_SECONDS: u32 = 28_800;
89
90const MICROVM_TIERS: &[MicrovmTier] = &[
91    MicrovmTier {
92        baseline_memory_mib: 512,
93        peak_memory_mib: 2048,
94        peak_vcpu: 1,
95        max_disk_mib: 8192,
96    },
97    MicrovmTier {
98        baseline_memory_mib: 1024,
99        peak_memory_mib: 4096,
100        peak_vcpu: 2,
101        max_disk_mib: 8192,
102    },
103    MicrovmTier {
104        baseline_memory_mib: 2048,
105        peak_memory_mib: 8192,
106        peak_vcpu: 4,
107        max_disk_mib: 8192,
108    },
109    MicrovmTier {
110        baseline_memory_mib: 4096,
111        peak_memory_mib: 16384,
112        peak_vcpu: 8,
113        max_disk_mib: 16384,
114    },
115    MicrovmTier {
116        baseline_memory_mib: 8192,
117        peak_memory_mib: 32768,
118        peak_vcpu: 16,
119        max_disk_mib: 32768,
120    },
121];
122
123/// Outbound network policy for a sandbox.
124#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
125#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
126#[serde(rename_all = "camelCase", tag = "mode")]
127pub enum SandboxEgress {
128    /// No outbound network access.
129    ///
130    /// Routed traffic only. Link-local is not outbound and no backend's egress control reaches
131    /// it, so this is not a boundary against instance metadata.
132    Deny,
133    /// Unrestricted outbound access to the public internet, and none to private ranges or the
134    /// deployment's own network.
135    ///
136    /// Link-local carries the same exception as `Deny`. AWS and Kubernetes deliver both halves.
137    /// Azure and GCP deliver the first only: one matches host patterns and the other is a single
138    /// switch, so neither can name an address range to exclude.
139    Allow,
140    /// Outbound access only to the listed hostnames.
141    ///
142    /// Azure alone expresses it: its egress proxy matches on host pattern. The others filter by
143    /// CIDR or carry a single switch, and both would approximate the list rather than keep it.
144    #[serde(rename_all = "camelCase")]
145    AllowDomains {
146        /// Hostnames the sandbox may reach
147        domains: Vec<String>,
148    },
149}
150
151impl SandboxEgress {
152    /// The single outbound switch for a backend that has no host matcher, or `None` for a mode a
153    /// boolean cannot carry.
154    ///
155    /// `AllowDomains` needs a host list, so it maps to nothing and each caller refuses it in its
156    /// own error naming the sandbox. One source for what a mode means, so a template and a session
157    /// cannot disagree on it.
158    pub fn internet_access_switch(&self) -> Option<bool> {
159        match self {
160            SandboxEgress::Allow => Some(true),
161            SandboxEgress::Deny => Some(false),
162            SandboxEgress::AllowDomains { .. } => None,
163        }
164    }
165}
166
167/// How long a session may live and when it is suspended.
168#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
169#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
170#[serde(rename_all = "camelCase", deny_unknown_fields)]
171pub struct SandboxSessionPolicy {
172    /// Wall-clock ceiling on a single session, after which the platform terminates it.
173    ///
174    /// Optional because not every backend has the primitive: Kubernetes has
175    /// `activeDeadlineSeconds` and AWS `maximumDurationInSeconds`, while neither Azure nor Local
176    /// expose one, so declaring a ceiling there is refused at plan time rather than accepted and
177    /// never applied. AWS caps it at 8 hours.
178    #[serde(default, skip_serializing_if = "Option::is_none")]
179    pub max_lifetime_seconds: Option<u32>,
180    /// Idle period after which the session is suspended, where the platform supports it
181    #[serde(skip_serializing_if = "Option::is_none")]
182    pub idle_suspend_seconds: Option<u32>,
183}
184
185/// What a platform's sandbox backend can actually do.
186///
187/// Published so portable code can branch before calling rather than discovering a gap through
188/// an error. Every field here corresponds to a capability that at least one platform lacks;
189/// create, exec and terminate are the guaranteed floor and are therefore not listed.
190#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
191#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
192#[serde(rename_all = "camelCase", deny_unknown_fields)]
193pub struct SandboxCapabilities {
194    /// Files can be moved in and out of a session
195    pub files: bool,
196    /// A later call can reach a session created by an earlier one
197    pub reconnect: bool,
198    /// An authenticated, port-scoped capability to reach a service inside the sandbox
199    pub preview: bool,
200    /// Session state can be suspended and resumed
201    pub suspend_resume: bool,
202    /// A session's full state can be captured and used to create another
203    pub snapshot: bool,
204    /// Egress can be restricted to a hostname allowlist
205    pub domain_egress_rules: bool,
206    /// Whether a declared `deny` is actually enforced, rather than accepted and dropped
207    pub egress_deny: bool,
208    /// The platform enforces the declared cpu, memory and disk ceilings
209    pub enforced_limits: bool,
210    /// The platform can cap how many processes a session runs
211    pub process_limit: bool,
212    /// The platform terminates a session at a declared wall-clock deadline
213    pub session_lifetime: bool,
214    /// A command runs in its own PID namespace and cannot see or signal the agent's processes.
215    ///
216    /// Only where an agent runs as root. Creating the namespace needs `CAP_SYS_ADMIN`, and the
217    /// Kubernetes sandbox pod drops every capability — which is also what denies `ptrace` by
218    /// construction, so granting it there would remove a lock to add one.
219    pub supervisor_pid_namespace: bool,
220    /// The process supervising a command is a different identity from the command.
221    ///
222    /// False where a command runs as the agent's own user: it can then read the supervisor's
223    /// environment and signal it. Separate from `supervisorPidNamespace`, which is about
224    /// visibility rather than identity — a backend can have one without the other.
225    pub supervisor_isolation: bool,
226}
227
228impl SandboxCapabilities {
229    /// Returns what the given platform's sandbox backend supports.
230    ///
231    /// Errors for platforms with no sandbox backend, rather than returning an all-false set —
232    /// "every capability is missing" and "this platform has no sandboxes" are different
233    /// conditions and an application should not have to tell them apart by inspection.
234    pub fn for_platform(platform: Platform) -> Result<Self> {
235        match platform {
236            Platform::Aws => Ok(Self {
237                files: true,
238                reconnect: true,
239                preview: true,
240                suspend_resume: true,
241                snapshot: false,
242                domain_egress_rules: false,
243                egress_deny: true,
244                enforced_limits: true,
245                // Nothing in the API bounds process count.
246                process_limit: false,
247                // `maximumDurationInSeconds` on `RunMicrovm`, which Lambda enforces by
248                // terminating the MicroVM. Capped at 8 hours by the service.
249                session_lifetime: true,
250                // Measured, not assumed: the agent inside a Lambda MicroVM runs as uid 0 with
251                // `CapEff: 00000000a80425fb`, the standard container default set, which excludes
252                // `CAP_SYS_ADMIN`. It can drop privilege (`CAP_SETUID`/`CAP_SETGID` are held) and
253                // it cannot create a namespace. No backend offers this today.
254                supervisor_pid_namespace: false,
255                // The agent runs as uid 0 and `setuid`s the command to uid 60000, so the command
256                // runs under a different identity than the process supervising it.
257                supervisor_isolation: true,
258            }),
259            Platform::Azure => Ok(Self {
260                files: true,
261                reconnect: true,
262                // A sandbox port carries a URL and an auth config, and the auth config offers two
263                // things: anonymous, or Entra ID with an allowlist of human email addresses.
264                // Neither is a credential scoped to a port for a fixed time, which is what a
265                // preview capability is. Returning the anonymous URL would publish the port.
266                preview: false,
267                suspend_resume: true,
268                // The one cloud of the five that could offer this, and the blocker is ours:
269                // `snapshot()` returns an id and `CreateSessionRequest` has no field to consume
270                // one, so no backend can complete the round trip. Nothing in the resource model
271                // owns such an artifact either, and Microsoft states snapshots are not garbage
272                // collected — an id with no owner is a bill that grows.
273                snapshot: false,
274                domain_egress_rules: true,
275                egress_deny: true,
276                enforced_limits: false,
277                process_limit: false,
278                // Auto-suspend and auto-delete exist; a wall-clock ceiling does not. Accepting
279                // `maxLifetimeSeconds` here would be the silent no-op the capability set exists
280                // to prevent, so this is a decision rather than a gap.
281                session_lifetime: false,
282                // No Alien process inside an Azure sandbox, so there is no supervisor to isolate.
283                supervisor_pid_namespace: false,
284                // No Alien process runs the command at all — the platform's own data plane does,
285                // so there is no separate supervisor identity to speak of.
286                supervisor_isolation: false,
287            }),
288            Platform::Gcp => Ok(Self::gcp_agent_platform()),
289            // Preview needs a gateway that validates a session-and-port capability, and that
290            // gateway does not exist yet.
291            Platform::Kubernetes => Ok(Self {
292                files: true,
293                reconnect: true,
294                preview: false,
295                suspend_resume: false,
296                snapshot: false,
297                domain_egress_rules: false,
298                egress_deny: true,
299                enforced_limits: true,
300                // A pid ceiling is a kubelet setting per node, not a pod field.
301                process_limit: false,
302                // `activeDeadlineSeconds` on the pod, which the kubelet enforces.
303                session_lifetime: true,
304                // The pod drops every capability, including the `CAP_SYS_ADMIN` the agent would
305                // need to unshare. That is also what denies `ptrace`, so this stays false rather
306                // than the pod being weakened to make it true.
307                supervisor_pid_namespace: false,
308                // The pod pins one uid (`run_as_user: 65534` on both pod and container) with
309                // `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no
310                // process can setuid to split the command off from a supervisor. No uid split is
311                // possible, so none exists.
312                supervisor_isolation: false,
313            }),
314            Platform::Local => Ok(Self {
315                files: true,
316                reconnect: true,
317                preview: true,
318                suspend_resume: false,
319                snapshot: false,
320                domain_egress_rules: false,
321                egress_deny: true,
322                enforced_limits: true,
323                // Docker's `--pids-limit`.
324                process_limit: true,
325                session_lifetime: false,
326                // Local has no in-sandbox agent: the manager drives Docker from outside, so
327                // there is no supervisor inside the sandbox to isolate from.
328                supervisor_pid_namespace: false,
329                // The supervisor is the manager on the host, outside the container entirely, and
330                // `docker exec` runs the command as the workload uid — a different identity by
331                // construction.
332                supervisor_isolation: true,
333            }),
334            Platform::Machines | Platform::Test => {
335                Err(AlienError::new(ErrorData::SandboxPlatformUnsupported {
336                    platform: platform.to_string(),
337                }))
338            }
339        }
340    }
341
342    /// What the GCP Agent Platform sandbox backend supports; the body of the `Platform::Gcp` arm.
343    pub fn gcp_agent_platform() -> Self {
344        Self {
345            // Agent file operations move over the session envelope.
346            files: true,
347            // Reaching a session across processes is safe because `generation` is derived from the
348            // container boot id read through the agent's health op, so a caller detects a container
349            // replaced under a stable session name rather than reconnecting to a blank one.
350            reconnect: true,
351            // No method mints a port-scoped ingress capability; the only ingress is `:execute`.
352            preview: false,
353            // `:pause` and `:resume` preserve the running container.
354            suspend_resume: true,
355            // A session's state can be captured and used to create another.
356            snapshot: true,
357            // Egress is shaped by VPC and DNS peering, which is not a hostname allowlist.
358            domain_egress_rules: false,
359            // A declared `deny` blocks both routed egress and DNS.
360            egress_deny: true,
361            // The declared ceilings are enforced, but by terminating the session on breach rather
362            // than by refusing the allocation — a caller reading `true` should expect the session
363            // to die, not a clean error at the point of the request.
364            enforced_limits: true,
365            // No ceiling on process count is observed.
366            process_limit: false,
367            // `ttl` maps to a session `expireTime` the platform terminates at.
368            session_lifetime: true,
369            // No PID-namespace isolation between the command and anything supervising it.
370            supervisor_pid_namespace: false,
371            // No separate supervisor identity: the command is not run under a different identity
372            // than the process supervising it.
373            supervisor_isolation: false,
374        }
375    }
376
377    /// Returns a typed error if the named capability is absent on this platform.
378    pub fn require(&self, capability: SandboxCapability, platform: Platform) -> Result<()> {
379        let available = match capability {
380            SandboxCapability::Files => self.files,
381            SandboxCapability::Reconnect => self.reconnect,
382            SandboxCapability::Preview => self.preview,
383            SandboxCapability::SuspendResume => self.suspend_resume,
384            SandboxCapability::Snapshot => self.snapshot,
385            SandboxCapability::DomainEgressRules => self.domain_egress_rules,
386            SandboxCapability::EgressDeny => self.egress_deny,
387            SandboxCapability::EnforcedLimits => self.enforced_limits,
388            SandboxCapability::ProcessLimit => self.process_limit,
389            SandboxCapability::SessionLifetime => self.session_lifetime,
390            SandboxCapability::SupervisorPidNamespace => self.supervisor_pid_namespace,
391            SandboxCapability::SupervisorIsolation => self.supervisor_isolation,
392        };
393
394        if available {
395            return Ok(());
396        }
397
398        Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
399            capability: capability.as_str().to_string(),
400            platform: platform.to_string(),
401        }))
402    }
403}
404
405/// Names a single sandbox capability, so an unsupported call can report which one it needed.
406#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
407#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
408#[serde(rename_all = "camelCase")]
409pub enum SandboxCapability {
410    /// Moving files in and out of a session
411    Files,
412    /// Reaching a session created by an earlier call
413    Reconnect,
414    /// An authenticated, port-scoped ingress capability
415    Preview,
416    /// Suspending and resuming session state
417    SuspendResume,
418    /// Capturing full session state
419    Snapshot,
420    /// Restricting egress to a hostname allowlist
421    DomainEgressRules,
422    /// Refusing outbound access when a sandbox declares none
423    EgressDeny,
424    /// Platform-enforced resource ceilings
425    EnforcedLimits,
426    /// A ceiling on the number of processes a session may run
427    ProcessLimit,
428    /// A wall-clock ceiling on a session, applied by the platform rather than by a caller
429    SessionLifetime,
430    /// A command runs in its own PID namespace, isolated from the agent supervising it
431    SupervisorPidNamespace,
432    /// A command runs under a different identity than the process supervising it
433    SupervisorIsolation,
434}
435
436impl SandboxCapability {
437    /// Returns the stable identifier used in errors and capability queries.
438    pub fn as_str(&self) -> &'static str {
439        match self {
440            Self::Files => "files",
441            Self::Reconnect => "reconnect",
442            Self::Preview => "preview",
443            Self::SuspendResume => "suspendResume",
444            Self::Snapshot => "snapshot",
445            Self::DomainEgressRules => "domainEgressRules",
446            Self::EgressDeny => "egressDeny",
447            Self::EnforcedLimits => "enforcedLimits",
448            Self::ProcessLimit => "processLimit",
449            Self::SessionLifetime => "sessionLifetime",
450            Self::SupervisorPidNamespace => "supervisorPidNamespace",
451            Self::SupervisorIsolation => "supervisorIsolation",
452        }
453    }
454}
455
456/// An isolated environment for running untrusted code, created per session at runtime.
457#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Builder)]
458#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
459#[serde(rename_all = "camelCase", deny_unknown_fields)]
460#[builder(start_fn = new)]
461pub struct Sandbox {
462    /// Identifier for the sandbox. Must contain only alphanumeric characters, hyphens, and
463    /// underscores ([A-Za-z0-9-_]). Maximum 64 characters.
464    #[builder(start_fn)]
465    pub id: String,
466    /// Where the sandbox's root filesystem comes from
467    pub code: SandboxCode,
468    /// Enforced resource ceilings.
469    ///
470    /// Optional because not every platform can enforce them, and a declaration that names none
471    /// takes the platform's own defaults. Naming them on a platform that cannot enforce them is
472    /// rejected at plan time rather than silently ignored.
473    #[serde(skip_serializing_if = "Option::is_none")]
474    pub limits: Option<SandboxLimits>,
475    /// Outbound network policy
476    pub egress: SandboxEgress,
477    /// Session lifetime and idle behaviour
478    pub session: SandboxSessionPolicy,
479    /// Ports eligible for a preview capability. A port not listed here can never be exposed,
480    /// so an application cannot widen its own ingress at runtime.
481    #[builder(default)]
482    #[serde(default, skip_serializing_if = "Vec::is_empty")]
483    pub preview_ports: Vec<u16>,
484}
485
486/// Whether the artifact being rendered restricts which network modes it accepts.
487///
488/// A sandbox is not emitted on a Kubernetes target, so nothing there routes egress through a
489/// connector and the default network stays a working answer. Every site that withholds the mode,
490/// explains the restriction, or renders a branch for it has to ask this one question — asking the
491/// stack directly is how they came to disagree.
492pub fn restricts_network_mode(stack: &crate::Stack, targets_kubernetes: bool) -> bool {
493    !targets_kubernetes && stack_needs_named_subnets_at_setup(stack)
494}
495
496/// Whether any sandbox in the stack forces setup to name subnets.
497///
498/// A restricted sandbox routes session egress through a VPC connector, and neither generator can
499/// enumerate the account default VPC's subnets, so that mode leaves the connector without any and
500/// it fails at create. Callers that render an artifact want [`restricts_network_mode`] instead:
501/// this one answers for the declaration, which on a Kubernetes target is not what gets emitted.
502pub fn stack_needs_named_subnets_at_setup(stack: &crate::Stack) -> bool {
503    stack.resources().any(|(_resource_id, resource)| {
504        resource
505            .config
506            .downcast_ref::<Sandbox>()
507            .is_some_and(|sandbox| !matches!(sandbox.egress, SandboxEgress::Allow))
508    })
509}
510
511impl Sandbox {
512    /// The resource type identifier for Sandbox
513    pub const RESOURCE_TYPE: ResourceType = ResourceType::from_static("sandbox");
514
515    /// Returns the sandbox's unique identifier.
516    pub fn id(&self) -> &str {
517        &self.id
518    }
519
520    /// The declared ceilings, or the defaults a platform applies when none were named.
521    ///
522    /// Backends want a concrete set: a sandbox with no declared ceilings still runs inside
523    /// whatever the platform gives it, and a backend that had to branch on `None` would end up
524    /// inventing its own default anyway.
525    pub fn resolved_limits(&self) -> SandboxLimits {
526        self.limits.clone().unwrap_or_else(default_limits)
527    }
528
529    /// Validates the declaration against what the target platform can enforce.
530    ///
531    /// Runs at plan time so an unenforceable limit or an unsupported egress mode fails before
532    /// anything is provisioned, rather than at the first exec.
533    pub fn validate_for_platform(&self, platform: Platform) -> Result<()> {
534        let capabilities = SandboxCapabilities::for_platform(platform)?;
535
536        // No backend builds a sandbox image from source: an empty image string schedules a pod
537        // that can never run, the silent no-op the capability contract forbids — the failure
538        // has to land here instead.
539        if let SandboxCode::Source { .. } = &self.code {
540            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
541                resource_id: self.id.clone(),
542                field: "code".to_string(),
543                value: "source".to_string(),
544                reason: "no sandbox backend builds an image from source yet; give code.image a \
545                         prebuilt reference"
546                    .to_string(),
547            }));
548        }
549
550        // Read before the limits, because the image is declared whether or not any are.
551        if platform == Platform::Azure {
552            self.azure_catalog_image()?;
553        }
554
555        let Some(limits) = self.limits.as_ref() else {
556            // Nothing declared, so nothing to enforce and nothing to reject.
557            return self.validate_capabilities(&capabilities, platform);
558        };
559
560        validate_quantity(&self.id, "cpu", &limits.cpu)?;
561        validate_quantity(&self.id, "memory", &limits.memory)?;
562        validate_quantity(&self.id, "disk", &limits.disk)?;
563
564        if let Some(max_processes) = limits.max_processes {
565            if max_processes == 0 {
566                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
567                    resource_id: self.id.clone(),
568                    field: "maxProcesses".to_string(),
569                    value: "0".to_string(),
570                    reason: "a sandbox that may run no processes cannot run code".to_string(),
571                }));
572            }
573            capabilities.require(SandboxCapability::ProcessLimit, platform)?;
574        }
575
576        // Declaring limits a platform ignores is worse than not declaring them: the stack reads
577        // as bounded while the sandbox is not.
578        capabilities.require(SandboxCapability::EnforcedLimits, platform)?;
579
580        if platform == Platform::Aws {
581            // Refused here rather than at emit so a customer sees it while planning, and so both
582            // package formats inherit the same answer.
583            self.microvm_tier()?;
584
585            // The ceiling is Lambda's, and it rejects the run rather than clamping — so a value
586            // outside it would pass planning, render into the package, and fail at the first
587            // session. Kubernetes takes the same field with no such bound, which is why this
588            // sits under the AWS gate rather than on the type.
589            if let Some(seconds) = self.session.max_lifetime_seconds {
590                if !(1..=AWS_MAX_SESSION_LIFETIME_SECONDS).contains(&seconds) {
591                    return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
592                        resource_id: self.id.clone(),
593                        field: "maxLifetimeSeconds".to_string(),
594                        value: seconds.to_string(),
595                        reason: format!(
596                            "AWS runs a MicroVM for between 1 and \
597                             {AWS_MAX_SESSION_LIFETIME_SECONDS} seconds"
598                        ),
599                    }));
600                }
601            }
602        }
603
604        self.validate_capabilities(&capabilities, platform)
605    }
606
607    /// The catalog disk image Azure creates a session from.
608    ///
609    /// Azure names a public catalog entry rather than pulling a reference, so a registry path,
610    /// tag or digest has nowhere to go. An allowlist, because the answer to "what else could be
611    /// in there" is a name the data plane rejects at the first session, long after the apply.
612    pub fn azure_catalog_image(&self) -> Result<&str> {
613        let refused = |value: &str, reason: &str| {
614            AlienError::new(ErrorData::SandboxLimitInvalid {
615                resource_id: self.id.clone(),
616                field: "code.image".to_string(),
617                value: value.to_string(),
618                reason: reason.to_string(),
619            })
620        };
621
622        let SandboxCode::Image { image } = &self.code else {
623            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
624                resource_id: self.id.clone(),
625                field: "code".to_string(),
626                value: "source".to_string(),
627                reason: "no sandbox backend builds an image from source yet".to_string(),
628            }));
629        };
630
631        let image = image.trim();
632        if image.is_empty() {
633            return Err(refused(image, "a sandbox has to name an image"));
634        }
635        if !image
636            .chars()
637            .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '-'))
638        {
639            return Err(refused(
640                image,
641                "Azure creates a session from a public catalog disk image, so code.image must be \
642                 a bare catalog name such as 'ubuntu'",
643            ));
644        }
645        Ok(image)
646    }
647
648    /// The MicroVM size that keeps every declared ceiling, or why none does.
649    ///
650    /// AWS sizes are discrete and a running MicroVM bursts to four times its baseline, so the
651    /// only tier that honours a ceiling is one whose peak fits inside it. A declaration no tier
652    /// satisfies is refused: shipping the nearest size would give the customer a sandbox that
653    /// exceeds the bound they wrote down.
654    pub fn microvm_tier(&self) -> Result<MicrovmTier> {
655        let Some(limits) = self.limits.as_ref() else {
656            // Nothing declared: AWS's own default baseline, which is also `default_limits`.
657            return Ok(MICROVM_TIERS[2]);
658        };
659
660        let memory_mib = quantity_mib(&limits.memory).ok_or_else(|| {
661            AlienError::new(ErrorData::SandboxLimitInvalid {
662                resource_id: self.id.clone(),
663                field: "memory".to_string(),
664                value: limits.memory.clone(),
665                reason: "AWS sizes a MicroVM in whole MiB".to_string(),
666            })
667        })?;
668        let disk_mib = quantity_mib(&limits.disk).ok_or_else(|| {
669            AlienError::new(ErrorData::SandboxLimitInvalid {
670                resource_id: self.id.clone(),
671                field: "disk".to_string(),
672                value: limits.disk.clone(),
673                reason: "AWS sizes a MicroVM's disk in whole MiB".to_string(),
674            })
675        })?;
676        let cpu_millicores = millicores(&limits.cpu).ok_or_else(|| {
677            AlienError::new(ErrorData::SandboxLimitInvalid {
678                resource_id: self.id.clone(),
679                field: "cpu".to_string(),
680                value: limits.cpu.clone(),
681                reason: "expected cores or millicores".to_string(),
682            })
683        })?;
684
685        // Memory and disk choose the size; cpu is then checked rather than used to choose.
686        // AWS couples cpu to memory at 2 GB per vCPU, so letting a low cpu ceiling select the
687        // size too would quietly hand back a machine four times smaller than the memory ceiling
688        // asked for, with nothing to indicate it.
689        let sized = |tier: &&MicrovmTier| {
690            tier.peak_memory_mib <= memory_mib && tier.max_disk_mib <= disk_mib
691        };
692
693        let tier = MICROVM_TIERS
694            .iter()
695            .rev()
696            .find(sized)
697            .copied()
698            .ok_or_else(|| {
699                AlienError::new(ErrorData::SandboxLimitInvalid {
700                    resource_id: self.id.clone(),
701                    field: "memory".to_string(),
702                    value: limits.memory.clone(),
703                    reason: format!(
704                        "a Lambda MicroVM bursts to four times its baseline, so the smallest \
705                         ceiling AWS can hold is 2Gi memory with 8Gi disk; '{}' memory and '{}' \
706                         disk fit no size",
707                        limits.memory, limits.disk
708                    ),
709                })
710            })?;
711
712        let required_millicores = i64::from(tier.peak_vcpu) * 1000;
713        if cpu_millicores < required_millicores {
714            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
715                resource_id: self.id.clone(),
716                field: "cpu".to_string(),
717                value: limits.cpu.clone(),
718                reason: format!(
719                    "AWS allocates one vCPU per 2GB, so a MicroVM sized to a '{}' memory ceiling \
720                     reaches {} vCPU; declare cpu '{}' or lower the memory ceiling",
721                    limits.memory, tier.peak_vcpu, tier.peak_vcpu
722                ),
723            }));
724        }
725
726        Ok(tier)
727    }
728
729    /// The capability checks that do not depend on declared limits.
730    fn validate_capabilities(
731        &self,
732        capabilities: &SandboxCapabilities,
733        platform: Platform,
734    ) -> Result<()> {
735        if matches!(self.egress, SandboxEgress::AllowDomains { .. }) {
736            capabilities.require(SandboxCapability::DomainEgressRules, platform)?;
737        }
738
739        // `allow` asks for no restriction, so a backend that ignores it fails loudly on the first
740        // blocked connection. `deny` asks for one, and a backend that ignores it puts untrusted
741        // code on the internet with nothing to notice — so only this direction is gated.
742        // An empty list is not a restriction anyone wrote down: it renders as a deny-all wearing
743        // an allowlist's label, which reads at a glance as the opposite of what it does.
744        if let SandboxEgress::AllowDomains { domains } = &self.egress {
745            if domains.is_empty() {
746                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
747                    resource_id: self.id.clone(),
748                    field: "egress.domains".to_string(),
749                    value: "[]".to_string(),
750                    reason: "an allowlist naming no domain denies everything; declare \
751                             egress: deny if that is what was meant"
752                        .to_string(),
753                }));
754            }
755        }
756
757        if matches!(self.egress, SandboxEgress::Deny) {
758            capabilities.require(SandboxCapability::EgressDeny, platform)?;
759        }
760
761        if !self.preview_ports.is_empty() {
762            capabilities.require(SandboxCapability::Preview, platform)?;
763        }
764
765        if self.session.idle_suspend_seconds.is_some() {
766            capabilities.require(SandboxCapability::SuspendResume, platform)?;
767        }
768
769        if self.session.max_lifetime_seconds.is_some() {
770            capabilities.require(SandboxCapability::SessionLifetime, platform)?;
771        }
772
773        Ok(())
774    }
775}
776
777/// Ceilings applied when a declaration names none.
778///
779/// Modest on purpose: an undeclared sandbox is one whose author did not think about sizing, and
780/// the safe reading of that is a small box rather than a generous one.
781fn default_limits() -> SandboxLimits {
782    SandboxLimits {
783        cpu: "1".to_string(),
784        memory: "2Gi".to_string(),
785        disk: "8Gi".to_string(),
786        max_processes: None,
787    }
788}
789
790/// Validates a Kubernetes-style resource quantity such as `500m`, `2Gi` or `1`.
791fn validate_quantity(resource_id: &str, field: &str, value: &str) -> Result<()> {
792    let invalid = |reason: &str| {
793        AlienError::new(ErrorData::SandboxLimitInvalid {
794            resource_id: resource_id.to_string(),
795            field: field.to_string(),
796            value: value.to_string(),
797            reason: reason.to_string(),
798        })
799    };
800
801    let digits_end = value
802        .find(|c: char| !c.is_ascii_digit() && c != '.')
803        .unwrap_or(value.len());
804    let (number, suffix) = value.split_at(digits_end);
805
806    let parsed: f64 = number
807        .parse()
808        .map_err(|_| invalid("expected a number, optionally followed by a unit suffix"))?;
809
810    if parsed <= 0.0 {
811        return Err(invalid("must be greater than zero"));
812    }
813
814    const SUFFIXES: &[&str] = &["", "m", "k", "M", "G", "T", "Ki", "Mi", "Gi", "Ti"];
815    if !SUFFIXES.contains(&suffix) {
816        return Err(invalid(
817            "unit must be one of m, k, M, G, T, Ki, Mi, Gi, Ti, or absent",
818        ));
819    }
820
821    Ok(())
822}
823
824/// Splits a quantity into its number and unit suffix.
825fn split_quantity(value: &str) -> Option<(f64, &str)> {
826    let trimmed = value.trim();
827    let digits_end = trimmed
828        .find(|c: char| !c.is_ascii_digit() && c != '.')
829        .unwrap_or(trimmed.len());
830    let (number, suffix) = trimmed.split_at(digits_end);
831    number.parse().ok().map(|number| (number, suffix))
832}
833
834/// A memory or disk quantity in whole MiB, rounded down.
835///
836/// Every suffix `validate_quantity` accepts is handled here. Reading only `Gi` and `Mi` and
837/// falling back for the rest would turn a declared `4G` into a different size than the customer
838/// asked for, which for a ceiling means a sandbox larger than its bound.
839pub fn quantity_mib(value: &str) -> Option<i64> {
840    let (number, suffix) = split_quantity(value)?;
841    let bytes = match suffix {
842        "" => number,
843        "k" => number * 1e3,
844        "M" => number * 1e6,
845        "G" => number * 1e9,
846        "T" => number * 1e12,
847        "Ki" => number * 1024.0,
848        "Mi" => number * 1024.0 * 1024.0,
849        "Gi" => number * 1024.0 * 1024.0 * 1024.0,
850        "Ti" => number * 1024.0 * 1024.0 * 1024.0 * 1024.0,
851        // `m` is a millicore suffix; memory has no use for it.
852        _ => return None,
853    };
854    Some((bytes / (1024.0 * 1024.0)) as i64)
855}
856
857/// A CPU quantity in millicores.
858pub fn millicores(value: &str) -> Option<i64> {
859    let (number, suffix) = split_quantity(value)?;
860    match suffix {
861        "" => Some((number * 1000.0) as i64),
862        "m" => Some(number as i64),
863        _ => None,
864    }
865}
866
867/// Outputs generated by a successfully provisioned Sandbox parent.
868#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
869#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
870#[serde(rename_all = "camelCase")]
871pub struct SandboxOutputs {
872    /// Name of the durable parent that sessions are created inside
873    pub parent_name: String,
874    /// Platform-specific identifier for the parent (image ARN, sandbox group id, namespace)
875    #[serde(skip_serializing_if = "Option::is_none")]
876    pub identifier: Option<String>,
877    /// Data-plane endpoint sessions are created through, where the platform has one
878    #[serde(skip_serializing_if = "Option::is_none")]
879    pub endpoint: Option<String>,
880}
881
882impl ResourceOutputsDefinition for SandboxOutputs {
883    fn get_resource_type(&self) -> ResourceType {
884        Sandbox::RESOURCE_TYPE
885    }
886
887    fn as_any(&self) -> &dyn Any {
888        self
889    }
890
891    fn box_clone(&self) -> Box<dyn ResourceOutputsDefinition> {
892        Box::new(self.clone())
893    }
894
895    fn outputs_eq(&self, other: &dyn ResourceOutputsDefinition) -> bool {
896        other.as_any().downcast_ref::<SandboxOutputs>() == Some(self)
897    }
898
899    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
900        serde_json::to_value(self)
901    }
902}
903
904impl ResourceDefinition for Sandbox {
905    fn get_resource_type(&self) -> ResourceType {
906        Self::RESOURCE_TYPE
907    }
908
909    fn id(&self) -> &str {
910        &self.id
911    }
912
913    fn get_dependencies(&self) -> Vec<ResourceRef> {
914        Vec::new()
915    }
916
917    fn validate_update(&self, new_config: &dyn ResourceDefinition) -> Result<()> {
918        let new_sandbox = new_config
919            .as_any()
920            .downcast_ref::<Sandbox>()
921            .ok_or_else(|| {
922                AlienError::new(ErrorData::UnexpectedResourceType {
923                    resource_id: self.id.clone(),
924                    expected: Self::RESOURCE_TYPE,
925                    actual: new_config.get_resource_type(),
926                })
927            })?;
928
929        if self.id != new_sandbox.id {
930            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
931                resource_id: self.id.clone(),
932                reason: "the 'id' field is immutable".to_string(),
933            }));
934        }
935
936        Ok(())
937    }
938
939    fn as_any(&self) -> &dyn Any {
940        self
941    }
942
943    fn as_any_mut(&mut self) -> &mut dyn Any {
944        self
945    }
946
947    fn box_clone(&self) -> Box<dyn ResourceDefinition> {
948        Box::new(self.clone())
949    }
950
951    fn resource_eq(&self, other: &dyn ResourceDefinition) -> bool {
952        other.as_any().downcast_ref::<Sandbox>() == Some(self)
953    }
954
955    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
956        serde_json::to_value(self)
957    }
958}
959
960#[cfg(test)]
961mod tests {
962    use super::*;
963
964    fn sandbox_with(egress: SandboxEgress, preview_ports: Vec<u16>) -> Sandbox {
965        Sandbox::new("agent-sbx".to_string())
966            .code(SandboxCode::Image {
967                image: "ubuntu".to_string(),
968            })
969            .limits(SandboxLimits {
970                cpu: "1".to_string(),
971                memory: "2Gi".to_string(),
972                disk: "20Gi".to_string(),
973                max_processes: None,
974            })
975            .egress(egress)
976            .session(SandboxSessionPolicy {
977                max_lifetime_seconds: None,
978                idle_suspend_seconds: None,
979            })
980            .preview_ports(preview_ports)
981            .build()
982    }
983
984    #[test]
985    fn resource_type_is_stable() {
986        assert_eq!(Sandbox::RESOURCE_TYPE.as_ref(), "sandbox");
987    }
988
989    #[test]
990    fn capability_sets_are_per_platform() {
991        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
992        assert!(
993            gcp.reconnect,
994            "generation from the container boot id makes a session reachable across processes"
995        );
996        assert!(!gcp.preview);
997        assert!(gcp.enforced_limits);
998
999        let azure = SandboxCapabilities::for_platform(Platform::Azure).expect("azure is supported");
1000        assert!(azure.files, "every backend moves files");
1001        assert!(gcp.files);
1002        // Azure is the only backend whose egress policy matches on host pattern, and the only
1003        // one where `deny` and a hostname list are the same object.
1004        assert!(azure.domain_egress_rules);
1005        assert!(azure.egress_deny);
1006        // The data plane takes no ceiling, so a declaration of one is refused rather than
1007        // accepted and dropped.
1008        assert!(!azure.enforced_limits);
1009        assert!(azure.suspend_resume);
1010        // Both stay false for reasons that are not "unbuilt": a snapshot id has nothing to
1011        // consume it on any backend, and an Azure port's auth is anonymous or a human allowlist,
1012        // neither of which is a port-scoped credential.
1013        assert!(!azure.snapshot);
1014        assert!(!azure.preview);
1015
1016        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1017        assert!(!aws.snapshot, "AWS has no user-callable session snapshot");
1018        assert!(aws.suspend_resume);
1019
1020        let k8s =
1021            SandboxCapabilities::for_platform(Platform::Kubernetes).expect("k8s is supported");
1022        assert!(
1023            !k8s.preview,
1024            "the session-scoped ingress gateway does not exist yet"
1025        );
1026    }
1027
1028    /// Whether the process supervising a command is a separate identity from the command.
1029    ///
1030    /// Values are measured, not inferred. AWS: the agent runs as uid 0 with
1031    /// `CapEff: 00000000a80425fb` and `setuid`s the command to uid 60000, so the two differ.
1032    /// Kubernetes: the sandbox pod pins `run_as_user: 65534` on both pod and container with
1033    /// `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no uid split is
1034    /// possible (`kubernetes_spec.rs`). Local: `docker exec` runs as the workload uid while the
1035    /// manager supervises from the host. Azure and Agent Platform have no in-sandbox supervisor.
1036    #[test]
1037    fn supervisor_isolation_is_per_platform() {
1038        let value = |platform| {
1039            SandboxCapabilities::for_platform(platform)
1040                .expect("supported")
1041                .supervisor_isolation
1042        };
1043
1044        assert!(
1045            value(Platform::Aws),
1046            "root agent setuids the command to 60000"
1047        );
1048        assert!(
1049            value(Platform::Local),
1050            "the supervisor is on the host, outside the container"
1051        );
1052        assert!(
1053            !value(Platform::Kubernetes),
1054            "a single pinned uid cannot be split"
1055        );
1056        assert!(!value(Platform::Azure), "no Alien process runs the command");
1057        assert!(
1058            !value(Platform::Gcp),
1059            "no separate supervisor identity runs the command"
1060        );
1061    }
1062
1063    /// The point of the field: AWS and GCP report the *same* `supervisor_pid_namespace` (neither
1064    /// has `CAP_SYS_ADMIN`), so that axis alone reads them as equivalent. They are not — AWS
1065    /// separates the command's identity from the supervisor's and Agent Platform does not.
1066    #[test]
1067    fn supervisor_isolation_separates_aws_from_a_subprocess_backend() {
1068        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1069        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1070
1071        assert_eq!(
1072            aws.supervisor_pid_namespace, gcp.supervisor_pid_namespace,
1073            "the older axis cannot tell them apart"
1074        );
1075        assert!(
1076            aws.supervisor_isolation,
1077            "AWS setuids the command off the supervisor"
1078        );
1079        assert!(
1080            !gcp.supervisor_isolation,
1081            "the command runs under no separate supervisor identity"
1082        );
1083    }
1084
1085    /// The Agent Platform row, each value against the behaviour it was measured from. `reconnect`
1086    /// is the tripwire: it is `true` only because `generation` is derived from the container boot
1087    /// id read through the agent's health op, so a caller detects a replaced container instead of
1088    /// reconnecting to a blank one. It is also the body of the `Platform::Gcp` arm, asserted below.
1089    #[test]
1090    fn gcp_agent_platform_row_matches_measured_backend() {
1091        let row = SandboxCapabilities::gcp_agent_platform();
1092
1093        assert!(row.files, "agent file ops move over the session envelope");
1094        assert!(
1095            row.reconnect,
1096            "generation is derived from the container boot id, so a session is reachable across \
1097             processes"
1098        );
1099        assert!(
1100            !row.preview,
1101            "the only ingress is :execute; no port-scoped capability"
1102        );
1103        assert!(
1104            row.suspend_resume,
1105            ":pause and :resume preserve the container"
1106        );
1107        assert!(
1108            row.snapshot,
1109            "session state can be captured and restored into a new session"
1110        );
1111        assert!(
1112            !row.domain_egress_rules,
1113            "VPC and DNS peering is not a hostname allowlist"
1114        );
1115        assert!(
1116            row.egress_deny,
1117            "a declared deny blocks both egress and DNS"
1118        );
1119        assert!(
1120            row.enforced_limits,
1121            "ceilings are enforced, by terminating the session on breach"
1122        );
1123        assert!(!row.process_limit, "no process-count ceiling is observed");
1124        assert!(row.session_lifetime, "ttl maps to a session expireTime");
1125        assert!(!row.supervisor_pid_namespace, "no PID-namespace isolation");
1126        assert!(
1127            !row.supervisor_isolation,
1128            "the command is not run under a separate supervisor identity"
1129        );
1130
1131        // Agent Platform is the registered GCP backend, so the arm returns exactly this row.
1132        let live = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1133        assert_eq!(
1134            live, row,
1135            "the Platform::Gcp arm is the Agent Platform capability row"
1136        );
1137    }
1138
1139    #[test]
1140    fn platforms_without_a_backend_are_an_error_not_an_empty_set() {
1141        let error = SandboxCapabilities::for_platform(Platform::Machines)
1142            .expect_err("Machines has no sandbox backend");
1143        assert_eq!(error.code, "SANDBOX_PLATFORM_UNSUPPORTED");
1144    }
1145
1146    #[test]
1147    fn unsupported_capability_names_platform_and_capability() {
1148        let capabilities = SandboxCapabilities::for_platform(Platform::Gcp).expect("supported");
1149        let error = capabilities
1150            .require(SandboxCapability::Preview, Platform::Gcp)
1151            .expect_err("GCP has no preview");
1152
1153        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1154        let rendered = error.to_string();
1155        assert!(
1156            rendered.contains("preview"),
1157            "names the capability: {rendered}"
1158        );
1159        assert!(rendered.contains("gcp"), "names the platform: {rendered}");
1160    }
1161
1162    /// Azure matches on hostname; AWS and Kubernetes match CIDRs, and Local and GCP have a
1163    /// switch rather than a filter. Accepting a hostname list on those four would leave a stack
1164    /// reading as restricted while the sandbox reaches the whole internet.
1165    #[test]
1166    fn a_hostname_allowlist_is_refused_everywhere_it_would_be_approximated() {
1167        let sandbox = sandbox_with(
1168            SandboxEgress::AllowDomains {
1169                domains: vec!["example.com".to_string()],
1170            },
1171            vec![],
1172        );
1173
1174        for platform in [
1175            Platform::Aws,
1176            Platform::Gcp,
1177            Platform::Kubernetes,
1178            Platform::Local,
1179        ] {
1180            let error = sandbox
1181                .validate_for_platform(platform)
1182                .expect_err("only Azure expresses a hostname allowlist");
1183            assert_eq!(
1184                error.code, "SANDBOX_CAPABILITY_UNSUPPORTED",
1185                "on {platform:?}"
1186            );
1187        }
1188
1189        assert!(
1190            SandboxCapabilities::for_platform(Platform::Azure)
1191                .expect("supported")
1192                .domain_egress_rules,
1193            "Azure's egress policy matches on host pattern"
1194        );
1195    }
1196
1197    /// `deny` is the declaration that carries a security promise, so a backend that cannot keep
1198    /// it has to refuse rather than accept it and run the code with open egress.
1199    #[test]
1200    fn a_denied_egress_is_refused_where_it_would_not_be_enforced() {
1201        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1202
1203        // GCP is asserted at the capability rather than through validation: this sandbox declares
1204        // ceilings GCP cannot enforce, so it is refused for a reason unrelated to egress.
1205        assert!(
1206            SandboxCapabilities::for_platform(Platform::Gcp)
1207                .expect("supported")
1208                .egress_deny
1209        );
1210
1211        for platform in [Platform::Aws, Platform::Kubernetes, Platform::Local] {
1212            sandbox
1213                .validate_for_platform(platform)
1214                .expect("deny is enforced here");
1215        }
1216
1217        // Declares no ceilings, which Azure refuses for its own reason, so this isolates egress.
1218        let egress_only = Sandbox::new("sbx".to_string())
1219            .code(SandboxCode::Image {
1220                image: "alpine".to_string(),
1221            })
1222            .egress(SandboxEgress::Deny)
1223            .session(SandboxSessionPolicy {
1224                max_lifetime_seconds: None,
1225                idle_suspend_seconds: None,
1226            })
1227            .build();
1228
1229        egress_only
1230            .validate_for_platform(Platform::Azure)
1231            .expect("Azure creates the sandbox under a Deny policy with full inspection");
1232    }
1233
1234    /// Ceilings are rejected per-platform where unsupported — rejected when *declared*. With
1235    /// limits mandatory that would read as "GCP sandboxes cannot exist", contradicting the
1236    /// create, exec, files and terminate GCP does support.
1237    #[test]
1238    fn a_platform_that_cannot_enforce_limits_still_takes_a_sandbox_without_them() {
1239        let declared = sandbox_with(SandboxEgress::Deny, Vec::new());
1240        declared
1241            .validate_for_platform(Platform::Azure)
1242            .expect_err("declaring ceilings Azure cannot enforce is rejected");
1243
1244        let undeclared = Sandbox::new("sbx".to_string())
1245            .code(SandboxCode::Image {
1246                image: "alpine".to_string(),
1247            })
1248            .egress(SandboxEgress::Deny)
1249            .session(SandboxSessionPolicy {
1250                max_lifetime_seconds: None,
1251                idle_suspend_seconds: None,
1252            })
1253            .build();
1254
1255        undeclared
1256            .validate_for_platform(Platform::Azure)
1257            .expect("a sandbox naming no ceilings takes the platform's own");
1258
1259        // A backend still gets a concrete set, so nothing downstream has to invent one.
1260        assert_eq!(undeclared.resolved_limits().cpu, "1");
1261    }
1262
1263    #[test]
1264    fn preview_ports_require_the_preview_capability() {
1265        let sandbox = sandbox_with(SandboxEgress::Deny, vec![8080]);
1266
1267        sandbox
1268            .validate_for_platform(Platform::Aws)
1269            .expect("AWS mints a port-scoped JWE");
1270
1271        let error = sandbox
1272            .validate_for_platform(Platform::Kubernetes)
1273            .expect_err("Kubernetes preview is deferred");
1274        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1275    }
1276
1277    #[test]
1278    fn gcp_accepts_a_sandbox_declaring_enforced_limits() {
1279        let sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1280        sandbox
1281            .validate_for_platform(Platform::Gcp)
1282            .expect("Agent Platform enforces declared ceilings, by terminating on breach");
1283    }
1284
1285    #[test]
1286    fn invalid_quantities_are_rejected_with_the_offending_field() {
1287        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1288        sandbox
1289            .limits
1290            .as_mut()
1291            .expect("the fixture declares limits")
1292            .memory = "2Gb".to_string();
1293
1294        let error = sandbox
1295            .validate_for_platform(Platform::Aws)
1296            .expect_err("Gb is not a valid suffix");
1297        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1298        assert!(error.to_string().contains("memory"));
1299
1300        sandbox
1301            .limits
1302            .as_mut()
1303            .expect("the fixture declares limits")
1304            .memory = "2Gi".to_string();
1305        sandbox
1306            .limits
1307            .as_mut()
1308            .expect("the fixture declares limits")
1309            .cpu = "0".to_string();
1310        let error = sandbox
1311            .validate_for_platform(Platform::Aws)
1312            .expect_err("zero cpu is not a ceiling");
1313        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1314    }
1315
1316    #[test]
1317    fn zero_max_processes_is_rejected() {
1318        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1319        sandbox
1320            .limits
1321            .as_mut()
1322            .expect("the fixture declares limits")
1323            .max_processes = Some(0);
1324
1325        let error = sandbox
1326            .validate_for_platform(Platform::Local)
1327            .expect_err("a sandbox must be able to run at least one process");
1328        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1329        assert!(error.to_string().contains("maxProcesses"));
1330    }
1331
1332    /// A process ceiling needs a container runtime. Kubernetes sets one per node rather than per
1333    /// pod, and neither MicroVMs nor Azure sandboxes expose one, so accepting the declaration
1334    /// anywhere else would mean carrying a bound nothing applies.
1335    #[test]
1336    fn a_process_ceiling_is_accepted_only_where_a_runtime_can_apply_it() {
1337        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1338        sandbox
1339            .limits
1340            .as_mut()
1341            .expect("the fixture declares limits")
1342            .max_processes = Some(256);
1343
1344        sandbox
1345            .validate_for_platform(Platform::Local)
1346            .expect("Docker takes a pids limit");
1347
1348        for platform in [Platform::Aws, Platform::Azure, Platform::Kubernetes] {
1349            let error = sandbox
1350                .validate_for_platform(platform)
1351                .expect_err("a process ceiling nothing applies must be refused");
1352            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1353        }
1354    }
1355
1356    /// Lambda rejects a run outside 1–28,800 rather than clamping it, so a value beyond that
1357    /// would pass planning, render into the package, and fail at the first session. Kubernetes
1358    /// takes the same field with no such bound, so the check is AWS's alone.
1359    #[test]
1360    fn a_lifetime_aws_would_reject_is_refused_while_planning() {
1361        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1362
1363        for seconds in [0, 28_801, 100_000] {
1364            sandbox.session.max_lifetime_seconds = Some(seconds);
1365            let error = sandbox
1366                .validate_for_platform(Platform::Aws)
1367                .expect_err("a lifetime outside what AWS runs is refused");
1368            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "{seconds}s");
1369
1370            // Kubernetes has no such ceiling, so the same declaration is fine there.
1371            sandbox
1372                .validate_for_platform(Platform::Kubernetes)
1373                .expect("the kubelet takes any activeDeadlineSeconds");
1374        }
1375
1376        sandbox.session.max_lifetime_seconds = Some(28_800);
1377        sandbox
1378            .validate_for_platform(Platform::Aws)
1379            .expect("the ceiling itself is allowed");
1380    }
1381
1382    /// An image reference Azure cannot honour is refused while planning, not at the first session.
1383    ///
1384    /// `code.image`'s own documentation gives a tag and a registry path as examples — exactly
1385    /// what Azure cannot take, so this is the shape a customer is most likely to declare.
1386    #[test]
1387    fn an_image_azure_cannot_pull_is_refused_while_planning() {
1388        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1389        // Azure enforces no declared ceiling, so a sandbox carrying limits is refused before the
1390        // image is ever read.
1391        sandbox.limits = None;
1392
1393        for image in [
1394            "ubuntu:24.04",
1395            "ghcr.io/myorg/sandbox:latest",
1396            "ubuntu@sha256:abc",
1397            "",
1398            "   ",
1399            "ubuntu latest",
1400            "ubuntu?x",
1401        ] {
1402            sandbox.code = SandboxCode::Image {
1403                image: image.to_string(),
1404            };
1405            let error = sandbox
1406                .validate_for_platform(Platform::Azure)
1407                .expect_err("an image Azure has nowhere to put is refused");
1408            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "image '{image}'");
1409
1410            // The same declaration is ordinary everywhere that pulls a reference.
1411            sandbox
1412                .validate_for_platform(Platform::Kubernetes)
1413                .expect("a registry reference is what every other backend takes");
1414        }
1415
1416        for image in ["ubuntu", "ubuntu-22.04", "debian_slim"] {
1417            sandbox.code = SandboxCode::Image {
1418                image: image.to_string(),
1419            };
1420            sandbox
1421                .validate_for_platform(Platform::Azure)
1422                .unwrap_or_else(|error| panic!("'{image}' is a catalog name: {error}"));
1423        }
1424
1425        // Surrounding space is trimmed rather than carried into the create body.
1426        sandbox.code = SandboxCode::Image {
1427            image: " ubuntu ".to_string(),
1428        };
1429        assert_eq!(
1430            sandbox
1431                .azure_catalog_image()
1432                .expect("a padded name is still a name"),
1433            "ubuntu"
1434        );
1435    }
1436
1437    /// A deadline is accepted only where the platform itself terminates on it — the kubelet's
1438    /// `activeDeadlineSeconds` and Lambda's `maximumDurationInSeconds`. Everywhere else it would
1439    /// need a reaper that does not exist, so it is refused rather than accepted and dropped.
1440    #[test]
1441    fn a_session_deadline_is_accepted_only_where_the_platform_applies_it() {
1442        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1443        sandbox.session.max_lifetime_seconds = Some(3600);
1444
1445        sandbox
1446            .validate_for_platform(Platform::Kubernetes)
1447            .expect("the kubelet enforces activeDeadlineSeconds");
1448        sandbox
1449            .validate_for_platform(Platform::Aws)
1450            .expect("Lambda terminates the MicroVM at maximumDurationInSeconds");
1451
1452        for platform in [Platform::Azure, Platform::Local] {
1453            let error = sandbox
1454                .validate_for_platform(platform)
1455                .expect_err("a deadline nothing applies must be refused");
1456            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1457        }
1458    }
1459
1460    /// A MicroVM bursts to four times its baseline with no way to opt out, so a ceiling is kept
1461    /// by choosing the size whose *peak* fits inside it. Sizing by baseline would hand back a
1462    /// sandbox that can reach four times what the customer declared.
1463    #[test]
1464    fn an_aws_size_is_chosen_so_its_peak_stays_inside_the_declared_ceiling() {
1465        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1466        let tier = sandbox
1467            .microvm_tier()
1468            .expect("2Gi/1cpu/20Gi is satisfiable");
1469
1470        assert_eq!(
1471            tier.peak_memory_mib, 2048,
1472            "the peak is the declared ceiling"
1473        );
1474        assert_eq!(
1475            tier.baseline_memory_mib, 512,
1476            "which is a quarter of it as the baseline"
1477        );
1478        assert!(tier.max_disk_mib <= 20 * 1024);
1479    }
1480
1481    /// AWS allocates one vCPU per 2GB, so a cpu ceiling below what the memory ceiling implies
1482    /// cannot be honoured together with it. Letting cpu choose the size instead would hand back a
1483    /// machine four times smaller than the memory asked for, with nothing to indicate it.
1484    #[test]
1485    fn a_cpu_ceiling_below_what_the_memory_implies_is_refused_not_quietly_downsized() {
1486        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1487        {
1488            let limits = sandbox
1489                .limits
1490                .as_mut()
1491                .expect("the fixture declares limits");
1492            limits.cpu = "1".to_string();
1493            limits.memory = "8Gi".to_string();
1494        }
1495
1496        let error = sandbox
1497            .microvm_tier()
1498            .expect_err("1 cpu and 8Gi cannot both be ceilings on AWS");
1499        assert!(
1500            error.to_string().contains("4 vCPU"),
1501            "the refusal must say what the memory ceiling implies: {error}"
1502        );
1503
1504        sandbox
1505            .limits
1506            .as_mut()
1507            .expect("the fixture declares limits")
1508            .cpu = "4".to_string();
1509        let tier = sandbox.microvm_tier().expect("4 cpu matches 8Gi");
1510        assert_eq!(tier.peak_memory_mib, 8192);
1511    }
1512
1513    /// Below AWS's smallest peak there is no size that holds the ceiling, and rounding up to the
1514    /// nearest one would silently exceed it.
1515    #[test]
1516    fn an_aws_ceiling_smaller_than_any_size_is_refused_rather_than_rounded() {
1517        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1518        sandbox
1519            .limits
1520            .as_mut()
1521            .expect("the fixture declares limits")
1522            .memory = "1Gi".to_string();
1523
1524        let error = sandbox
1525            .validate_for_platform(Platform::Aws)
1526            .expect_err("no MicroVM size peaks at or below 1Gi");
1527        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1528        assert!(
1529            error.to_string().contains("2Gi"),
1530            "the refusal must say what the smallest holdable ceiling is: {error}"
1531        );
1532    }
1533
1534    /// `Source` is a public part of the type that no backend builds: an empty image string
1535    /// schedules a pod that can never run, so the refusal has to happen at plan time and on
1536    /// every platform, not in one emitter.
1537    #[test]
1538    fn source_code_is_refused_everywhere_rather_than_producing_a_broken_manifest() {
1539        let sandbox = Sandbox::new("agent".to_string())
1540            .code(SandboxCode::Source {
1541                src: "./sandbox".to_string(),
1542                toolchain: ToolchainConfig::Docker {
1543                    dockerfile: None,
1544                    build_args: None,
1545                    target: None,
1546                },
1547            })
1548            .egress(SandboxEgress::Deny)
1549            .session(SandboxSessionPolicy {
1550                max_lifetime_seconds: None,
1551                idle_suspend_seconds: None,
1552            })
1553            .build();
1554
1555        for platform in [
1556            Platform::Aws,
1557            Platform::Azure,
1558            Platform::Gcp,
1559            Platform::Kubernetes,
1560            Platform::Local,
1561        ] {
1562            let error = sandbox
1563                .validate_for_platform(platform)
1564                .expect_err("no backend builds a sandbox image from source");
1565            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1566            assert!(
1567                error.to_string().contains("code.image"),
1568                "the refusal must say what to write instead: {error}"
1569            );
1570        }
1571    }
1572
1573    /// `validate_quantity` accepts nine suffixes. Reading only `Gi` and `Mi` would size a
1574    /// declared `4G` as though it were `4Gi`, which for a ceiling means exceeding it.
1575    #[test]
1576    fn every_accepted_unit_converts_rather_than_falling_back() {
1577        assert_eq!(quantity_mib("2Gi"), Some(2048));
1578        assert_eq!(quantity_mib("512Mi"), Some(512));
1579        assert_eq!(quantity_mib("4G"), Some(3814));
1580        assert_eq!(quantity_mib("1Ti"), Some(1024 * 1024));
1581        assert_eq!(millicores("1"), Some(1000));
1582        assert_eq!(millicores("500m"), Some(500));
1583    }
1584
1585    #[test]
1586    fn unknown_fields_are_rejected() {
1587        let json = r#"{
1588            "id": "sbx",
1589            "code": {"type": "image", "image": "ubuntu:24.04"},
1590            "limits": {"cpu": "1", "memory": "2Gi", "disk": "20Gi"},
1591            "egress": {"mode": "deny"},
1592            "session": {},
1593            "unexpected": true
1594        }"#;
1595
1596        serde_json::from_str::<Sandbox>(json).expect_err("deny_unknown_fields must reject");
1597    }
1598
1599    #[test]
1600    fn serialization_roundtrips() {
1601        let sandbox = sandbox_with(
1602            SandboxEgress::AllowDomains {
1603                domains: vec!["example.com".to_string()],
1604            },
1605            vec![8080, 9090],
1606        );
1607
1608        let json = serde_json::to_string(&sandbox).expect("serializes");
1609        let restored: Sandbox = serde_json::from_str(&json).expect("deserializes");
1610        assert_eq!(sandbox, restored);
1611    }
1612
1613    #[test]
1614    fn id_is_immutable_across_updates() {
1615        let original = sandbox_with(SandboxEgress::Deny, vec![]);
1616        let renamed = Sandbox::new("other".to_string())
1617            .code(SandboxCode::Image {
1618                image: "ubuntu".to_string(),
1619            })
1620            .limits(
1621                original
1622                    .limits
1623                    .clone()
1624                    .expect("the fixture declares limits"),
1625            )
1626            .egress(SandboxEgress::Deny)
1627            .session(SandboxSessionPolicy {
1628                max_lifetime_seconds: None,
1629                idle_suspend_seconds: None,
1630            })
1631            .build();
1632
1633        original
1634            .validate_update(&original.clone())
1635            .expect("an unchanged config is a valid update");
1636        original
1637            .validate_update(&renamed)
1638            .expect_err("renaming a sandbox is not an update");
1639    }
1640
1641    /// Azure declares an idle-suspend policy but not a wall-clock ceiling.
1642    ///
1643    /// The two travel together in `SandboxSessionPolicy` and are gated separately on purpose:
1644    /// Azure suspends on idle and has no maximum lifetime, so accepting one and refusing the
1645    /// other is the honest split rather than an inconsistency.
1646    #[test]
1647    fn azure_takes_an_idle_policy_and_still_refuses_a_lifetime_ceiling() {
1648        let with_policy = |session: SandboxSessionPolicy| {
1649            Sandbox::new("sbx".to_string())
1650                .code(SandboxCode::Image {
1651                    image: "ubuntu".to_string(),
1652                })
1653                .egress(SandboxEgress::Allow)
1654                .session(session)
1655                .build()
1656                .validate_for_platform(Platform::Azure)
1657        };
1658
1659        with_policy(SandboxSessionPolicy {
1660            max_lifetime_seconds: None,
1661            idle_suspend_seconds: Some(900),
1662        })
1663        .expect("Azure suspends a session on idle");
1664
1665        let error = with_policy(SandboxSessionPolicy {
1666            max_lifetime_seconds: Some(3600),
1667            idle_suspend_seconds: None,
1668        })
1669        .expect_err("Azure has no wall-clock ceiling to enforce one with");
1670        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1671        assert!(
1672            error.message.contains("sessionLifetime"),
1673            "names the capability: {}",
1674            error.message
1675        );
1676    }
1677
1678    /// An allowlist naming nothing is a deny-all wearing an allowlist's label.
1679    ///
1680    /// It renders as a `Deny` default with no rules — the shape the Azure provider adds a
1681    /// catch-all to avoid — and a reader scanning the declaration sees "allowDomains" and reads
1682    /// the opposite of what it does.
1683    #[test]
1684    fn an_allowlist_with_no_domains_is_refused() {
1685        let declared = |domains: Vec<String>| {
1686            Sandbox::new("sbx".to_string())
1687                .code(SandboxCode::Image {
1688                    image: "ubuntu".to_string(),
1689                })
1690                .egress(SandboxEgress::AllowDomains { domains })
1691                .session(SandboxSessionPolicy {
1692                    max_lifetime_seconds: None,
1693                    idle_suspend_seconds: None,
1694                })
1695                .build()
1696                .validate_for_platform(Platform::Azure)
1697        };
1698
1699        let error = declared(vec![]).expect_err("an empty allowlist must be refused");
1700        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1701
1702        declared(vec!["api.example.com".to_string()])
1703            .expect("a named domain is what an allowlist is for");
1704    }
1705
1706    /// The two expressible modes map to the boolean; a host list maps to nothing so the caller has
1707    /// to refuse rather than silently pick a side.
1708    #[test]
1709    fn internet_access_switch_maps_only_the_two_expressible_modes() {
1710        assert_eq!(SandboxEgress::Allow.internet_access_switch(), Some(true));
1711        assert_eq!(SandboxEgress::Deny.internet_access_switch(), Some(false));
1712        assert_eq!(
1713            SandboxEgress::AllowDomains {
1714                domains: vec!["api.example.com".to_string()]
1715            }
1716            .internet_access_switch(),
1717            None,
1718            "a host list has no boolean and must not be approximated"
1719        );
1720    }
1721}