Skip to main content

alien_core/resources/
sandbox.rs

1//! Sandbox resource for running untrusted code in an isolated environment.
2//!
3//! The declaration provisions a durable parent, and the application creates and destroys
4//! individual sandboxes through its binding at runtime.
5//!
6//! The capability set differs per platform and is published rather than assumed. Calling an
7//! unsupported capability is a typed error naming both the platform and the capability, so a
8//! portable application can branch on `SandboxCapabilities` before it calls.
9
10use crate::error::{ErrorData, Result};
11use crate::resource::{ResourceDefinition, ResourceOutputsDefinition, ResourceRef, ResourceType};
12use crate::resources::ToolchainConfig;
13use crate::Platform;
14use alien_error::AlienError;
15use bon::Builder;
16use serde::{Deserialize, Serialize};
17use sha2::{Digest, Sha256};
18use std::any::Any;
19use std::fmt::Debug;
20
21/// Specifies where the sandbox's root filesystem comes from.
22#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
23#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
24#[serde(rename_all = "camelCase", tag = "type")]
25pub enum SandboxCode {
26    /// A prebuilt container image used as the sandbox root filesystem.
27    #[serde(rename_all = "camelCase")]
28    Image {
29        /// Image reference (e.g. `ubuntu:24.04`, `ghcr.io/myorg/sandbox:latest`). AWS wants an
30        /// `s3://` bundle; Azure takes a catalog name such as `ubuntu` or an amd64 registry
31        /// image, told apart by syntax. Each refuses what it cannot take while planning.
32        image: String,
33    },
34    /// A Dockerfile `alien build` builds into the sandbox's base image.
35    ///
36    /// AWS only, and docker only: the base image is a root filesystem, not a binary laid on one.
37    /// `alien release` pushes it and the bundle layers the sandbox agent on afterwards.
38    #[serde(rename_all = "camelCase")]
39    Source {
40        /// The source directory to build from
41        src: String,
42        /// Toolchain configuration with type-safe options
43        toolchain: ToolchainConfig,
44    },
45}
46
47/// Hard ceilings enforced on a sandbox.
48///
49/// These are limits, not scheduling requests. Untrusted code does not respect a hint, so every
50/// field is enforced by the platform and a platform that cannot enforce one is rejected at plan
51/// time rather than silently ignoring it.
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
54#[serde(rename_all = "camelCase", deny_unknown_fields)]
55pub struct SandboxLimits {
56    /// CPU ceiling in cores or millicores (e.g. `"1"`, `"500m"`)
57    pub cpu: String,
58    /// Memory ceiling (e.g. `"2Gi"`, `"512Mi"`)
59    pub memory: String,
60    /// Disk ceiling (e.g. `"20Gi"`)
61    pub disk: String,
62    /// Maximum number of processes, which bounds fork bombs.
63    ///
64    /// Optional because only a container runtime has the primitive: Kubernetes sets a pid ceiling
65    /// per node, not per pod, and neither AWS MicroVMs nor Azure sandboxes expose one. Declaring
66    /// it on a platform that cannot apply it is refused at plan time.
67    #[serde(default, skip_serializing_if = "Option::is_none")]
68    pub max_processes: Option<u32>,
69}
70
71/// One of the five sizes a Lambda MicroVM can be built at.
72///
73/// AWS has no ceiling knob: `minimumMemoryInMiB` sets a *baseline* and a running MicroVM bursts
74/// vertically to four times it with no way to opt out. A declared ceiling is therefore honoured by
75/// picking the tier whose **peak** stays inside it, not the tier whose baseline matches it.
76#[derive(Debug, Clone, Copy, PartialEq, Eq)]
77pub struct MicrovmTier {
78    /// What `minimumMemoryInMiB` is set to.
79    pub baseline_memory_mib: i64,
80    /// The most memory the MicroVM can reach, in MiB.
81    pub peak_memory_mib: i64,
82    /// The most vCPU the MicroVM can reach.
83    pub peak_vcpu: u32,
84    /// The most disk the MicroVM can use, in MiB.
85    pub max_disk_mib: i64,
86}
87
88/// The published sizes, smallest first. Baseline memory to vCPU is 2 GB per vCPU, peak is four
89/// times baseline, and disk is fixed per tier rather than independently selectable.
90/// Longest life AWS will run a MicroVM for, from `RunMicrovm`'s `maximumDurationInSeconds`.
91const AWS_MAX_LIFETIME_SECONDS: u32 = 28_800;
92
93/// Azure's sandbox sizing rule, quoted from the data plane's own refusal of an oversized request:
94/// *CPU must be n×250m for n=1..64 (0.25–16 cores); Memory ≤ cores × 2Gi; Disk ≤ cores × 20Gi*.
95const AZURE_CPU_STEP_MILLICORES: i64 = 250;
96const AZURE_MAX_CPU_MILLICORES: i64 = 16_000;
97const AZURE_MEMORY_MIB_PER_CORE: i64 = 2 * 1024;
98const AZURE_DISK_MIB_PER_CORE: i64 = 20 * 1024;
99
100const MICROVM_TIERS: &[MicrovmTier] = &[
101    MicrovmTier {
102        baseline_memory_mib: 512,
103        peak_memory_mib: 2048,
104        peak_vcpu: 1,
105        max_disk_mib: 8192,
106    },
107    MicrovmTier {
108        baseline_memory_mib: 1024,
109        peak_memory_mib: 4096,
110        peak_vcpu: 2,
111        max_disk_mib: 8192,
112    },
113    MicrovmTier {
114        baseline_memory_mib: 2048,
115        peak_memory_mib: 8192,
116        peak_vcpu: 4,
117        max_disk_mib: 8192,
118    },
119    MicrovmTier {
120        baseline_memory_mib: 4096,
121        peak_memory_mib: 16384,
122        peak_vcpu: 8,
123        max_disk_mib: 16384,
124    },
125    MicrovmTier {
126        baseline_memory_mib: 8192,
127        peak_memory_mib: 32768,
128        peak_vcpu: 16,
129        max_disk_mib: 32768,
130    },
131];
132
133/// Outbound network policy for a sandbox.
134#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
135#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
136#[serde(rename_all = "camelCase", tag = "mode")]
137pub enum SandboxEgress {
138    /// No outbound network access.
139    ///
140    /// Routed traffic only. Link-local is not outbound and no backend's egress control reaches
141    /// it, so this is not a boundary against instance metadata.
142    ///
143    /// Nor, on AWS, against DNS: while the connector VPC has DNS support on, a session resolves
144    /// names through that VPC's resolver, which no security group filters, so a query name can
145    /// carry data out.
146    Deny,
147    /// Unrestricted outbound access to the public internet, and none to private ranges or the
148    /// deployment's own network.
149    ///
150    /// Link-local carries the same exception as `Deny`. AWS and Kubernetes deliver both halves.
151    /// Azure and GCP deliver the first only: one matches host patterns and the other is a single
152    /// switch, so neither can name an address range to exclude.
153    Allow,
154    /// Outbound access only to the listed hostnames.
155    ///
156    /// Azure alone expresses it: its egress proxy matches on host pattern. The others filter by
157    /// CIDR or carry a single switch, and both would approximate the list rather than keep it.
158    #[serde(rename_all = "camelCase")]
159    AllowDomains {
160        /// Hostnames the sandbox may reach
161        domains: Vec<String>,
162    },
163}
164
165impl SandboxEgress {
166    /// The single outbound switch for a backend that has no host matcher, or `None` for a mode a
167    /// boolean cannot carry.
168    ///
169    /// `AllowDomains` needs a host list, so it maps to nothing and each caller refuses it in its
170    /// own error naming the sandbox. One source for what a mode means, so a template and a sandbox
171    /// cannot disagree on it.
172    pub fn internet_access_switch(&self) -> Option<bool> {
173        match self {
174            SandboxEgress::Allow => Some(true),
175            SandboxEgress::Deny => Some(false),
176            SandboxEgress::AllowDomains { .. } => None,
177        }
178    }
179}
180
181/// How long a sandbox may live and when it is paused.
182///
183/// Declaration-time ceilings, not per-request values: every sandbox created through this
184/// declaration's binding is held to them, whatever a caller asks for at runtime.
185#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
186#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
187#[serde(rename_all = "camelCase", deny_unknown_fields)]
188pub struct SandboxLifecyclePolicy {
189    /// Wall-clock ceiling on a single sandbox, after which the platform terminates it.
190    ///
191    /// Optional because not every backend has the primitive: Kubernetes has
192    /// `activeDeadlineSeconds` and AWS `maximumDurationInSeconds`, while neither Azure nor Local
193    /// expose one, so declaring a ceiling there is refused at plan time rather than accepted and
194    /// never applied. AWS caps it at 8 hours.
195    #[serde(default, skip_serializing_if = "Option::is_none")]
196    pub max_lifetime_seconds: Option<u32>,
197    /// Idle period after which the sandbox is paused, where the platform supports it
198    #[serde(skip_serializing_if = "Option::is_none")]
199    pub idle_pause_seconds: Option<u32>,
200}
201
202/// What a platform's sandbox backend can actually do.
203///
204/// Published so portable code can branch before calling rather than discovering a gap through
205/// an error. Every field here corresponds to a capability that at least one platform lacks;
206/// create, exec and terminate are the guaranteed floor and are therefore not listed.
207#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
208#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
209#[serde(rename_all = "camelCase", deny_unknown_fields)]
210pub struct SandboxCapabilities {
211    /// Files can be moved in and out of a sandbox
212    pub files: bool,
213    /// A later call can reach a sandbox created by an earlier one
214    pub reconnect: bool,
215    /// A command can be started, polled and cancelled across separate calls, so it outlives the
216    /// one that started it. False where nothing inside the sandbox owns the process in between.
217    pub jobs: bool,
218    /// An authenticated, port-scoped capability to reach a service inside the sandbox
219    pub preview: bool,
220    /// Sandbox state can be paused and resumed
221    pub pause_resume: bool,
222    /// A sandbox's full state can be captured and used to create another
223    pub snapshot: bool,
224    /// Egress can be restricted to a hostname allowlist
225    pub domain_egress_rules: bool,
226    /// Whether a declared `deny` is actually enforced, rather than accepted and dropped.
227    /// It covers routed traffic; the `deny` mode says where DNS still resolves.
228    pub egress_deny: bool,
229    /// The platform enforces the declared cpu, memory and disk ceilings
230    pub enforced_limits: bool,
231    /// The platform can cap how many processes a sandbox runs
232    pub process_limit: bool,
233    /// The platform terminates a sandbox at a declared wall-clock deadline
234    pub sandbox_lifetime: bool,
235    /// A command runs in its own PID namespace and cannot see or signal the agent's processes.
236    ///
237    /// Only where an agent runs as root. Creating the namespace needs `CAP_SYS_ADMIN`, and the
238    /// Kubernetes sandbox pod drops every capability — which is also what denies `ptrace` by
239    /// construction, so granting it there would remove a lock to add one.
240    pub supervisor_pid_namespace: bool,
241    /// The process supervising a command is a different identity from the command.
242    ///
243    /// False where a command runs as the agent's own user: it can then read the supervisor's
244    /// environment and signal it. Separate from `supervisorPidNamespace`, which is about
245    /// visibility rather than identity — a backend can have one without the other.
246    pub supervisor_isolation: bool,
247}
248
249impl SandboxCapabilities {
250    /// Returns what the given platform's sandbox backend supports.
251    ///
252    /// Errors for platforms with no sandbox backend, rather than returning an all-false set —
253    /// "every capability is missing" and "this platform has no sandboxes" are different
254    /// conditions and an application should not have to tell them apart by inspection.
255    pub fn for_platform(platform: Platform) -> Result<Self> {
256        match platform {
257            Platform::Aws => Ok(Self {
258                files: true,
259                reconnect: true,
260                jobs: true,
261                preview: true,
262                pause_resume: true,
263                snapshot: false,
264                domain_egress_rules: false,
265                // Routed egress only: while the connector VPC has DNS support on, a session
266                // resolves names through its resolver, which no security group filters.
267                egress_deny: true,
268                enforced_limits: true,
269                // Nothing in the API bounds process count.
270                process_limit: false,
271                // `maximumDurationInSeconds` on `RunMicrovm`, which Lambda enforces by
272                // terminating the MicroVM. Capped at 8 hours by the service.
273                sandbox_lifetime: true,
274                // Measured, not assumed: the agent inside a Lambda MicroVM runs as uid 0 with
275                // `CapEff: 00000000a80425fb`, the standard container default set, which excludes
276                // `CAP_SYS_ADMIN`. It can drop privilege (`CAP_SETUID`/`CAP_SETGID` are held) and
277                // it cannot create a namespace. No backend offers this today.
278                supervisor_pid_namespace: false,
279                // The agent runs as uid 0 and `setuid`s the command to uid 60000, so the command
280                // runs under a different identity than the process supervising it.
281                supervisor_isolation: true,
282            }),
283            Platform::Azure => Ok(Self::azure()),
284            Platform::Gcp => Ok(Self::gcp_agent_platform()),
285            // Preview needs a gateway that validates a sandbox-and-port capability, which this
286            // backend has none of.
287            Platform::Kubernetes => Ok(Self {
288                files: true,
289                reconnect: true,
290                jobs: true,
291                preview: false,
292                pause_resume: false,
293                snapshot: false,
294                domain_egress_rules: false,
295                egress_deny: true,
296                enforced_limits: true,
297                // A pid ceiling is a kubelet setting per node, not a pod field.
298                process_limit: false,
299                // `activeDeadlineSeconds` on the pod, which the kubelet enforces.
300                sandbox_lifetime: true,
301                // The pod drops every capability, including the `CAP_SYS_ADMIN` the agent would
302                // need to unshare. That is also what denies `ptrace`, so this stays false rather
303                // than the pod being weakened to make it true.
304                supervisor_pid_namespace: false,
305                // The pod pins one uid (`run_as_user: 65534` on both pod and container) with
306                // `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no
307                // process can setuid to split the command off from a supervisor. No uid split is
308                // possible, so none exists.
309                supervisor_isolation: false,
310            }),
311            Platform::Local => Ok(Self {
312                files: true,
313                reconnect: true,
314                // Nothing runs inside the sandbox: the manager drives Docker from outside it.
315                jobs: false,
316                preview: true,
317                pause_resume: false,
318                snapshot: false,
319                domain_egress_rules: false,
320                egress_deny: true,
321                enforced_limits: true,
322                // Docker's `--pids-limit`.
323                process_limit: true,
324                sandbox_lifetime: false,
325                // Local has no in-sandbox agent: the manager drives Docker from outside, so
326                // there is no supervisor inside the sandbox to isolate from.
327                supervisor_pid_namespace: false,
328                // The supervisor is the manager on the host, outside the container entirely, and
329                // `docker exec` runs the command as the workload uid — a different identity by
330                // construction.
331                supervisor_isolation: true,
332            }),
333            Platform::Machines | Platform::Test => {
334                Err(AlienError::new(ErrorData::SandboxPlatformUnsupported {
335                    platform: platform.to_string(),
336                }))
337            }
338        }
339    }
340
341    /// What the Azure sandbox backend supports; the body of the `Platform::Azure` arm.
342    pub fn azure() -> Self {
343        Self {
344            files: true,
345            reconnect: true,
346            // No Alien process runs inside the sandbox to own a command between two calls.
347            jobs: false,
348            // A sandbox port carries a URL and an auth config, and the auth config offers two
349            // things: anonymous, or Entra ID with an allowlist of human email addresses.
350            // Neither is a credential scoped to a port for a fixed time, which is what a
351            // preview capability is. Returning the anonymous URL would publish the port.
352            preview: false,
353            pause_resume: true,
354            // False for a client reason, not a cloud one: this client has no snapshot call, and
355            // `CreateSandboxRequest` has no field to consume the id it would return. Also
356            // unclaimed: Microsoft does not garbage-collect snapshots, so an id is a bill that grows.
357            snapshot: false,
358            domain_egress_rules: true,
359            egress_deny: true,
360            // Enforced inside the sandbox, not at create: an over-allocation raises `MemoryError`
361            // while the sandbox keeps running. `azure_sandbox_limits` checks the continuous
362            // sizing rule at plan time instead of matching a tier.
363            enforced_limits: true,
364            process_limit: false,
365            // Auto-suspend and auto-delete exist; a wall-clock ceiling does not. Accepting
366            // `maxLifetimeSeconds` here would be the silent no-op the capability set exists
367            // to prevent, so this is a decision rather than a gap.
368            sandbox_lifetime: false,
369            // No Alien process inside an Azure sandbox, so there is no supervisor to isolate.
370            supervisor_pid_namespace: false,
371            // No Alien process runs the command at all — the platform's own data plane does,
372            // so there is no separate supervisor identity to speak of.
373            supervisor_isolation: false,
374        }
375    }
376
377    /// What the GCP Agent Platform sandbox backend supports; the body of the `Platform::Gcp` arm.
378    pub fn gcp_agent_platform() -> Self {
379        Self {
380            // Agent file operations move over the sandbox envelope.
381            files: true,
382            // Reaching a sandbox across processes is safe because `generation` is derived from the
383            // container boot id read through the agent's health op, so a caller detects a container
384            // replaced under a stable sandbox name rather than reconnecting to a blank one.
385            reconnect: true,
386            jobs: true,
387            // No method mints a port-scoped ingress capability; the only ingress is `:execute`.
388            preview: false,
389            // `:resume` can return a fresh container while reporting success, so a pause does not
390            // keep the sandbox's state.
391            pause_resume: false,
392            // The create path never sends `sandbox_environment_snapshot`, so no sandbox state is
393            // reachable through the trait; declared false until the client carries it.
394            snapshot: false,
395            // Egress is shaped by VPC and DNS peering, which is not a hostname allowlist.
396            domain_egress_rules: false,
397            // A declared `deny` blocks both routed egress and DNS.
398            egress_deny: true,
399            // The declared ceilings are enforced, but by terminating the sandbox on breach rather
400            // than by refusing the allocation — a caller reading `true` should expect the sandbox
401            // to die, not a clean error at the point of the request.
402            enforced_limits: true,
403            // No ceiling on process count is observed.
404            process_limit: false,
405            // `ttl` maps to a sandbox `expireTime` the platform terminates at.
406            sandbox_lifetime: true,
407            // No PID-namespace isolation between the command and anything supervising it.
408            supervisor_pid_namespace: false,
409            // No separate supervisor identity: the command is not run under a different identity
410            // than the process supervising it.
411            supervisor_isolation: false,
412        }
413    }
414
415    /// Returns a typed error if the named capability is absent on this platform.
416    pub fn require(&self, capability: SandboxCapability, platform: Platform) -> Result<()> {
417        let available = match capability {
418            SandboxCapability::Files => self.files,
419            SandboxCapability::Reconnect => self.reconnect,
420            SandboxCapability::Jobs => self.jobs,
421            SandboxCapability::Preview => self.preview,
422            SandboxCapability::PauseResume => self.pause_resume,
423            SandboxCapability::Snapshot => self.snapshot,
424            SandboxCapability::DomainEgressRules => self.domain_egress_rules,
425            SandboxCapability::EgressDeny => self.egress_deny,
426            SandboxCapability::EnforcedLimits => self.enforced_limits,
427            SandboxCapability::ProcessLimit => self.process_limit,
428            SandboxCapability::SandboxLifetime => self.sandbox_lifetime,
429            SandboxCapability::SupervisorPidNamespace => self.supervisor_pid_namespace,
430            SandboxCapability::SupervisorIsolation => self.supervisor_isolation,
431        };
432
433        if available {
434            return Ok(());
435        }
436
437        Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
438            capability: capability.as_str().to_string(),
439            platform: platform.to_string(),
440        }))
441    }
442}
443
444/// Names a single sandbox capability, so an unsupported call can report which one it needed.
445#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
446#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
447#[serde(rename_all = "camelCase")]
448pub enum SandboxCapability {
449    /// Moving files in and out of a sandbox
450    Files,
451    /// Reaching a sandbox created by an earlier call
452    Reconnect,
453    /// Starting, polling and cancelling a command across separate calls
454    Jobs,
455    /// An authenticated, port-scoped ingress capability
456    Preview,
457    /// Pausing and resuming sandbox state
458    PauseResume,
459    /// Capturing full sandbox state
460    Snapshot,
461    /// Restricting egress to a hostname allowlist
462    DomainEgressRules,
463    /// Refusing outbound access when a sandbox declares none
464    EgressDeny,
465    /// Platform-enforced resource ceilings
466    EnforcedLimits,
467    /// A ceiling on the number of processes a sandbox may run
468    ProcessLimit,
469    /// A wall-clock ceiling on a sandbox, applied by the platform rather than by a caller
470    SandboxLifetime,
471    /// A command runs in its own PID namespace, isolated from the agent supervising it
472    SupervisorPidNamespace,
473    /// A command runs under a different identity than the process supervising it
474    SupervisorIsolation,
475}
476
477impl SandboxCapability {
478    /// Returns the stable identifier used in errors and capability queries.
479    pub fn as_str(&self) -> &'static str {
480        match self {
481            Self::Files => "files",
482            Self::Reconnect => "reconnect",
483            Self::Jobs => "jobs",
484            Self::Preview => "preview",
485            Self::PauseResume => "pauseResume",
486            Self::Snapshot => "snapshot",
487            Self::DomainEgressRules => "domainEgressRules",
488            Self::EgressDeny => "egressDeny",
489            Self::EnforcedLimits => "enforcedLimits",
490            Self::ProcessLimit => "processLimit",
491            Self::SandboxLifetime => "sandboxLifetime",
492            Self::SupervisorPidNamespace => "supervisorPidNamespace",
493            Self::SupervisorIsolation => "supervisorIsolation",
494        }
495    }
496}
497
498/// An isolated environment for running untrusted code, created at runtime.
499#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Builder)]
500#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
501#[serde(rename_all = "camelCase", deny_unknown_fields)]
502#[builder(start_fn = new)]
503pub struct Sandbox {
504    /// Identifier for the sandbox. Must contain only alphanumeric characters, hyphens, and
505    /// underscores ([A-Za-z0-9-_]). Maximum 64 characters.
506    #[builder(start_fn)]
507    pub id: String,
508    /// Where the sandbox's root filesystem comes from
509    pub code: SandboxCode,
510    /// Private ECR base image an AWS build pulls; `code.image` names only the S3 bundle, so this is
511    /// what the cross-account grant opens. Live only: the grant needs the customer account,
512    /// which registration reports. Absent means the base image is pulled anonymously.
513    #[serde(skip_serializing_if = "Option::is_none")]
514    pub private_base_image: Option<String>,
515    /// Enforced resource ceilings.
516    ///
517    /// Optional because not every platform can enforce them, and a declaration that names none
518    /// takes the platform's own defaults. Naming them on a platform that cannot enforce them is
519    /// rejected at plan time rather than silently ignored.
520    #[serde(skip_serializing_if = "Option::is_none")]
521    pub limits: Option<SandboxLimits>,
522    /// Outbound network policy
523    pub egress: SandboxEgress,
524    /// Sandbox lifetime ceiling and idle behaviour
525    pub lifecycle: SandboxLifecyclePolicy,
526    /// Ports eligible for a preview capability. An application reaches its sandbox through the
527    /// provider, so it cannot widen its own ingress at runtime; a holder of a remote binding's
528    /// credentials is bounded by no port condition, which is why a remote sandbox declares none.
529    #[builder(default)]
530    #[serde(default, skip_serializing_if = "Vec::is_empty")]
531    pub preview_ports: Vec<u16>,
532}
533
534/// Whether the artifact being rendered restricts which network modes it accepts.
535///
536/// Cloud setup needs explicit subnets for restricted sandbox connectors. Kubernetes targets do
537/// not emit these cloud backends.
538pub fn restricts_network_mode(stack: &crate::Stack, targets_kubernetes: bool) -> bool {
539    !targets_kubernetes && stack_needs_named_subnets_at_setup(stack)
540}
541
542/// Whether any setup-owned resource forces setup to name subnets.
543///
544/// Restricted sandbox connectors require subnet IDs. Callers rendering an artifact want
545/// [`restricts_network_mode`] instead:
546/// this one answers for the declaration, which on a Kubernetes target is not what gets emitted.
547pub fn stack_needs_named_subnets_at_setup(stack: &crate::Stack) -> bool {
548    stack.resources().any(|(_resource_id, resource)| {
549        resource
550            .config
551            .downcast_ref::<Sandbox>()
552            .is_some_and(|sandbox| !matches!(sandbox.egress, SandboxEgress::Allow))
553    })
554}
555
556impl Sandbox {
557    /// The resource type identifier for Sandbox
558    pub const RESOURCE_TYPE: ResourceType = ResourceType::from_static("sandbox");
559
560    /// Returns the sandbox's unique identifier.
561    pub fn id(&self) -> &str {
562        &self.id
563    }
564
565    /// The declared ceilings, or the defaults a platform applies when none were named.
566    ///
567    /// Backends want a concrete set: a sandbox with no declared ceilings still runs inside
568    /// whatever the platform gives it, and a backend that had to branch on `None` would end up
569    /// inventing its own default anyway.
570    pub fn resolved_limits(&self) -> SandboxLimits {
571        self.limits.clone().unwrap_or_else(default_limits)
572    }
573
574    /// Validates the declaration against what the target platform can enforce.
575    ///
576    /// Runs at plan time so an unenforceable limit or an unsupported egress mode fails before
577    /// anything is provisioned, rather than at the first exec.
578    pub fn validate_for_platform(&self, platform: Platform) -> Result<()> {
579        let capabilities = SandboxCapabilities::for_platform(platform)?;
580
581        // `alien build` builds an AWS sandbox's base image, so source is a declaration there and
582        // the emitters refuse it only if it reaches them unbuilt. Everywhere else the image is
583        // pulled as declared, and an empty image string would schedule a pod that can never run.
584        if matches!(&self.code, SandboxCode::Source { .. }) && platform != Platform::Aws {
585            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
586                resource_id: self.id.clone(),
587                field: "code".to_string(),
588                value: "source".to_string(),
589                reason: format!(
590                    "no sandbox backend builds an image from source on {platform}; give \
591                     code.image a prebuilt reference"
592                ),
593            }));
594        }
595
596        // Elsewhere `code.image` is pulled directly, so a second reference would be a grant
597        // nothing reads.
598        if self.private_base_image.is_some() && platform != Platform::Aws {
599            return Err(AlienError::new(ErrorData::SandboxCapabilityUnsupported {
600                capability: "privateBaseImage".to_string(),
601                platform: platform.to_string(),
602            }));
603        }
604
605        // Read before the limits, because the image is declared whether or not any are.
606        if platform == Platform::Azure {
607            self.azure_image()?;
608        }
609
610        let Some(limits) = self.limits.as_ref() else {
611            // Nothing declared, so nothing to enforce and nothing to reject.
612            return self.validate_capabilities(&capabilities, platform);
613        };
614
615        validate_quantity(&self.id, "cpu", &limits.cpu)?;
616        validate_quantity(&self.id, "memory", &limits.memory)?;
617        validate_quantity(&self.id, "disk", &limits.disk)?;
618
619        if let Some(max_processes) = limits.max_processes {
620            if max_processes == 0 {
621                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
622                    resource_id: self.id.clone(),
623                    field: "maxProcesses".to_string(),
624                    value: "0".to_string(),
625                    reason: "a sandbox that may run no processes cannot run code".to_string(),
626                }));
627            }
628            capabilities.require(SandboxCapability::ProcessLimit, platform)?;
629        }
630
631        // Declaring limits a platform ignores is worse than not declaring them: the stack reads
632        // as bounded while the sandbox is not.
633        capabilities.require(SandboxCapability::EnforcedLimits, platform)?;
634
635        if platform == Platform::Azure {
636            self.azure_sandbox_limits()?;
637        }
638
639        if platform == Platform::Aws {
640            // Refused here rather than at emit so a customer sees it while planning, and so both
641            // package formats inherit the same answer.
642            self.microvm_tier()?;
643
644            // The ceiling is Lambda's, and it rejects the run rather than clamping — so a value
645            // outside it would pass planning, render into the package, and fail at the first
646            // sandbox. Kubernetes takes the same field with no such bound, which is why this
647            // sits under the AWS gate rather than on the type.
648            if let Some(seconds) = self.lifecycle.max_lifetime_seconds {
649                if !(1..=AWS_MAX_LIFETIME_SECONDS).contains(&seconds) {
650                    return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
651                        resource_id: self.id.clone(),
652                        field: "maxLifetimeSeconds".to_string(),
653                        value: seconds.to_string(),
654                        reason: format!(
655                            "AWS runs a MicroVM for between 1 and \
656                             {AWS_MAX_LIFETIME_SECONDS} seconds"
657                        ),
658                    }));
659                }
660            }
661        }
662
663        self.validate_capabilities(&capabilities, platform)
664    }
665
666    /// What Azure creates this sandbox from: a catalog name or a registry image, told apart by
667    /// [`classify_azure_sandbox_image`]. Refused while planning, because a value the data plane
668    /// rejects would otherwise surface at the first sandbox, long after the apply.
669    pub fn azure_image(&self) -> Result<AzureSandboxImage<'_>> {
670        let SandboxCode::Image { image } = &self.code else {
671            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
672                resource_id: self.id.clone(),
673                field: "code".to_string(),
674                value: "source".to_string(),
675                reason: "no sandbox backend builds an image from source yet".to_string(),
676            }));
677        };
678
679        classify_azure_sandbox_image(image).ok_or_else(|| {
680            let image = image.trim();
681            AlienError::new(ErrorData::SandboxLimitInvalid {
682                resource_id: self.id.clone(),
683                field: "code.image".to_string(),
684                value: image.to_string(),
685                reason: if image.is_empty() {
686                    "a sandbox has to name an image".to_string()
687                } else {
688                    "Azure creates a sandbox from a catalog name such as 'ubuntu' or from a \
689                     registry image such as 'docker.io/library/python:3.14-slim'"
690                        .to_string()
691                },
692            })
693        })
694    }
695
696    /// Checks the declared ceilings against Azure's sizing rule (the `AZURE_*` constants above).
697    /// Refused at plan time, like [`Self::microvm_tier`], so a bad value is a declaration to fix
698    /// rather than a runtime fault at create.
699    pub fn azure_sandbox_limits(&self) -> Result<()> {
700        let Some(limits) = self.limits.as_ref() else {
701            // Nothing declared means the binding substitutes Alien's own default sizing, which is
702            // inside the rule — asserted where those constants live, since this cannot see them.
703            return Ok(());
704        };
705
706        let refused = |field: &str, value: &str, reason: &str| {
707            AlienError::new(ErrorData::SandboxLimitInvalid {
708                resource_id: self.id.clone(),
709                field: field.to_string(),
710                value: value.to_string(),
711                reason: reason.to_string(),
712            })
713        };
714
715        let cpu_millicores = millicores(&limits.cpu)
716            .ok_or_else(|| refused("cpu", &limits.cpu, "expected cores or millicores"))?;
717
718        // The multiple is checked, not just the range: `333m` sits inside 0.25–16 cores and is
719        // still refused on the wire, so a bounds-only check would pass a declaration that fails
720        // at create.
721        if cpu_millicores % AZURE_CPU_STEP_MILLICORES != 0
722            || !(AZURE_CPU_STEP_MILLICORES..=AZURE_MAX_CPU_MILLICORES).contains(&cpu_millicores)
723        {
724            return Err(refused(
725                "cpu",
726                &limits.cpu,
727                "Azure allocates cpu in steps of 250m from 250m to 16000m",
728            ));
729        }
730
731        // Both ceilings are derived from the cpu, so they cannot be checked before it is known.
732        let memory_ceiling_mib = cpu_millicores * AZURE_MEMORY_MIB_PER_CORE / 1000;
733        let disk_ceiling_mib = cpu_millicores * AZURE_DISK_MIB_PER_CORE / 1000;
734
735        let memory_mib = quantity_mib(&limits.memory)
736            .ok_or_else(|| refused("memory", &limits.memory, "Azure sizes memory in whole MiB"))?;
737        if memory_mib > memory_ceiling_mib {
738            return Err(refused(
739                "memory",
740                &limits.memory,
741                &format!(
742                    "Azure allows at most 2Gi of memory per core, or {memory_ceiling_mib}Mi \
743                          at the declared cpu"
744                ),
745            ));
746        }
747
748        let disk_mib = quantity_mib(&limits.disk)
749            .ok_or_else(|| refused("disk", &limits.disk, "Azure sizes disk in whole MiB"))?;
750        if disk_mib > disk_ceiling_mib {
751            return Err(refused(
752                "disk",
753                &limits.disk,
754                &format!(
755                    "Azure allows at most 20Gi of disk per core, or {disk_ceiling_mib}Mi at \
756                          the declared cpu"
757                ),
758            ));
759        }
760
761        Ok(())
762    }
763
764    /// The MicroVM size that keeps every declared ceiling, or why none does.
765    ///
766    /// AWS sizes are discrete and a running MicroVM bursts to four times its baseline, so the
767    /// only tier that honours a ceiling is one whose peak fits inside it. A declaration no tier
768    /// satisfies is refused: shipping the nearest size would give the customer a sandbox that
769    /// exceeds the bound they wrote down.
770    pub fn microvm_tier(&self) -> Result<MicrovmTier> {
771        let Some(limits) = self.limits.as_ref() else {
772            // Nothing declared: AWS's own default baseline, which is also `default_limits`.
773            return Ok(MICROVM_TIERS[2]);
774        };
775
776        let memory_mib = quantity_mib(&limits.memory).ok_or_else(|| {
777            AlienError::new(ErrorData::SandboxLimitInvalid {
778                resource_id: self.id.clone(),
779                field: "memory".to_string(),
780                value: limits.memory.clone(),
781                reason: "AWS sizes a MicroVM in whole MiB".to_string(),
782            })
783        })?;
784        let disk_mib = quantity_mib(&limits.disk).ok_or_else(|| {
785            AlienError::new(ErrorData::SandboxLimitInvalid {
786                resource_id: self.id.clone(),
787                field: "disk".to_string(),
788                value: limits.disk.clone(),
789                reason: "AWS sizes a MicroVM's disk in whole MiB".to_string(),
790            })
791        })?;
792        let cpu_millicores = millicores(&limits.cpu).ok_or_else(|| {
793            AlienError::new(ErrorData::SandboxLimitInvalid {
794                resource_id: self.id.clone(),
795                field: "cpu".to_string(),
796                value: limits.cpu.clone(),
797                reason: "expected cores or millicores".to_string(),
798            })
799        })?;
800
801        // Memory and disk choose the size; cpu is then checked rather than used to choose.
802        // AWS couples cpu to memory at 2 GB per vCPU, so letting a low cpu ceiling select the
803        // size too would quietly hand back a machine four times smaller than the memory ceiling
804        // asked for, with nothing to indicate it.
805        let sized = |tier: &&MicrovmTier| {
806            tier.peak_memory_mib <= memory_mib && tier.max_disk_mib <= disk_mib
807        };
808
809        let tier = MICROVM_TIERS
810            .iter()
811            .rev()
812            .find(sized)
813            .copied()
814            .ok_or_else(|| {
815                AlienError::new(ErrorData::SandboxLimitInvalid {
816                    resource_id: self.id.clone(),
817                    field: "memory".to_string(),
818                    value: limits.memory.clone(),
819                    reason: format!(
820                        "a Lambda MicroVM bursts to four times its baseline, so the smallest \
821                         ceiling AWS can hold is 2Gi memory with 8Gi disk; '{}' memory and '{}' \
822                         disk fit no size",
823                        limits.memory, limits.disk
824                    ),
825                })
826            })?;
827
828        let required_millicores = i64::from(tier.peak_vcpu) * 1000;
829        if cpu_millicores < required_millicores {
830            return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
831                resource_id: self.id.clone(),
832                field: "cpu".to_string(),
833                value: limits.cpu.clone(),
834                reason: format!(
835                    "AWS allocates one vCPU per 2GB, so a MicroVM sized to a '{}' memory ceiling \
836                     reaches {} vCPU; declare cpu '{}' or lower the memory ceiling",
837                    limits.memory, tier.peak_vcpu, tier.peak_vcpu
838                ),
839            }));
840        }
841
842        Ok(tier)
843    }
844
845    /// The capability checks that do not depend on declared limits.
846    fn validate_capabilities(
847        &self,
848        capabilities: &SandboxCapabilities,
849        platform: Platform,
850    ) -> Result<()> {
851        if matches!(self.egress, SandboxEgress::AllowDomains { .. }) {
852            capabilities.require(SandboxCapability::DomainEgressRules, platform)?;
853        }
854
855        // `allow` asks for no restriction, so a backend that ignores it fails loudly on the first
856        // blocked connection. `deny` asks for one, and a backend that ignores it puts untrusted
857        // code on the internet with nothing to notice — so only this direction is gated.
858        // An empty list is not a restriction anyone wrote down: it renders as a deny-all wearing
859        // an allowlist's label, which reads at a glance as the opposite of what it does.
860        if let SandboxEgress::AllowDomains { domains } = &self.egress {
861            if domains.is_empty() {
862                return Err(AlienError::new(ErrorData::SandboxLimitInvalid {
863                    resource_id: self.id.clone(),
864                    field: "egress.domains".to_string(),
865                    value: "[]".to_string(),
866                    reason: "an allowlist naming no domain denies everything; declare \
867                             egress: deny if that is what was meant"
868                        .to_string(),
869                }));
870            }
871        }
872
873        if matches!(self.egress, SandboxEgress::Deny) {
874            capabilities.require(SandboxCapability::EgressDeny, platform)?;
875        }
876
877        if !self.preview_ports.is_empty() {
878            capabilities.require(SandboxCapability::Preview, platform)?;
879        }
880
881        if self.lifecycle.idle_pause_seconds.is_some() {
882            capabilities.require(SandboxCapability::PauseResume, platform)?;
883        }
884
885        if self.lifecycle.max_lifetime_seconds.is_some() {
886            capabilities.require(SandboxCapability::SandboxLifetime, platform)?;
887        }
888
889        Ok(())
890    }
891}
892
893/// Ceilings applied when a declaration names none.
894///
895/// Modest on purpose: an undeclared sandbox is one whose author did not think about sizing, and
896/// the safe reading of that is a small box rather than a generous one.
897fn default_limits() -> SandboxLimits {
898    SandboxLimits {
899        cpu: "1".to_string(),
900        memory: "2Gi".to_string(),
901        disk: "8Gi".to_string(),
902        max_processes: None,
903    }
904}
905
906/// Validates a Kubernetes-style resource quantity such as `500m`, `2Gi` or `1`.
907fn validate_quantity(resource_id: &str, field: &str, value: &str) -> Result<()> {
908    let invalid = |reason: &str| {
909        AlienError::new(ErrorData::SandboxLimitInvalid {
910            resource_id: resource_id.to_string(),
911            field: field.to_string(),
912            value: value.to_string(),
913            reason: reason.to_string(),
914        })
915    };
916
917    let digits_end = value
918        .find(|c: char| !c.is_ascii_digit() && c != '.')
919        .unwrap_or(value.len());
920    let (number, suffix) = value.split_at(digits_end);
921
922    let parsed: f64 = number
923        .parse()
924        .map_err(|_| invalid("expected a number, optionally followed by a unit suffix"))?;
925
926    if parsed <= 0.0 {
927        return Err(invalid("must be greater than zero"));
928    }
929
930    const SUFFIXES: &[&str] = &["", "m", "k", "M", "G", "T", "Ki", "Mi", "Gi", "Ti"];
931    if !SUFFIXES.contains(&suffix) {
932        return Err(invalid(
933            "unit must be one of m, k, M, G, T, Ki, Mi, Gi, Ti, or absent",
934        ));
935    }
936
937    Ok(())
938}
939
940/// Splits a quantity into its number and unit suffix.
941fn split_quantity(value: &str) -> Option<(f64, &str)> {
942    let trimmed = value.trim();
943    let digits_end = trimmed
944        .find(|c: char| !c.is_ascii_digit() && c != '.')
945        .unwrap_or(trimmed.len());
946    let (number, suffix) = trimmed.split_at(digits_end);
947    number.parse().ok().map(|number| (number, suffix))
948}
949
950/// A memory or disk quantity in whole MiB, rounded down.
951///
952/// Every suffix `validate_quantity` accepts is handled here. Reading only `Gi` and `Mi` and
953/// falling back for the rest would turn a declared `4G` into a different size than the customer
954/// asked for, which for a ceiling means a sandbox larger than its bound.
955pub fn quantity_mib(value: &str) -> Option<i64> {
956    let (number, suffix) = split_quantity(value)?;
957    let bytes = match suffix {
958        "" => number,
959        "k" => number * 1e3,
960        "M" => number * 1e6,
961        "G" => number * 1e9,
962        "T" => number * 1e12,
963        "Ki" => number * 1024.0,
964        "Mi" => number * 1024.0 * 1024.0,
965        "Gi" => number * 1024.0 * 1024.0 * 1024.0,
966        "Ti" => number * 1024.0 * 1024.0 * 1024.0 * 1024.0,
967        // `m` is a millicore suffix; memory has no use for it.
968        _ => return None,
969    };
970    Some((bytes / (1024.0 * 1024.0)) as i64)
971}
972
973/// A CPU quantity in millicores.
974pub fn millicores(value: &str) -> Option<i64> {
975    let (number, suffix) = split_quantity(value)?;
976    match suffix {
977        "" => Some((number * 1000.0) as i64),
978        "m" => Some(number as i64),
979        _ => None,
980    }
981}
982
983/// Outputs generated by a successfully provisioned Sandbox parent.
984#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
985#[cfg_attr(feature = "openapi", derive(utoipa::ToSchema))]
986#[serde(rename_all = "camelCase")]
987pub struct SandboxOutputs {
988    /// Name of the durable parent that sandboxes are created inside
989    pub parent_name: String,
990    /// Platform-specific identifier for the parent (image ARN, sandbox group id, namespace)
991    #[serde(skip_serializing_if = "Option::is_none")]
992    pub identifier: Option<String>,
993    /// Data-plane endpoint sandboxes are created through, where the platform has one
994    #[serde(skip_serializing_if = "Option::is_none")]
995    pub endpoint: Option<String>,
996}
997
998impl ResourceOutputsDefinition for SandboxOutputs {
999    fn get_resource_type(&self) -> ResourceType {
1000        Sandbox::RESOURCE_TYPE
1001    }
1002
1003    fn as_any(&self) -> &dyn Any {
1004        self
1005    }
1006
1007    fn box_clone(&self) -> Box<dyn ResourceOutputsDefinition> {
1008        Box::new(self.clone())
1009    }
1010
1011    fn outputs_eq(&self, other: &dyn ResourceOutputsDefinition) -> bool {
1012        other.as_any().downcast_ref::<SandboxOutputs>() == Some(self)
1013    }
1014
1015    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1016        serde_json::to_value(self)
1017    }
1018}
1019
1020impl ResourceDefinition for Sandbox {
1021    fn get_resource_type(&self) -> ResourceType {
1022        Self::RESOURCE_TYPE
1023    }
1024
1025    fn id(&self) -> &str {
1026        &self.id
1027    }
1028
1029    fn get_dependencies(&self) -> Vec<ResourceRef> {
1030        Vec::new()
1031    }
1032
1033    fn validate_update(&self, new_config: &dyn ResourceDefinition) -> Result<()> {
1034        let new_sandbox = new_config
1035            .as_any()
1036            .downcast_ref::<Sandbox>()
1037            .ok_or_else(|| {
1038                AlienError::new(ErrorData::UnexpectedResourceType {
1039                    resource_id: self.id.clone(),
1040                    expected: Self::RESOURCE_TYPE,
1041                    actual: new_config.get_resource_type(),
1042                })
1043            })?;
1044
1045        if self.id != new_sandbox.id {
1046            return Err(AlienError::new(ErrorData::InvalidResourceUpdate {
1047                resource_id: self.id.clone(),
1048                reason: "the 'id' field is immutable".to_string(),
1049            }));
1050        }
1051
1052        Ok(())
1053    }
1054
1055    fn as_any(&self) -> &dyn Any {
1056        self
1057    }
1058
1059    fn as_any_mut(&mut self) -> &mut dyn Any {
1060        self
1061    }
1062
1063    fn box_clone(&self) -> Box<dyn ResourceDefinition> {
1064        Box::new(self.clone())
1065    }
1066
1067    fn resource_eq(&self, other: &dyn ResourceDefinition) -> bool {
1068        other.as_any().downcast_ref::<Sandbox>() == Some(self)
1069    }
1070
1071    fn to_json_value(&self) -> serde_json::Result<serde_json::Value> {
1072        serde_json::to_value(self)
1073    }
1074}
1075
1076/// What an Azure sandbox starts from.
1077#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1078pub enum AzureSandboxImage<'a> {
1079    /// A public catalog disk image, such as `ubuntu`, which the data plane names directly.
1080    Catalog(&'a str),
1081    /// A registry image, which the controller builds into a disk image in the sandbox's group
1082    /// before any sandbox can start from it.
1083    Registry(&'a str),
1084}
1085
1086impl<'a> AzureSandboxImage<'a> {
1087    /// The declared value, trimmed.
1088    pub fn as_str(&self) -> &'a str {
1089        match self {
1090            Self::Catalog(value) | Self::Registry(value) => value,
1091        }
1092    }
1093}
1094
1095/// Label key the controller writes on every disk image it builds, and the provider finds it by.
1096pub const AZURE_DISK_IMAGE_LABEL: &str = "alienImage";
1097
1098/// Classifies a declared `code.image` for Azure, or `None` when it is neither kind. A bare
1099/// `[A-Za-z0-9._-]+` is checked first and always a catalog name; anything else must carry `/`,
1100/// `:` or `@` and parse as an OCI reference.
1101pub fn classify_azure_sandbox_image(image: &str) -> Option<AzureSandboxImage<'_>> {
1102    let image = image.trim();
1103    if image.is_empty() {
1104        return None;
1105    }
1106    if image
1107        .chars()
1108        .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '-'))
1109    {
1110        return Some(AzureSandboxImage::Catalog(image));
1111    }
1112    (image.contains(|c| matches!(c, '/' | ':' | '@')) && OCI_REFERENCE.is_match(image))
1113        .then_some(AzureSandboxImage::Registry(image))
1114}
1115
1116/// The label value naming the disk image built from `reference`: a digest, since a label value
1117/// may not carry a reference's `/`, `:` and `@`.
1118pub fn azure_disk_image_label(reference: &str) -> String {
1119    let digest = format!("{:x}", Sha256::digest(reference.trim().as_bytes()));
1120    digest[..32].to_string()
1121}
1122
1123/// The distribution reference grammar: `[host[:port]/]path[:tag][@digest]`.
1124static OCI_REFERENCE: std::sync::LazyLock<regex::Regex> = std::sync::LazyLock::new(|| {
1125    let domain_component = r"(?:[a-zA-Z0-9]|[a-zA-Z0-9][a-zA-Z0-9-]*[a-zA-Z0-9])";
1126    let domain = format!(r"{domain_component}(?:\.{domain_component})*(?::[0-9]+)?");
1127    let path_component = r"[a-z0-9]+(?:(?:[._]|__|-+)[a-z0-9]+)*";
1128    let name = format!(r"(?:{domain}/)?{path_component}(?:/{path_component})*");
1129    let tag = r"[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}";
1130    let digest = r"[A-Za-z][A-Za-z0-9]*(?:[-_+.][A-Za-z][A-Za-z0-9]*)*:[0-9a-fA-F]{32,}";
1131    regex::Regex::new(&format!(r"^{name}(?::{tag})?(?:@{digest})?$"))
1132        .expect("the OCI reference grammar compiles")
1133});
1134
1135/// The one token a sandbox bundle URI may carry, replaced with the deploying region.
1136///
1137/// AWS builds a MicroVM image only from a bucket in the image's own region, so a vendor
1138/// publishing to every supported region needs one stored URI that resolves per region.
1139pub const BUNDLE_REGION_TOKEN: &str = "{region}";
1140
1141/// A bundle URI split around its region token, or carried whole when it has none.
1142#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1143pub enum BundleUri<'a> {
1144    /// No token: emitted exactly as it is today.
1145    Literal(&'a str),
1146    /// The text either side of the token, for an emitter to rejoin around its own region
1147    /// expression.
1148    Regional { before: &'a str, after: &'a str },
1149}
1150
1151/// The prefix a rebuild's new key still sits under: everything above the file name and the
1152/// version segment beneath it. `None` when nothing sits there, meaning no prefix can be granted
1153/// without also granting objects a rebuild never reads. Shared so both emitters agree on it.
1154pub fn stable_bundle_key_prefix(key: &str) -> Option<&str> {
1155    let (above_file, _) = key.rsplit_once('/')?;
1156    let (above_version, _) = above_file.rsplit_once('/')?;
1157    Some(above_version)
1158}
1159
1160/// Reads a sandbox bundle URI, refusing anything an image build would only reject later.
1161///
1162/// The token is accepted in the bucket alone. A key-position token would name an object that does
1163/// not exist, and any other brace is a typo that would otherwise reach S3 verbatim and fail ~160s
1164/// into the build — which is the failure this whole check exists to move to plan time.
1165pub fn parse_bundle_uri(uri: &str) -> std::result::Result<BundleUri<'_>, String> {
1166    let path = uri
1167        .strip_prefix("s3://")
1168        .ok_or_else(|| format!("'{uri}' is not an s3:// URI"))?;
1169    let (bucket, key) = path
1170        .split_once('/')
1171        .ok_or_else(|| format!("'{uri}' names a bucket with no object key"))?;
1172
1173    // Both emitters interpolate this path into the build role's resource ARN, where `*` and `?`
1174    // are IAM wildcards rather than literal characters. S3 accepts them in a key, so a bundle
1175    // published under one would silently widen the grant past the bundle it names.
1176    if path.contains('*') || path.contains('?') {
1177        return Err(format!(
1178            "'{uri}' carries an IAM wildcard; the bundle's path is interpolated into the build \
1179             role's grant, so '*' and '?' would widen it past the bundle"
1180        ));
1181    }
1182
1183    if key.contains('{') || key.contains('}') {
1184        return Err(format!(
1185            "'{uri}' places a token in the object key; {BUNDLE_REGION_TOKEN} is accepted in the \
1186             bucket name alone"
1187        ));
1188    }
1189
1190    let Some((before, after)) = bucket.split_once(BUNDLE_REGION_TOKEN) else {
1191        if bucket.contains('{') || bucket.contains('}') {
1192            return Err(format!(
1193                "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the \
1194                 only one"
1195            ));
1196        }
1197        return Ok(BundleUri::Literal(uri));
1198    };
1199
1200    if after.contains(BUNDLE_REGION_TOKEN) {
1201        return Err(format!("'{uri}' repeats {BUNDLE_REGION_TOKEN}"));
1202    }
1203    if before.contains('{') || before.contains('}') || after.contains('{') || after.contains('}') {
1204        return Err(format!(
1205            "'{uri}' carries a token this build does not know; {BUNDLE_REGION_TOKEN} is the only one"
1206        ));
1207    }
1208
1209    Ok(BundleUri::Regional {
1210        before: &uri[.."s3://".len() + before.len()],
1211        after: &uri["s3://".len() + before.len() + BUNDLE_REGION_TOKEN.len()..],
1212    })
1213}
1214
1215/// Where a private ECR image's region comes from.
1216#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1217pub enum EcrImageRegion<'a> {
1218    /// A region named in the host.
1219    Literal(&'a str),
1220    /// [`BUNDLE_REGION_TOKEN`] in the host: the region the deployment renders.
1221    Deployment,
1222}
1223
1224/// The ECR repository a private image reference is pulled from.
1225#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1226pub struct EcrImageRepository<'a> {
1227    pub account_id: &'a str,
1228    pub region: EcrImageRegion<'a>,
1229    /// The repository name, which may carry `/`; never a tag or digest.
1230    pub repository: &'a str,
1231}
1232
1233impl EcrImageRepository<'_> {
1234    /// The repository's ARN, with `region` standing in for a region the host leaves to the
1235    /// deployment. The partition is the deployment's: a GovCloud host ends `.amazonaws.com` too.
1236    pub fn arn(&self, partition: &str, region: &str) -> String {
1237        let region = match self.region {
1238            EcrImageRegion::Literal(region) => region,
1239            EcrImageRegion::Deployment => region,
1240        };
1241        format!(
1242            "arn:{partition}:ecr:{region}:{}:repository/{}",
1243            self.account_id, self.repository
1244        )
1245    }
1246}
1247
1248/// Reads `privateBaseImage` as the one repository a build role may pull from.
1249///
1250/// The name is interpolated into an IAM ARN, so it is held to ECR's own repository grammar: that
1251/// refuses `*` and `?`, which would widen the grant, and `$` and braces, which a CloudFormation
1252/// `Sub` or a Terraform template would read as an expression.
1253pub fn parse_ecr_image_repository(
1254    image: &str,
1255) -> std::result::Result<EcrImageRepository<'_>, String> {
1256    let refuse = |reason: &str| format!("privateBaseImage '{image}' {reason}");
1257    let (host, path) = image
1258        .split_once('/')
1259        .ok_or_else(|| refuse("names no repository"))?;
1260    let (account_id, rest) = host.split_once(".dkr.ecr.").ok_or_else(|| {
1261        refuse("is not served by a private ECR registry (<account>.dkr.ecr.<region>.amazonaws.com)")
1262    })?;
1263    // `.com.cn` first: a China host ends with the shorter suffix too.
1264    let region = rest
1265        .strip_suffix(".amazonaws.com.cn")
1266        .or_else(|| rest.strip_suffix(".amazonaws.com"))
1267        .ok_or_else(|| refuse("is not served by a private ECR registry (<account>.dkr.ecr.<region>.amazonaws.com)"))?;
1268    if account_id.len() != 12 || !account_id.bytes().all(|b| b.is_ascii_digit()) {
1269        return Err(refuse("names no 12-digit account in its registry host"));
1270    }
1271    let region = if region == BUNDLE_REGION_TOKEN {
1272        EcrImageRegion::Deployment
1273    } else if !region.is_empty()
1274        && region
1275            .bytes()
1276            .all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'-')
1277    {
1278        EcrImageRegion::Literal(region)
1279    } else {
1280        return Err(refuse(&format!(
1281            "names no region in its registry host; give one or {BUNDLE_REGION_TOKEN}"
1282        )));
1283    };
1284
1285    let repository = match path.split_once('@') {
1286        Some((repository, _digest)) => repository,
1287        None => match path.rsplit_once('/') {
1288            Some((parent, last)) => match last.split_once(':') {
1289                Some((name, _tag)) => &path[..parent.len() + 1 + name.len()],
1290                None => path,
1291            },
1292            None => path.split_once(':').map_or(path, |(name, _tag)| name),
1293        },
1294    };
1295    let valid_segment = |segment: &str| {
1296        !segment.is_empty()
1297            && segment.bytes().all(|b| {
1298                b.is_ascii_lowercase() || b.is_ascii_digit() || matches!(b, b'.' | b'_' | b'-')
1299            })
1300    };
1301    if !repository.split('/').all(valid_segment) {
1302        return Err(refuse(
1303            "names a repository outside ECR's grammar (lowercase letters, digits, '.', '_', '-', and '/' between them)",
1304        ));
1305    }
1306    Ok(EcrImageRepository {
1307        account_id,
1308        region,
1309        repository,
1310    })
1311}
1312
1313#[cfg(test)]
1314mod tests {
1315    use super::*;
1316
1317    #[test]
1318    fn private_database_setup_accepts_the_default_network() {
1319        for lifecycle in [
1320            crate::ResourceLifecycle::Frozen,
1321            crate::ResourceLifecycle::Live,
1322        ] {
1323            let stack = crate::Stack::new("database".to_string())
1324                .add(
1325                    crate::Postgres::new("metadata".to_string()).build(),
1326                    lifecycle,
1327                )
1328                .build();
1329            assert!(!restricts_network_mode(&stack, false));
1330            assert!(!restricts_network_mode(&stack, true));
1331        }
1332        assert!(!restricts_network_mode(
1333            &crate::Stack::new("empty".to_string()).build(),
1334            false,
1335        ));
1336    }
1337
1338    /// The vectors' `repositoryKey`: registry host and repository, never the tag or digest, so a
1339    /// new tag keeps the key. A reference the parser refuses stays whole.
1340    fn repository_key(image: &str) -> String {
1341        match parse_ecr_image_repository(image) {
1342            Ok(parsed) => {
1343                let host = image.split_once('/').map_or(image, |(host, _)| host);
1344                format!("{host}/{}", parsed.repository)
1345            }
1346            Err(_) => image.to_string(),
1347        }
1348    }
1349
1350    #[test]
1351    fn a_private_base_image_names_one_repository() {
1352        let vectors: serde_json::Value = serde_json::from_str(include_str!(
1353            "../../tests/fixtures/ecr-image-repository-parity.json"
1354        ))
1355        .expect("the ECR repository vectors must be JSON");
1356        let field = |case: &serde_json::Value, name: &str| {
1357            case[name]
1358                .as_str()
1359                .unwrap_or_else(|| panic!("vector must carry {name}: {case}"))
1360                .to_string()
1361        };
1362
1363        for case in vectors["accepted"].as_array().expect("accepted vectors") {
1364            let image = field(case, "image");
1365            let region = field(case, "region");
1366            let region = if region == BUNDLE_REGION_TOKEN {
1367                EcrImageRegion::Deployment
1368            } else {
1369                EcrImageRegion::Literal(&region)
1370            };
1371            assert_eq!(
1372                parse_ecr_image_repository(&image),
1373                Ok(EcrImageRepository {
1374                    account_id: &field(case, "accountId"),
1375                    region,
1376                    repository: &field(case, "repository"),
1377                }),
1378                "{image}"
1379            );
1380            assert_eq!(
1381                repository_key(&image),
1382                field(case, "repositoryKey"),
1383                "{image}"
1384            );
1385        }
1386
1387        for case in vectors["refused"].as_array().expect("refused vectors") {
1388            let image = field(case, "image");
1389            assert!(
1390                parse_ecr_image_repository(&image).is_err(),
1391                "{image} must be refused"
1392            );
1393            assert_eq!(
1394                repository_key(&image),
1395                field(case, "repositoryKey"),
1396                "{image}"
1397            );
1398        }
1399    }
1400
1401    #[test]
1402    fn a_repository_arn_takes_the_deployment_region_only_where_the_host_leaves_it() {
1403        let regional =
1404            parse_ecr_image_repository("123456789012.dkr.ecr.{region}.amazonaws.com/team/base:1")
1405                .expect("parses");
1406        let pinned =
1407            parse_ecr_image_repository("123456789012.dkr.ecr.eu-west-1.amazonaws.com/team/base:1")
1408                .expect("parses");
1409        assert_eq!(
1410            regional.arn("aws-us-gov", "us-gov-west-1"),
1411            "arn:aws-us-gov:ecr:us-gov-west-1:123456789012:repository/team/base"
1412        );
1413        assert_eq!(
1414            pinned.arn("aws", "us-east-1"),
1415            "arn:aws:ecr:eu-west-1:123456789012:repository/team/base"
1416        );
1417    }
1418
1419    /// A wildcard reaching the grant would widen it past the bundle, and it widens the Frozen
1420    /// object grant as readily as the Live prefix — both interpolate the path into the ARN.
1421    #[test]
1422    fn a_uri_carrying_an_iam_wildcard_is_refused() {
1423        for uri in [
1424            "s3://acme/team-*/v1/bundle.zip",
1425            "s3://acme/sandbox-bundle/f00d/bundle?.zip",
1426            "s3://acme-*/sandbox-bundle/f00d/bundle.zip",
1427        ] {
1428            let error = parse_bundle_uri(uri).expect_err("a wildcard must be refused");
1429            assert!(error.contains("IAM wildcard"), "for {uri}: {error}");
1430        }
1431
1432        parse_bundle_uri("s3://acme/sandbox-bundle/f00d/bundle.zip")
1433            .expect("an ordinary key still parses");
1434    }
1435
1436    /// The rule both package formats grant by, pinned here rather than in either. The
1437    /// near-misses the cases separate: the key's first segment grants objects a rebuild never
1438    /// reads, and the object's own directory pins the version segment that moves.
1439    #[test]
1440    fn a_grantable_prefix_stops_above_the_segment_that_moves() {
1441        assert_eq!(
1442            stable_bundle_key_prefix("sandbox-bundle/f00dcafe/bundle.zip"),
1443            Some("sandbox-bundle")
1444        );
1445        assert_eq!(
1446            stable_bundle_key_prefix("artifacts/team-a/sandbox/f00dcafe/bundle.zip"),
1447            Some("artifacts/team-a/sandbox"),
1448            "a deeper key narrows the prefix, it never widens to the first segment"
1449        );
1450
1451        // Nothing sits above the version segment, so no prefix a moved bundle stays inside
1452        // exists. Emitting the object grant instead installs a role that denies the next rebuild.
1453        assert_eq!(stable_bundle_key_prefix("agents/bundle.zip"), None);
1454        assert_eq!(stable_bundle_key_prefix("bundle.zip"), None);
1455    }
1456
1457    fn sandbox_with(egress: SandboxEgress, preview_ports: Vec<u16>) -> Sandbox {
1458        Sandbox::new("agent-sbx".to_string())
1459            .code(SandboxCode::Image {
1460                image: "ubuntu".to_string(),
1461            })
1462            .limits(SandboxLimits {
1463                cpu: "1".to_string(),
1464                memory: "2Gi".to_string(),
1465                disk: "20Gi".to_string(),
1466                max_processes: None,
1467            })
1468            .egress(egress)
1469            .lifecycle(SandboxLifecyclePolicy {
1470                max_lifetime_seconds: None,
1471                idle_pause_seconds: None,
1472            })
1473            .preview_ports(preview_ports)
1474            .build()
1475    }
1476
1477    /// A URI with no token must come back whole, because every bundle configured today has none
1478    /// and emitting one differently would change every existing customer's template.
1479    #[test]
1480    fn a_uri_without_a_token_is_carried_whole() {
1481        assert_eq!(
1482            parse_bundle_uri("s3://acme-artifacts-us-east-2/agents/bundle.zip"),
1483            Ok(BundleUri::Literal(
1484                "s3://acme-artifacts-us-east-2/agents/bundle.zip"
1485            ))
1486        );
1487    }
1488
1489    /// The split has to rejoin to the original with the region in place, or an emitter builds a
1490    /// URI that is subtly not the one the vendor configured.
1491    #[test]
1492    fn a_regional_uri_splits_either_side_of_the_token() {
1493        let BundleUri::Regional { before, after } =
1494            parse_bundle_uri("s3://acme-artifacts-{region}/agents/bundle.zip")
1495                .expect("the token is accepted in the bucket")
1496        else {
1497            panic!("a bucket-position token must split");
1498        };
1499
1500        assert_eq!(before, "s3://acme-artifacts-");
1501        assert_eq!(after, "/agents/bundle.zip");
1502        assert_eq!(
1503            format!("{before}us-east-2{after}"),
1504            "s3://acme-artifacts-us-east-2/agents/bundle.zip",
1505            "the halves must rejoin to the URI the vendor meant"
1506        );
1507    }
1508
1509    /// Each of these reaches S3 verbatim and dies ~160s into an image build if it is not refused
1510    /// here, which is the whole reason this runs at plan time.
1511    #[test]
1512    fn a_token_this_build_cannot_resolve_is_refused() {
1513        for uri in [
1514            "s3://acme-artifacts-{regio}/bundle.zip",
1515            "s3://acme-artifacts/{region}/bundle.zip",
1516            "s3://acme-artifacts-{region}-{region}/bundle.zip",
1517            "s3://acme-artifacts/bundle-{version}.zip",
1518            "s3://acme}-artifacts-{region}/bundle.zip",
1519            "s3://acme{-artifacts-{region}/bundle.zip",
1520        ] {
1521            assert!(
1522                parse_bundle_uri(uri).is_err(),
1523                "'{uri}' must be refused before it can reach an image build"
1524            );
1525        }
1526    }
1527
1528    #[test]
1529    fn resource_type_is_stable() {
1530        assert_eq!(Sandbox::RESOURCE_TYPE.as_ref(), "sandbox");
1531    }
1532
1533    #[test]
1534    fn capability_sets_are_per_platform() {
1535        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1536        assert!(
1537            gcp.reconnect,
1538            "generation from the container boot id makes a sandbox reachable across processes"
1539        );
1540        assert!(!gcp.preview);
1541        assert!(gcp.enforced_limits);
1542
1543        let azure = SandboxCapabilities::for_platform(Platform::Azure).expect("azure is supported");
1544        assert!(azure.files, "every backend moves files");
1545        assert!(gcp.files);
1546        // Azure is the only backend whose egress policy matches on host pattern, and the only
1547        // one where `deny` and a hostname list are the same object.
1548        assert!(azure.domain_egress_rules);
1549        assert!(azure.egress_deny);
1550        // The data plane honours a continuous cpu/memory/disk surface and refuses anything
1551        // outside `n×250m`, with the rule in the message.
1552        assert!(azure.enforced_limits);
1553        assert!(azure.pause_resume);
1554        // Both stay false for reasons that are not "unbuilt": a snapshot id has nothing to
1555        // consume it on any backend, and an Azure port's auth is anonymous or a human allowlist,
1556        // neither of which is a port-scoped credential.
1557        assert!(!azure.snapshot);
1558        assert!(!azure.preview);
1559
1560        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1561        assert!(!aws.snapshot, "AWS has no user-callable sandbox snapshot");
1562        assert!(aws.pause_resume);
1563
1564        let k8s =
1565            SandboxCapabilities::for_platform(Platform::Kubernetes).expect("k8s is supported");
1566        assert!(
1567            !k8s.preview,
1568            "the sandbox-scoped ingress gateway does not exist yet"
1569        );
1570    }
1571
1572    /// Whether the process supervising a command is a separate identity from the command.
1573    ///
1574    /// Values are measured, not inferred. AWS: the agent runs as uid 0 with
1575    /// `CapEff: 00000000a80425fb` and `setuid`s the command to uid 60000, so the two differ.
1576    /// Kubernetes: the sandbox pod pins `run_as_user: 65534` on both pod and container with
1577    /// `capabilities.drop: [ALL]` and `allow_privilege_escalation: false`, so no uid split is
1578    /// possible (`kubernetes_spec.rs`). Local: `docker exec` runs as the workload uid while the
1579    /// manager supervises from the host. Azure and Agent Platform have no in-sandbox supervisor.
1580    #[test]
1581    fn supervisor_isolation_is_per_platform() {
1582        let value = |platform| {
1583            SandboxCapabilities::for_platform(platform)
1584                .expect("supported")
1585                .supervisor_isolation
1586        };
1587
1588        assert!(
1589            value(Platform::Aws),
1590            "root agent setuids the command to 60000"
1591        );
1592        assert!(
1593            value(Platform::Local),
1594            "the supervisor is on the host, outside the container"
1595        );
1596        assert!(
1597            !value(Platform::Kubernetes),
1598            "a single pinned uid cannot be split"
1599        );
1600        assert!(!value(Platform::Azure), "no Alien process runs the command");
1601        assert!(
1602            !value(Platform::Gcp),
1603            "no separate supervisor identity runs the command"
1604        );
1605    }
1606
1607    /// The point of the field: AWS and GCP report the *same* `supervisor_pid_namespace` (neither
1608    /// has `CAP_SYS_ADMIN`), so that axis alone reads them as equivalent. They are not — AWS
1609    /// separates the command's identity from the supervisor's and Agent Platform does not.
1610    #[test]
1611    fn supervisor_isolation_separates_aws_from_a_subprocess_backend() {
1612        let aws = SandboxCapabilities::for_platform(Platform::Aws).expect("aws is supported");
1613        let gcp = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1614
1615        assert_eq!(
1616            aws.supervisor_pid_namespace, gcp.supervisor_pid_namespace,
1617            "the older axis cannot tell them apart"
1618        );
1619        assert!(
1620            aws.supervisor_isolation,
1621            "AWS setuids the command off the supervisor"
1622        );
1623        assert!(
1624            !gcp.supervisor_isolation,
1625            "the command runs under no separate supervisor identity"
1626        );
1627    }
1628
1629    /// The Agent Platform row, each value against the behaviour it was measured from. `reconnect`
1630    /// is the tripwire: it is `true` only because `generation` is derived from the container boot
1631    /// id read through the agent's health op, so a caller detects a replaced container instead of
1632    /// reconnecting to a blank one. It is also the body of the `Platform::Gcp` arm, asserted below.
1633    #[test]
1634    fn gcp_agent_platform_row_matches_measured_backend() {
1635        let row = SandboxCapabilities::gcp_agent_platform();
1636
1637        assert!(row.files, "agent file ops move over the sandbox envelope");
1638        assert!(
1639            row.reconnect,
1640            "generation is derived from the container boot id, so a sandbox is reachable across \
1641             processes"
1642        );
1643        assert!(
1644            !row.preview,
1645            "the only ingress is :execute; no port-scoped capability"
1646        );
1647        assert!(
1648            !row.pause_resume,
1649            ":resume can return a fresh container, so a pause keeps no state"
1650        );
1651        assert!(
1652            !row.snapshot,
1653            "the create path never sends a snapshot, so none is reachable through the trait"
1654        );
1655        assert!(
1656            !row.domain_egress_rules,
1657            "VPC and DNS peering is not a hostname allowlist"
1658        );
1659        assert!(
1660            row.egress_deny,
1661            "a declared deny blocks both egress and DNS"
1662        );
1663        assert!(
1664            row.enforced_limits,
1665            "ceilings are enforced, by terminating the sandbox on breach"
1666        );
1667        assert!(!row.process_limit, "no process-count ceiling is observed");
1668        assert!(row.sandbox_lifetime, "ttl maps to a sandbox expireTime");
1669        assert!(!row.supervisor_pid_namespace, "no PID-namespace isolation");
1670        assert!(
1671            !row.supervisor_isolation,
1672            "the command is not run under a separate supervisor identity"
1673        );
1674
1675        // Agent Platform is the registered GCP backend, so the arm returns exactly this row.
1676        let live = SandboxCapabilities::for_platform(Platform::Gcp).expect("gcp is supported");
1677        assert_eq!(
1678            live, row,
1679            "the Platform::Gcp arm is the Agent Platform capability row"
1680        );
1681    }
1682
1683    #[test]
1684    fn platforms_without_a_backend_are_an_error_not_an_empty_set() {
1685        let error = SandboxCapabilities::for_platform(Platform::Machines)
1686            .expect_err("Machines has no sandbox backend");
1687        assert_eq!(error.code, "SANDBOX_PLATFORM_UNSUPPORTED");
1688    }
1689
1690    #[test]
1691    fn unsupported_capability_names_platform_and_capability() {
1692        let capabilities = SandboxCapabilities::for_platform(Platform::Gcp).expect("supported");
1693        let error = capabilities
1694            .require(SandboxCapability::Preview, Platform::Gcp)
1695            .expect_err("GCP has no preview");
1696
1697        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1698        let rendered = error.to_string();
1699        assert!(
1700            rendered.contains("preview"),
1701            "names the capability: {rendered}"
1702        );
1703        assert!(rendered.contains("gcp"), "names the platform: {rendered}");
1704    }
1705
1706    /// Azure matches on hostname; AWS and Kubernetes match CIDRs, and Local and GCP have a
1707    /// switch rather than a filter. Accepting a hostname list on those four would leave a stack
1708    /// reading as restricted while the sandbox reaches the whole internet.
1709    #[test]
1710    fn a_hostname_allowlist_is_refused_everywhere_it_would_be_approximated() {
1711        let sandbox = sandbox_with(
1712            SandboxEgress::AllowDomains {
1713                domains: vec!["example.com".to_string()],
1714            },
1715            vec![],
1716        );
1717
1718        for platform in [
1719            Platform::Aws,
1720            Platform::Gcp,
1721            Platform::Kubernetes,
1722            Platform::Local,
1723        ] {
1724            let error = sandbox
1725                .validate_for_platform(platform)
1726                .expect_err("only Azure expresses a hostname allowlist");
1727            assert_eq!(
1728                error.code, "SANDBOX_CAPABILITY_UNSUPPORTED",
1729                "on {platform:?}"
1730            );
1731        }
1732
1733        assert!(
1734            SandboxCapabilities::for_platform(Platform::Azure)
1735                .expect("supported")
1736                .domain_egress_rules,
1737            "Azure's egress policy matches on host pattern"
1738        );
1739    }
1740
1741    /// `deny` is the declaration that carries a security promise, so a backend that cannot keep
1742    /// it has to refuse rather than accept it and run the code with open egress.
1743    #[test]
1744    fn a_denied_egress_is_refused_where_it_would_not_be_enforced() {
1745        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1746
1747        // GCP is asserted at the capability rather than through validation: this sandbox declares
1748        // ceilings GCP cannot enforce, so it is refused for a reason unrelated to egress.
1749        assert!(
1750            SandboxCapabilities::for_platform(Platform::Gcp)
1751                .expect("supported")
1752                .egress_deny
1753        );
1754
1755        for platform in [Platform::Aws, Platform::Kubernetes, Platform::Local] {
1756            sandbox
1757                .validate_for_platform(platform)
1758                .expect("deny is enforced here");
1759        }
1760
1761        // Declares no ceilings, which Azure refuses for its own reason, so this isolates egress.
1762        let egress_only = Sandbox::new("sbx".to_string())
1763            .code(SandboxCode::Image {
1764                image: "alpine".to_string(),
1765            })
1766            .egress(SandboxEgress::Deny)
1767            .lifecycle(SandboxLifecyclePolicy {
1768                max_lifetime_seconds: None,
1769                idle_pause_seconds: None,
1770            })
1771            .build();
1772
1773        egress_only
1774            .validate_for_platform(Platform::Azure)
1775            .expect("Azure creates the sandbox under a Deny policy with full inspection");
1776    }
1777
1778    /// A sandbox naming no ceilings is valid on every platform and still resolves to a concrete
1779    /// set. The rule Azure applies to ceilings that *are* declared is pinned separately, by
1780    /// `azure_sizes_follow_the_rule_the_data_plane_states`.
1781    #[test]
1782    fn a_sandbox_declaring_no_ceilings_takes_the_platforms_own() {
1783        let undeclared = Sandbox::new("sbx".to_string())
1784            .code(SandboxCode::Image {
1785                image: "alpine".to_string(),
1786            })
1787            .egress(SandboxEgress::Deny)
1788            .lifecycle(SandboxLifecyclePolicy {
1789                max_lifetime_seconds: None,
1790                idle_pause_seconds: None,
1791            })
1792            .build();
1793
1794        undeclared
1795            .validate_for_platform(Platform::Azure)
1796            .expect("a sandbox naming no ceilings takes the platform's own");
1797
1798        // A backend still gets a concrete set, so nothing downstream has to invent one.
1799        assert_eq!(undeclared.resolved_limits().cpu, "1");
1800    }
1801
1802    /// Pins Azure's own sizing rule: `250m`, `1500m` and `4000m` are valid, `32000m` and `333m`
1803    /// are refused. Checked at plan time so a bad value is a declaration to fix, not a package
1804    /// that renders without error and dies at the first sandbox.
1805    #[test]
1806    fn azure_sizes_follow_the_rule_the_data_plane_states() {
1807        let sized = |cpu: &str, memory: &str, disk: &str| {
1808            let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1809            let limits = sandbox
1810                .limits
1811                .as_mut()
1812                .expect("the fixture declares limits");
1813            limits.cpu = cpu.to_string();
1814            limits.memory = memory.to_string();
1815            limits.disk = disk.to_string();
1816            sandbox.validate_for_platform(Platform::Azure)
1817        };
1818
1819        sized("250m", "512Mi", "5120Mi").expect("the smallest step the data plane accepts");
1820        sized("4000m", "8192Mi", "40960Mi").expect("cpu, memory and disk are all honoured");
1821        sized("16000m", "32Gi", "320Gi").expect("the top of the range");
1822
1823        // Same off-step case `azure_sandbox_limits` checks the multiple for, not just the range.
1824        let off_step = sized("333m", "512Mi", "5120Mi").expect_err("333m is not a step of 250m");
1825        assert_eq!(off_step.code, "SANDBOX_LIMIT_INVALID", "{off_step}");
1826        assert!(off_step.to_string().contains("cpu"), "{off_step}");
1827
1828        let too_big = sized("32000m", "64Gi", "640Gi").expect_err("32 cores is over the ceiling");
1829        assert_eq!(too_big.code, "SANDBOX_LIMIT_INVALID", "{too_big}");
1830
1831        // Both ceilings are derived from the cpu, so the same memory passes at one size and fails
1832        // at another - which is what makes them worth checking rather than bounding absolutely.
1833        sized("1000m", "2Gi", "20Gi").expect("2Gi is exactly one core's worth");
1834        let over_memory = sized("250m", "2Gi", "5120Mi").expect_err("2Gi needs a full core");
1835        assert_eq!(over_memory.code, "SANDBOX_LIMIT_INVALID", "{over_memory}");
1836        assert!(over_memory.to_string().contains("memory"), "{over_memory}");
1837
1838        let over_disk = sized("250m", "512Mi", "20Gi").expect_err("20Gi needs a full core");
1839        assert!(over_disk.to_string().contains("disk"), "{over_disk}");
1840    }
1841
1842    #[test]
1843    fn preview_ports_require_the_preview_capability() {
1844        let sandbox = sandbox_with(SandboxEgress::Deny, vec![8080]);
1845
1846        sandbox
1847            .validate_for_platform(Platform::Aws)
1848            .expect("AWS mints a port-scoped JWE");
1849
1850        let error = sandbox
1851            .validate_for_platform(Platform::Kubernetes)
1852            .expect_err("Kubernetes preview is deferred");
1853        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1854    }
1855
1856    /// A grant nothing reads is the silent no-op the capability contract exists to prevent, and
1857    /// here it is worse than useless: the reader would take it for a base image that needs
1858    /// authenticating while the platform pulls `code.image` itself.
1859    #[test]
1860    fn a_private_base_image_is_refused_off_aws() {
1861        let mut sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1862        sandbox.code = SandboxCode::Image {
1863            image: "s3://acme-artifacts/agents/bundle.zip".to_string(),
1864        };
1865        sandbox.private_base_image =
1866            Some("123456789012.dkr.ecr.{region}.amazonaws.com/acme:tag".to_string());
1867
1868        sandbox
1869            .validate_for_platform(Platform::Aws)
1870            .expect("AWS builds its image from a bundle, so a base image sits behind code.image");
1871
1872        for platform in [
1873            Platform::Gcp,
1874            Platform::Azure,
1875            Platform::Kubernetes,
1876            Platform::Local,
1877        ] {
1878            let error = sandbox
1879                .validate_for_platform(platform)
1880                .expect_err("a backend that builds no image must refuse a base image for one");
1881            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1882            assert!(
1883                error.to_string().contains("privateBaseImage"),
1884                "the refusal must name the field the user declared: {error}"
1885            );
1886        }
1887    }
1888
1889    #[test]
1890    fn gcp_accepts_a_sandbox_declaring_enforced_limits() {
1891        let sandbox = sandbox_with(SandboxEgress::Allow, vec![]);
1892        sandbox
1893            .validate_for_platform(Platform::Gcp)
1894            .expect("Agent Platform enforces declared ceilings, by terminating on breach");
1895    }
1896
1897    #[test]
1898    fn invalid_quantities_are_rejected_with_the_offending_field() {
1899        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1900        sandbox
1901            .limits
1902            .as_mut()
1903            .expect("the fixture declares limits")
1904            .memory = "2Gb".to_string();
1905
1906        let error = sandbox
1907            .validate_for_platform(Platform::Aws)
1908            .expect_err("Gb is not a valid suffix");
1909        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1910        assert!(error.to_string().contains("memory"));
1911
1912        sandbox
1913            .limits
1914            .as_mut()
1915            .expect("the fixture declares limits")
1916            .memory = "2Gi".to_string();
1917        sandbox
1918            .limits
1919            .as_mut()
1920            .expect("the fixture declares limits")
1921            .cpu = "0".to_string();
1922        let error = sandbox
1923            .validate_for_platform(Platform::Aws)
1924            .expect_err("zero cpu is not a ceiling");
1925        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1926    }
1927
1928    #[test]
1929    fn zero_max_processes_is_rejected() {
1930        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1931        sandbox
1932            .limits
1933            .as_mut()
1934            .expect("the fixture declares limits")
1935            .max_processes = Some(0);
1936
1937        let error = sandbox
1938            .validate_for_platform(Platform::Local)
1939            .expect_err("a sandbox must be able to run at least one process");
1940        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
1941        assert!(error.to_string().contains("maxProcesses"));
1942    }
1943
1944    /// A process ceiling needs a container runtime. Kubernetes sets one per node rather than per
1945    /// pod, and neither MicroVMs nor Azure sandboxes expose one, so accepting the declaration
1946    /// anywhere else would mean carrying a bound nothing applies.
1947    #[test]
1948    fn a_process_ceiling_is_accepted_only_where_a_runtime_can_apply_it() {
1949        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1950        sandbox
1951            .limits
1952            .as_mut()
1953            .expect("the fixture declares limits")
1954            .max_processes = Some(256);
1955
1956        sandbox
1957            .validate_for_platform(Platform::Local)
1958            .expect("Docker takes a pids limit");
1959
1960        for platform in [Platform::Aws, Platform::Azure, Platform::Kubernetes] {
1961            let error = sandbox
1962                .validate_for_platform(platform)
1963                .expect_err("a process ceiling nothing applies must be refused");
1964            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
1965        }
1966    }
1967
1968    /// Lambda rejects a run outside 1–28,800 rather than clamping it, so a value beyond that
1969    /// would pass planning, render into the package, and fail at the first sandbox. Kubernetes
1970    /// takes the same field with no such bound, so the check is AWS's alone.
1971    #[test]
1972    fn a_lifetime_aws_would_reject_is_refused_while_planning() {
1973        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1974
1975        for seconds in [0, 28_801, 100_000] {
1976            sandbox.lifecycle.max_lifetime_seconds = Some(seconds);
1977            let error = sandbox
1978                .validate_for_platform(Platform::Aws)
1979                .expect_err("a lifetime outside what AWS runs is refused");
1980            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "{seconds}s");
1981
1982            // Kubernetes has no such ceiling, so the same declaration is fine there.
1983            sandbox
1984                .validate_for_platform(Platform::Kubernetes)
1985                .expect("the kubelet takes any activeDeadlineSeconds");
1986        }
1987
1988        sandbox.lifecycle.max_lifetime_seconds = Some(28_800);
1989        sandbox
1990            .validate_for_platform(Platform::Aws)
1991            .expect("the ceiling itself is allowed");
1992    }
1993
1994    /// An image Azure can take neither as a catalog name nor as a registry image is refused while
1995    /// planning, not at the first sandbox.
1996    #[test]
1997    fn an_image_azure_cannot_pull_is_refused_while_planning() {
1998        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
1999        // Azure enforces no declared ceiling, so a sandbox carrying limits is refused before the
2000        // image is ever read.
2001        sandbox.limits = None;
2002
2003        // A digest too short to be one, blanks, a space and a query string.
2004        for image in ["ubuntu@sha256:abc", "", "   ", "ubuntu latest", "ubuntu?x"] {
2005            sandbox.code = SandboxCode::Image {
2006                image: image.to_string(),
2007            };
2008            let error = sandbox
2009                .validate_for_platform(Platform::Azure)
2010                .expect_err("an image Azure has nowhere to put is refused");
2011            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID", "image '{image}'");
2012        }
2013
2014        for image in ["ubuntu", "ubuntu-22.04", "debian_slim"] {
2015            sandbox.code = SandboxCode::Image {
2016                image: image.to_string(),
2017            };
2018            sandbox
2019                .validate_for_platform(Platform::Azure)
2020                .unwrap_or_else(|error| panic!("'{image}' is a catalog name: {error}"));
2021        }
2022
2023        for image in [
2024            "ubuntu:24.04",
2025            "ghcr.io/myorg/sandbox:latest",
2026            "docker.io/library/python:3.14-slim",
2027            "localhost:5000/team/agent@sha256:51dafde81dbdb6ebde285137a295cf18a47ca95234fe388a343719cb97305b3d",
2028        ] {
2029            sandbox.code = SandboxCode::Image {
2030                image: image.to_string(),
2031            };
2032            sandbox
2033                .validate_for_platform(Platform::Azure)
2034                .unwrap_or_else(|error| panic!("'{image}' is a registry image: {error}"));
2035        }
2036
2037        // Surrounding space is trimmed rather than carried into the create body.
2038        sandbox.code = SandboxCode::Image {
2039            image: " ubuntu ".to_string(),
2040        };
2041        assert_eq!(
2042            sandbox
2043                .azure_image()
2044                .expect("a padded name is still a name"),
2045            AzureSandboxImage::Catalog("ubuntu")
2046        );
2047    }
2048
2049    /// Every value the catalog allowlist `[A-Za-z0-9._-]+` accepts stays a catalog name, so a
2050    /// declaration that planned under it keeps its meaning. Exhaustive to three characters, then a
2051    /// fixed-seed sample of longer ones, some with surrounding space.
2052    #[test]
2053    fn every_value_the_catalog_allowlist_accepted_is_still_a_catalog_name() {
2054        const ALPHABET: &[u8] =
2055            b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789._-";
2056
2057        let assert_catalog = |value: &str| {
2058            assert_eq!(
2059                classify_azure_sandbox_image(value),
2060                Some(AzureSandboxImage::Catalog(value.trim())),
2061                "'{value}'"
2062            );
2063        };
2064
2065        for a in ALPHABET {
2066            assert_catalog(&String::from_utf8(vec![*a]).unwrap());
2067            for b in ALPHABET {
2068                assert_catalog(&String::from_utf8(vec![*a, *b]).unwrap());
2069                for c in ALPHABET {
2070                    assert_catalog(&String::from_utf8(vec![*a, *b, *c]).unwrap());
2071                }
2072            }
2073        }
2074
2075        let mut state: u64 = 0x9E37_79B9_7F4A_7C15;
2076        let mut next = || {
2077            state ^= state << 13;
2078            state ^= state >> 7;
2079            state ^= state << 17;
2080            state
2081        };
2082        for _ in 0..20_000 {
2083            let len = 4 + (next() % 60) as usize;
2084            let mut value: String = (0..len)
2085                .map(|_| ALPHABET[(next() % ALPHABET.len() as u64) as usize] as char)
2086                .collect();
2087            if next() % 4 == 0 {
2088                value = format!("  {value}\t");
2089            }
2090            assert_catalog(&value);
2091        }
2092    }
2093
2094    /// The label is how the provider finds the image the controller built, so both must derive
2095    /// the same one; it also has to fit a label value, which a raw reference does not.
2096    #[test]
2097    fn the_disk_image_label_is_stable_and_label_safe() {
2098        let label = azure_disk_image_label("docker.io/library/python:3.14-slim");
2099        assert_eq!(
2100            label,
2101            azure_disk_image_label(" docker.io/library/python:3.14-slim ")
2102        );
2103        assert_ne!(
2104            label,
2105            azure_disk_image_label("docker.io/library/python:3.13-slim")
2106        );
2107        assert_eq!(label.len(), 32);
2108        assert!(label.chars().all(|c| c.is_ascii_hexdigit()), "{label}");
2109    }
2110
2111    /// A deadline is accepted only where the platform itself terminates on it — the kubelet's
2112    /// `activeDeadlineSeconds` and Lambda's `maximumDurationInSeconds`. Everywhere else it would
2113    /// need a reaper that does not exist, so it is refused rather than accepted and dropped.
2114    #[test]
2115    fn a_sandbox_deadline_is_accepted_only_where_the_platform_applies_it() {
2116        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2117        sandbox.lifecycle.max_lifetime_seconds = Some(3600);
2118
2119        sandbox
2120            .validate_for_platform(Platform::Kubernetes)
2121            .expect("the kubelet enforces activeDeadlineSeconds");
2122        sandbox
2123            .validate_for_platform(Platform::Aws)
2124            .expect("Lambda terminates the MicroVM at maximumDurationInSeconds");
2125
2126        for platform in [Platform::Azure, Platform::Local] {
2127            let error = sandbox
2128                .validate_for_platform(platform)
2129                .expect_err("a deadline nothing applies must be refused");
2130            assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2131        }
2132    }
2133
2134    /// A MicroVM bursts to four times its baseline with no way to opt out, so a ceiling is kept
2135    /// by choosing the size whose *peak* fits inside it. Sizing by baseline would hand back a
2136    /// sandbox that can reach four times what the customer declared.
2137    #[test]
2138    fn an_aws_size_is_chosen_so_its_peak_stays_inside_the_declared_ceiling() {
2139        let sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2140        let tier = sandbox
2141            .microvm_tier()
2142            .expect("2Gi/1cpu/20Gi is satisfiable");
2143
2144        assert_eq!(
2145            tier.peak_memory_mib, 2048,
2146            "the peak is the declared ceiling"
2147        );
2148        assert_eq!(
2149            tier.baseline_memory_mib, 512,
2150            "which is a quarter of it as the baseline"
2151        );
2152        assert!(tier.max_disk_mib <= 20 * 1024);
2153    }
2154
2155    /// AWS allocates one vCPU per 2GB, so a cpu ceiling below what the memory ceiling implies
2156    /// cannot be honoured together with it. Letting cpu choose the size instead would hand back a
2157    /// machine four times smaller than the memory asked for, with nothing to indicate it.
2158    #[test]
2159    fn a_cpu_ceiling_below_what_the_memory_implies_is_refused_not_quietly_downsized() {
2160        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2161        {
2162            let limits = sandbox
2163                .limits
2164                .as_mut()
2165                .expect("the fixture declares limits");
2166            limits.cpu = "1".to_string();
2167            limits.memory = "8Gi".to_string();
2168        }
2169
2170        let error = sandbox
2171            .microvm_tier()
2172            .expect_err("1 cpu and 8Gi cannot both be ceilings on AWS");
2173        assert!(
2174            error.to_string().contains("4 vCPU"),
2175            "the refusal must say what the memory ceiling implies: {error}"
2176        );
2177
2178        sandbox
2179            .limits
2180            .as_mut()
2181            .expect("the fixture declares limits")
2182            .cpu = "4".to_string();
2183        let tier = sandbox.microvm_tier().expect("4 cpu matches 8Gi");
2184        assert_eq!(tier.peak_memory_mib, 8192);
2185    }
2186
2187    /// Below AWS's smallest peak there is no size that holds the ceiling, and rounding up to the
2188    /// nearest one would silently exceed it.
2189    #[test]
2190    fn an_aws_ceiling_smaller_than_any_size_is_refused_rather_than_rounded() {
2191        let mut sandbox = sandbox_with(SandboxEgress::Deny, vec![]);
2192        sandbox
2193            .limits
2194            .as_mut()
2195            .expect("the fixture declares limits")
2196            .memory = "1Gi".to_string();
2197
2198        let error = sandbox
2199            .validate_for_platform(Platform::Aws)
2200            .expect_err("no MicroVM size peaks at or below 1Gi");
2201        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2202        assert!(
2203            error.to_string().contains("2Gi"),
2204            "the refusal must say what the smallest holdable ceiling is: {error}"
2205        );
2206    }
2207
2208    /// `alien build` builds an AWS sandbox's base image, so source is a declaration there. On
2209    /// every other platform the image is pulled as declared, and an unbuilt source would schedule
2210    /// a pod that can never run, so the refusal still has to happen at plan time.
2211    #[test]
2212    fn source_code_is_refused_off_aws_rather_than_producing_a_broken_manifest() {
2213        let sandbox = Sandbox::new("agent".to_string())
2214            .code(SandboxCode::Source {
2215                src: "./sandbox".to_string(),
2216                toolchain: ToolchainConfig::Docker {
2217                    dockerfile: None,
2218                    build_args: None,
2219                    target: None,
2220                },
2221            })
2222            .egress(SandboxEgress::Deny)
2223            .lifecycle(SandboxLifecyclePolicy {
2224                max_lifetime_seconds: None,
2225                idle_pause_seconds: None,
2226            })
2227            .build();
2228
2229        sandbox
2230            .validate_for_platform(Platform::Aws)
2231            .expect("an AWS sandbox base image is built by `alien build`");
2232
2233        for platform in [
2234            Platform::Azure,
2235            Platform::Gcp,
2236            Platform::Kubernetes,
2237            Platform::Local,
2238        ] {
2239            let error = sandbox
2240                .validate_for_platform(platform)
2241                .expect_err("no backend builds a sandbox image from source here");
2242            assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2243            assert!(
2244                error.to_string().contains("code.image"),
2245                "the refusal must say what to write instead: {error}"
2246            );
2247            assert!(
2248                error.to_string().contains(&platform.to_string()),
2249                "the refusal must name the platform that cannot build it: {error}"
2250            );
2251        }
2252    }
2253
2254    /// `validate_quantity` accepts nine suffixes. Reading only `Gi` and `Mi` would size a
2255    /// declared `4G` as though it were `4Gi`, which for a ceiling means exceeding it.
2256    #[test]
2257    fn every_accepted_unit_converts_rather_than_falling_back() {
2258        assert_eq!(quantity_mib("2Gi"), Some(2048));
2259        assert_eq!(quantity_mib("512Mi"), Some(512));
2260        assert_eq!(quantity_mib("4G"), Some(3814));
2261        assert_eq!(quantity_mib("1Ti"), Some(1024 * 1024));
2262        assert_eq!(millicores("1"), Some(1000));
2263        assert_eq!(millicores("500m"), Some(500));
2264    }
2265
2266    #[test]
2267    fn unknown_fields_are_rejected() {
2268        let json = r#"{
2269            "id": "sbx",
2270            "code": {"type": "image", "image": "ubuntu:24.04"},
2271            "limits": {"cpu": "1", "memory": "2Gi", "disk": "20Gi"},
2272            "egress": {"mode": "deny"},
2273            "lifecycle": {},
2274            "unexpected": true
2275        }"#;
2276
2277        serde_json::from_str::<Sandbox>(json).expect_err("deny_unknown_fields must reject");
2278    }
2279
2280    #[test]
2281    fn serialization_roundtrips() {
2282        let sandbox = sandbox_with(
2283            SandboxEgress::AllowDomains {
2284                domains: vec!["example.com".to_string()],
2285            },
2286            vec![8080, 9090],
2287        );
2288
2289        let json = serde_json::to_string(&sandbox).expect("serializes");
2290        let restored: Sandbox = serde_json::from_str(&json).expect("deserializes");
2291        assert_eq!(sandbox, restored);
2292    }
2293
2294    #[test]
2295    fn id_is_immutable_across_updates() {
2296        let original = sandbox_with(SandboxEgress::Deny, vec![]);
2297        let renamed = Sandbox::new("other".to_string())
2298            .code(SandboxCode::Image {
2299                image: "ubuntu".to_string(),
2300            })
2301            .limits(
2302                original
2303                    .limits
2304                    .clone()
2305                    .expect("the fixture declares limits"),
2306            )
2307            .egress(SandboxEgress::Deny)
2308            .lifecycle(SandboxLifecyclePolicy {
2309                max_lifetime_seconds: None,
2310                idle_pause_seconds: None,
2311            })
2312            .build();
2313
2314        original
2315            .validate_update(&original.clone())
2316            .expect("an unchanged config is a valid update");
2317        original
2318            .validate_update(&renamed)
2319            .expect_err("renaming a sandbox is not an update");
2320    }
2321
2322    /// Azure declares an idle-pause policy but not a wall-clock ceiling.
2323    ///
2324    /// The two travel together in `SandboxLifecyclePolicy` and are gated separately on purpose:
2325    /// Azure pauses on idle and has no maximum lifetime, so accepting one and refusing the
2326    /// other is the honest split rather than an inconsistency.
2327    #[test]
2328    fn azure_takes_an_idle_policy_and_still_refuses_a_lifetime_ceiling() {
2329        let with_policy = |lifecycle: SandboxLifecyclePolicy| {
2330            Sandbox::new("sbx".to_string())
2331                .code(SandboxCode::Image {
2332                    image: "ubuntu".to_string(),
2333                })
2334                .egress(SandboxEgress::Allow)
2335                .lifecycle(lifecycle)
2336                .build()
2337                .validate_for_platform(Platform::Azure)
2338        };
2339
2340        with_policy(SandboxLifecyclePolicy {
2341            max_lifetime_seconds: None,
2342            idle_pause_seconds: Some(900),
2343        })
2344        .expect("Azure pauses a sandbox on idle");
2345
2346        let error = with_policy(SandboxLifecyclePolicy {
2347            max_lifetime_seconds: Some(3600),
2348            idle_pause_seconds: None,
2349        })
2350        .expect_err("Azure has no wall-clock ceiling to enforce one with");
2351        assert_eq!(error.code, "SANDBOX_CAPABILITY_UNSUPPORTED");
2352        assert!(
2353            error.message.contains("sandboxLifetime"),
2354            "names the capability: {}",
2355            error.message
2356        );
2357    }
2358
2359    /// An allowlist naming nothing is a deny-all wearing an allowlist's label.
2360    ///
2361    /// It renders as a `Deny` default with no rules — the shape the Azure provider adds a
2362    /// catch-all to avoid — and a reader scanning the declaration sees "allowDomains" and reads
2363    /// the opposite of what it does.
2364    #[test]
2365    fn an_allowlist_with_no_domains_is_refused() {
2366        let declared = |domains: Vec<String>| {
2367            Sandbox::new("sbx".to_string())
2368                .code(SandboxCode::Image {
2369                    image: "ubuntu".to_string(),
2370                })
2371                .egress(SandboxEgress::AllowDomains { domains })
2372                .lifecycle(SandboxLifecyclePolicy {
2373                    max_lifetime_seconds: None,
2374                    idle_pause_seconds: None,
2375                })
2376                .build()
2377                .validate_for_platform(Platform::Azure)
2378        };
2379
2380        let error = declared(vec![]).expect_err("an empty allowlist must be refused");
2381        assert_eq!(error.code, "SANDBOX_LIMIT_INVALID");
2382
2383        declared(vec!["api.example.com".to_string()])
2384            .expect("a named domain is what an allowlist is for");
2385    }
2386
2387    /// The two expressible modes map to the boolean; a host list maps to nothing so the caller has
2388    /// to refuse rather than silently pick a side.
2389    #[test]
2390    fn internet_access_switch_maps_only_the_two_expressible_modes() {
2391        assert_eq!(SandboxEgress::Allow.internet_access_switch(), Some(true));
2392        assert_eq!(SandboxEgress::Deny.internet_access_switch(), Some(false));
2393        assert_eq!(
2394            SandboxEgress::AllowDomains {
2395                domains: vec!["api.example.com".to_string()]
2396            }
2397            .internet_access_switch(),
2398            None,
2399            "a host list has no boolean and must not be approximated"
2400        );
2401    }
2402}