smolvm_protocol/lib.rs
1//! Protocol types for smolvm host-guest communication.
2//!
3//! This crate defines the wire protocol for vsock communication between
4//! the smolvm host and the guest agent (smolvm-agent).
5//!
6//! # Protocol Overview
7//!
8//! Communication uses JSON-encoded messages over vsock. Each message is
9//! prefixed with a 4-byte big-endian length header.
10//!
11//! ```text
12//! +----------------+-------------------+
13//! | Length (4 BE) | JSON payload |
14//! +----------------+-------------------+
15//! ```
16
17#![deny(missing_docs)]
18
19use serde::{Deserialize, Serialize};
20
21/// One S3-compatible bucket to mount inside the workload container.
22///
23/// Structured rather than a shell command: the agent mounts it natively, so
24/// nothing has to be installed in the image and no command is interpolated.
25#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
26pub struct S3Volume {
27 /// Service endpoint, e.g. `https://s3.us-east-1.amazonaws.com` or a
28 /// self-hosted `http://minio:9000`.
29 pub endpoint: String,
30 /// Region used for request signing.
31 pub region: String,
32 /// Bucket to mount.
33 pub bucket: String,
34 /// Optional key prefix, so a mount can expose one sub-tree of a bucket.
35 #[serde(default)]
36 pub prefix: String,
37 /// Absolute path inside the container where the bucket appears.
38 pub mountpoint: String,
39 /// Mount read-only; the kernel then rejects writes before they reach us.
40 #[serde(default)]
41 pub read_only: bool,
42 /// Access key. Absent (with the secret) means anonymous access.
43 #[serde(default, skip_serializing_if = "Option::is_none")]
44 pub access_key_id: Option<String>,
45 /// Secret key paired with `access_key_id`.
46 #[serde(default, skip_serializing_if = "Option::is_none")]
47 pub secret_access_key: Option<String>,
48 /// Session token, when temporary credentials are in use.
49 #[serde(default, skip_serializing_if = "Option::is_none")]
50 pub session_token: Option<String>,
51}
52
53pub mod credentials;
54pub mod forkpoint;
55pub mod guest_env;
56pub mod host_pattern;
57pub mod image_ref;
58pub mod intercept;
59pub mod publish_socket;
60pub mod retry;
61pub mod secrets;
62
63pub use credentials::{CredentialBinding, CredentialPolicy};
64pub use image_ref::{image_repo, normalize_image_ref};
65pub use intercept::InterceptEndpoint;
66pub use secrets::{SecretRef, SecretSourceKind};
67
68/// Serde helper for encoding `Vec<u8>` as a base64 string in JSON.
69///
70/// Without this, serde_json serializes `Vec<u8>` as a JSON array of numbers
71/// (e.g., `[104,101,108,108,111]`), which inflates binary data by ~4x.
72/// Base64 encoding reduces this to ~1.33x.
73pub mod base64_bytes {
74 use base64::{engine::general_purpose::STANDARD, Engine};
75 use serde::{Deserialize, Deserializer, Serializer};
76
77 /// Serialize `Vec<u8>` as a base64 string.
78 pub fn serialize<S: Serializer>(data: &[u8], serializer: S) -> Result<S::Ok, S::Error> {
79 serializer.serialize_str(&STANDARD.encode(data))
80 }
81
82 /// Deserialize a base64 string into `Vec<u8>`.
83 pub fn deserialize<'de, D: Deserializer<'de>>(deserializer: D) -> Result<Vec<u8>, D::Error> {
84 let s = String::deserialize(deserializer)?;
85 STANDARD.decode(&s).map_err(serde::de::Error::custom)
86 }
87}
88
89/// Protocol version.
90pub const PROTOCOL_VERSION: u32 = 1;
91
92/// The agent can freeze internal filesystems before acknowledging shutdown.
93pub const QUIESCED_SHUTDOWN_CAPABILITY: &str = "quiesced-shutdown-v1";
94
95/// virtiofs tag under which the host exposes the Rosetta 2 Linux runtime to the
96/// guest. Shared host↔guest so the launcher's `krun_add_virtiofs` tag and the
97/// guest agent's `mount -t virtiofs` source can't drift apart.
98pub const ROSETTA_TAG: &str = "rosetta";
99
100/// Guest mount point for the Rosetta 2 Linux runtime. The ptrace wrapper execs
101/// `<ROSETTA_GUEST_PATH>/rosetta` (the translator), so this path is baked into
102/// both the wrapper and the `binfmt_misc` registration.
103pub const ROSETTA_GUEST_PATH: &str = "/mnt/rosetta";
104
105/// Maximum frame size (32 MB - layer exports use chunked streaming).
106pub const MAX_FRAME_SIZE: u32 = 32 * 1024 * 1024;
107
108/// Chunk size for streaming layer data (~16 MB raw, ~21 MB as base64 JSON).
109pub const LAYER_CHUNK_SIZE: usize = 16 * 1024 * 1024;
110
111/// Files at or below this size are written with a single `FileWrite`
112/// message. Larger files must stream via
113/// `FileWriteBegin` + `FileWriteChunk` so no single frame approaches
114/// [`MAX_FRAME_SIZE`] (base64 + JSON inflation is ~1.4x).
115///
116/// Chosen to keep the single-shot frame comfortably under the frame
117/// limit while preserving the fast-path latency for small config
118/// files / scripts / keys.
119pub const FILE_WRITE_SINGLE_SHOT_MAX: usize = 1024 * 1024;
120
121/// Payload bytes per streaming upload chunk. Deliberately small —
122/// equal to [`FILE_WRITE_SINGLE_SHOT_MAX`] — so each chunk's encoded
123/// frame (~1.4 MB) fits inside typical kernel Unix-socket send
124/// buffers (`SO_SNDBUF` defaults on the order of 200–256 KiB but
125/// can grow). Larger chunks would force `write_all` to spin waiting
126/// for the agent to drain, and any latency spike trips the 10 s
127/// write timeout with `EAGAIN` — exactly the failure David
128/// reproduced before this fix landed.
129///
130/// Note: [`LAYER_CHUNK_SIZE`] is 16 MiB for agent→host (download)
131/// streaming, which works because the host side of the socket has
132/// more headroom than the guest side. Upload streaming is the
133/// asymmetric case and needs a smaller chunk.
134pub const FILE_WRITE_CHUNK_SIZE: usize = FILE_WRITE_SINGLE_SHOT_MAX;
135
136// The single-shot threshold must be <= the chunk size. They can be equal (a
137// 1 MiB file is a single shot; a 1 MiB + 1 byte file streams as two chunks),
138// but SINGLE_SHOT > CHUNK would be incoherent — a file slightly over the shot
139// threshold would have to stream as a single oversized chunk. Enforced at
140// compile time so a bad edit fails the build rather than a test.
141const _: () = assert!(FILE_WRITE_SINGLE_SHOT_MAX <= FILE_WRITE_CHUNK_SIZE);
142
143/// Hard ceiling on a single file transfer in either direction.
144///
145/// On the write path: enforced at `FileWriteBegin` by the agent —
146/// `total_size > FILE_TRANSFER_MAX_TOTAL` is rejected before any
147/// staging file is created.
148///
149/// On the read path: enforced by the host's `read_file` loop —
150/// after the first chunk that pushes the accumulated total past the
151/// cap, the call bails with an error and the partial buffer is
152/// dropped. This protects the host process from OOM if the guest
153/// (compromised or merely buggy) streams unbounded data.
154///
155/// 4 GiB matches the order-of-magnitude of the default overlay disk and the
156/// `gpu_vram_mib` cap. It is the compiled-in default. The host-side transfer
157/// paths (read/export, including `pack create --from-vm`) raise it at runtime
158/// via `SMOLVM_FILE_TRANSFER_MAX_BYTES` (see
159/// `agent::client::file_transfer_max_total`) so a VM snapshot whose overlay
160/// carries a large dependency tree (e.g. a torch + CUDA-wheels environment is
161/// ~5 GiB) can be packed without lowering the DoS bound for everyone else. The
162/// guest agent's write-path check runs inside the VM and cannot see that host
163/// env var, so it always enforces this const. Callers that need to move larger
164/// blobs routinely should stage via a virtiofs mount instead of `cp`.
165pub const FILE_TRANSFER_MAX_TOTAL: u64 = 4 * 1024 * 1024 * 1024;
166
167/// Filename of the virtiofs-visible marker the agent creates when it is
168/// ready to accept vsock connections.
169///
170/// The host polls for this file through its virtiofs mount of the guest
171/// rootfs. The agent writes it (and optionally a symlink from `/oldroot/`)
172/// during deferred init, just before opening the vsock listener.
173///
174/// Both sides must agree on this name; keeping it here prevents silent drift.
175pub const AGENT_READY_MARKER: &str = ".smolvm-ready";
176
177/// Well-known vsock ports.
178pub mod ports {
179 /// Control channel for workload VMs.
180 pub const WORKLOAD_CONTROL: u32 = 5000;
181 /// Log streaming from workload VMs.
182 pub const WORKLOAD_LOGS: u32 = 5001;
183 /// Agent control port (for OCI operations and management).
184 pub const AGENT_CONTROL: u32 = 6000;
185 /// SSH agent forwarding (host SSH_AUTH_SOCK bridged to guest).
186 pub const SSH_AGENT: u32 = 6001;
187 /// DNS filtering proxy (guest forwards DNS queries to host for filtering).
188 pub const DNS_FILTER: u32 = 6002;
189 /// Docker socket bridge: the guest listens on this vsock port and proxies
190 /// each connection to the in-guest `/var/run/docker.sock`, so the host can
191 /// reach the guest's Docker daemon over a host-side Unix socket
192 /// (`DOCKER_HOST=unix://…`). Inbound (host connects in), like the agent
193 /// control channel — unlike the outbound SSH/DNS/CUDA bridges.
194 pub const DOCKER: u32 = 6003;
195 /// Readiness doorbell: the guest connects OUT to this host port the instant it
196 /// finishes init, and the host's `accept()` fires — an event-driven readiness
197 /// signal that needs no writable filesystem, DAX window, or extra virtio
198 /// device. Outbound (guest connects in), like the SSH/DNS/CUDA bridges. The
199 /// primary readiness signal; the ready-marker file and the control-channel
200 /// ping remain as fallbacks. A fork clone resumes past the ring, so it never
201 /// dials this — clones detect readiness by pinging the restored agent.
202 pub const AGENT_READY: u32 = 6004;
203 /// CUDA-over-vsock (experimental): guest CUDA client forwards Driver-API
204 /// calls to a host CUDA server that runs them on the host NVIDIA GPU.
205 pub const CUDA: u32 = 7000;
206
207 /// Base vsock port for user-published Unix-socket bridges
208 /// (`--expose-socket` / `--mount-socket`). Each published socket is assigned
209 /// `PUBLISH_SOCKET_BASE + index`. Kept clear of the fixed ports above (and of
210 /// CUDA at 7000) so a reasonable number of sockets never collides.
211 pub const PUBLISH_SOCKET_BASE: u32 = 6100;
212
213 /// Maximum number of user-published sockets per VM. Bounds the vsock-port
214 /// window (`6100..6100+MAX`) below CUDA's 7000.
215 pub const PUBLISH_SOCKET_MAX: usize = 64;
216}
217
218/// vsock CID constants.
219pub mod cid {
220 /// Host CID (always 2).
221 pub const HOST: u32 = 2;
222 /// Guest CID (always 3 for the first/only guest).
223 pub const GUEST: u32 = 3;
224 /// Any CID (for listening).
225 pub const ANY: u32 = u32::MAX;
226}
227
228/// fsnotify event masks, mirroring the kernel's `FS_*` bits in
229/// `include/linux/fsnotify_backend.h`. Shared by the host watcher (which maps a
230/// host filesystem event to one of these) and the guest agent (which forwards
231/// the raw bits to `/proc/smolvm-fsnotify`). Only the subset relevant to
232/// file-watching tools is defined.
233pub mod fsnotify_mask {
234 /// File was modified.
235 pub const FS_MODIFY: u32 = 0x0000_0002;
236 /// Metadata changed (chmod/chown/utimes).
237 pub const FS_ATTRIB: u32 = 0x0000_0004;
238 /// Writable file was closed.
239 pub const FS_CLOSE_WRITE: u32 = 0x0000_0008;
240 /// File was moved away from the watched dir.
241 pub const FS_MOVED_FROM: u32 = 0x0000_0040;
242 /// File was moved into the watched dir.
243 pub const FS_MOVED_TO: u32 = 0x0000_0080;
244 /// Subfile was created.
245 pub const FS_CREATE: u32 = 0x0000_0100;
246 /// Subfile was deleted.
247 pub const FS_DELETE: u32 = 0x0000_0200;
248 /// Event occurred against a directory.
249 pub const FS_ISDIR: u32 = 0x4000_0000;
250}
251
252/// A single host-originated filesystem change to replay into the guest.
253#[derive(Debug, Clone, Serialize, Deserialize)]
254pub struct FsNotifyEvent {
255 /// Guest-side absolute path the event occurred on (virtiofs staging path).
256 pub path: String,
257 /// `fsnotify_mask::FS_*` bitmask for the event.
258 pub mask: u32,
259}
260
261// ============================================================================
262// Agent Protocol (OCI Operations)
263// ============================================================================
264
265fn shutdown_progress_disabled(progress: &bool) -> bool {
266 !progress
267}
268
269#[cfg(test)]
270mod shutdown_compat_tests {
271 use super::*;
272
273 #[test]
274 fn legacy_shutdown_wire_shape_is_unchanged() {
275 let wire = r#"{"method":"shutdown"}"#;
276 assert!(matches!(
277 serde_json::from_str::<AgentRequest>(wire).unwrap(),
278 AgentRequest::Shutdown { progress: false }
279 ));
280 assert_eq!(
281 serde_json::to_string(&AgentRequest::Shutdown { progress: false }).unwrap(),
282 wire
283 );
284 }
285
286 #[test]
287 fn older_agent_can_accept_opt_in_request() {
288 #[derive(Deserialize)]
289 #[serde(tag = "method", rename_all = "snake_case")]
290 enum LegacyRequest {
291 Shutdown,
292 }
293 let wire = serde_json::to_string(&AgentRequest::Shutdown { progress: true }).unwrap();
294 assert!(matches!(
295 serde_json::from_str::<LegacyRequest>(&wire).unwrap(),
296 LegacyRequest::Shutdown
297 ));
298 }
299
300 #[test]
301 fn traced_shutdown_preserves_progress_opt_in() {
302 let envelope = Envelope::with_trace_id(
303 AgentRequest::Shutdown { progress: true },
304 Some("shutdown-test".into()),
305 );
306 let decoded: Envelope<AgentRequest> =
307 decode_message(&encode_message(&envelope).unwrap()).unwrap();
308 assert!(matches!(
309 decoded.body,
310 AgentRequest::Shutdown { progress: true }
311 ));
312 assert_eq!(decoded.trace_id.as_deref(), Some("shutdown-test"));
313 }
314}
315
316/// Agent supports guarded online growth of mounted managed ext4 filesystems.
317pub const ONLINE_FILESYSTEM_GROWTH_CAPABILITY: &str = "online-filesystem-growth-v1";
318
319/// Agent can online and verify CPUs after the VMM creates them.
320pub const ONLINE_CPU_GROWTH_CAPABILITY: &str = "online-cpu-growth-v1";
321/// Guest can offline non-boot CPUs and verify the resulting online set.
322pub const OFFLINE_CPU_SHRINK_CAPABILITY: &str = "offline-cpu-shrink-v1";
323
324/// Agent can online and verify a RAM range already added by the VMM.
325pub const ONLINE_MEMORY_GROWTH_CAPABILITY: &str = "online-memory-growth-v1";
326
327/// Managed writable disk, never an arbitrary guest path.
328#[derive(Debug, Clone, Copy, Serialize, Deserialize)]
329#[serde(rename_all = "snake_case")]
330pub enum ManagedDisk {
331 /// Persistent workspace and image storage disk.
332 Storage,
333 /// Writable guest root filesystem overlay disk.
334 Overlay,
335}
336
337/// Agent request types (for image management and OCI operations).
338#[derive(Debug, Clone, Serialize, Deserialize)]
339#[serde(tag = "method", rename_all = "snake_case")]
340pub enum AgentRequest {
341 /// Offline trailing CPUs after the host verifies runtime shrink support.
342 OfflineCpus {
343 /// Retained online CPU count, starting at CPU0; must be nonzero.
344 target_count: u8,
345 },
346 /// Online existing guest memory blocks; never create or offline memory.
347 OnlineMemory {
348 /// Guest physical address of the first block.
349 start_address: u64,
350 /// Length of the existing, block-aligned range in bytes.
351 length_bytes: u64,
352 },
353 /// Online CPUs already created by the VMM, without offlining existing CPUs.
354 OnlineCpus {
355 /// Total CPU count, using consecutive CPU IDs starting at zero.
356 target_count: u8,
357 },
358 /// Grow a mounted filesystem after the VMM publishes its new capacity.
359 /// Never formats, repairs, unmounts, or shrinks the filesystem.
360 GrowFilesystem {
361 /// Managed disk whose mounted filesystem should grow.
362 disk: ManagedDisk,
363 /// Exact capacity in bytes already published by the VMM.
364 expected_bytes: u64,
365 },
366 /// Ping to check if agent is alive.
367 Ping,
368
369 /// Inject host-originated fsnotify events into the guest.
370 ///
371 /// virtiofs does not deliver host-side file changes to the guest as
372 /// fsnotify/inotify events, so inotify-based hot-reload (Vite, webpack,
373 /// nodemon) never fires when a mounted file is edited on the host. The host
374 /// watches the mount source and sends the resulting events here; the agent
375 /// writes them to `/proc/smolvm-fsnotify`, which fires the matching event on
376 /// the guest inode so watchers on the (bind-mounted) container path wake up.
377 /// Each `path` is a guest-side absolute path (the virtiofs staging path),
378 /// `mask` an `fsnotify_mask::FS_*` bitmask.
379 FsNotify {
380 /// Host-originated filesystem changes to replay as guest fsnotify events.
381 #[serde(default)]
382 events: Vec<FsNotifyEvent>,
383 },
384
385 /// Pull an OCI image and extract layers.
386 Pull {
387 /// Image reference (e.g., "alpine:latest", "docker.io/library/ubuntu:22.04").
388 image: String,
389 /// OCI platform to pull (e.g., "linux/arm64", "linux/amd64").
390 oci_platform: Option<String>,
391 /// Optional registry authentication credentials.
392 #[serde(default, skip_serializing_if = "Option::is_none")]
393 auth: Option<RegistryAuth>,
394 /// Proxy URL applied to the registry client (sets HTTP_PROXY and HTTPS_PROXY).
395 #[serde(default, skip_serializing_if = "Option::is_none")]
396 proxy: Option<String>,
397 /// Comma-separated NO_PROXY list of hosts/CIDRs that bypass the proxy.
398 #[serde(default, skip_serializing_if = "Option::is_none")]
399 no_proxy: Option<String>,
400 },
401
402 /// Query if an image exists locally.
403 Query {
404 /// Image reference.
405 image: String,
406 },
407
408 /// List all cached images.
409 ListImages,
410
411 /// Run garbage collection on unused layers.
412 GarbageCollect {
413 /// If true, only report what would be deleted.
414 dry_run: bool,
415 /// If true, delete all image manifests and configs first,
416 /// making all layers unreferenced so they get collected.
417 #[serde(default)]
418 purge_all: bool,
419 },
420
421 /// Prepare overlay rootfs for a workload.
422 PrepareOverlay {
423 /// Image reference.
424 image: String,
425 /// Unique workload ID for the overlay.
426 workload_id: String,
427 },
428
429 /// Clean up overlay rootfs for a workload.
430 CleanupOverlay {
431 /// Workload ID to clean up.
432 workload_id: String,
433 },
434
435 /// Format the storage disk (first-time setup).
436 FormatStorage,
437
438 /// Get storage disk status.
439 StorageStatus,
440
441 /// Report the guest's own view of machine memory, read from
442 /// `/proc/meminfo`.
443 ///
444 /// The host cannot answer this. macOS charges the VMM's `phys_footprint`
445 /// as `internal + compressed`, and the compressed part is counted at the
446 /// pages' *uncompressed* size, so an idle guest whose memory has been
447 /// compressed reports a footprint several times the bytes it actually
448 /// occupies. Only the guest's allocator knows what the machine is using.
449 MemoryStatus,
450
451 /// Test network connectivity directly from the agent (not via chroot).
452 /// Used to debug TSI networking.
453 NetworkTest {
454 /// URL to test (e.g., "http://1.1.1.1")
455 url: String,
456 },
457
458 /// Shutdown the agent.
459 Shutdown {
460 /// Opt into progress frames while storage is being synchronized.
461 #[serde(default, skip_serializing_if = "shutdown_progress_disabled")]
462 progress: bool,
463 },
464
465 /// Export a layer as a tar archive.
466 ///
467 /// Used by `smolvm pack` to extract OCI layers for packaging.
468 /// The agent streams the layer tar data back via LayerData responses.
469 ExportLayer {
470 /// Image digest (sha256:...).
471 image_digest: String,
472 /// Layer index (0-based).
473 layer_index: usize,
474 },
475
476 /// Merge a stack of directories into a single tar archive.
477 ///
478 /// `pack create --from-vm` ships the machine's image layers plus whatever the
479 /// container has written since as ONE flattened layer. The merge has to apply
480 /// whiteouts and opaque markers exactly as the runtime would, so it is done
481 /// with a read-only overlay mount rather than a file copy.
482 ///
483 /// The agent owns this rather than the host driving `mount(8)` over `VmExec`,
484 /// because `mount(8)` rejects a `lowerdir=` value beyond ~255 bytes — about
485 /// three OCI layer paths — while the agent can append each layer separately
486 /// through the same `fsconfig` path the runtime container mount uses.
487 FlattenLayers {
488 /// Directories to merge, bottom -> top. Entries that are missing or empty
489 /// are dropped, so a caller may include a container overlay's upper dir
490 /// without first checking whether the machine ever wrote to it.
491 lowerdirs: Vec<String>,
492 /// Guest path to write the tar archive to, or `None` to stream the
493 /// archive straight back as `DataChunk` responses.
494 ///
495 /// Streaming is what a large export wants. The flattened tar is as large
496 /// as the image it came from, so staging it in the guest means the disk
497 /// has to hold both the expanded rootfs and a second full copy of it as
498 /// an archive — the export sizes that disk by guessing, and a big enough
499 /// image runs it out of space.
500 #[serde(default)]
501 output: Option<String>,
502 },
503
504 /// Wait until the workload has declared a branchpoint, returning the ready
505 /// marker's contents (its profile lines) in `data.contents`.
506 BranchpointWait {
507 /// Give up after this many milliseconds.
508 timeout_ms: u64,
509 },
510 /// Put a negotiated helper into its restore-safe loop before capture.
511 BranchpointArm,
512 /// Return a parked source to its ordinary wait after capture.
513 BranchpointPark,
514 /// Release a restored clone: write the release marker for the generation
515 /// recorded in its ready marker, carrying the clone's identity, and wait
516 /// for the helper to acknowledge.
517 BranchpointRelease {
518 /// The clone's parameters in dotenv form, appended to the marker.
519 #[serde(default)]
520 env_dotenv: Option<String>,
521 },
522 /// Assign and release a held clone in one idempotent step: claim it with
523 /// `activation_token` (a retry with the same token completes a partial
524 /// commit; a different token is refused), install the per-clone
525 /// parameters, then write the release marker.
526 BranchpointActivate {
527 /// Per-clone parameters, dotenv form, written to `env_path`.
528 env_dotenv: String,
529 /// The same parameters in shell-sourceable form, written to `branch_env_path`.
530 env_sourceable: String,
531 /// Guest path of the dotenv file (under the clone's overlay for image machines).
532 env_path: String,
533 /// Guest path of the sourceable file.
534 branch_env_path: String,
535 /// Directory that must already exist, typically the clone's merged
536 /// overlay root; activation refuses rather than fabricating it.
537 require_dir: Option<String>,
538 /// Directory to create before writing the env files.
539 env_dir: String,
540 /// Token identifying this activation attempt.
541 activation_token: String,
542 },
543 /// Wait until a released clone's workload publishes its worker-ready token.
544 BranchpointWaitWorkerReady {
545 /// The token the workload must publish; any other is a mismatch.
546 token: String,
547 /// Give up after this many milliseconds.
548 timeout_ms: u64,
549 },
550
551 /// Execute a command directly in the VM (not in a container).
552 ///
553 /// This runs the command in the agent's Alpine rootfs without any
554 /// container isolation. Useful for VM-level operations and debugging.
555 VmExec {
556 /// Command and arguments.
557 command: Vec<String>,
558 /// Environment variables.
559 #[serde(default)]
560 env: Vec<(String, String)>,
561 /// Working directory in the VM.
562 workdir: Option<String>,
563 /// Timeout in milliseconds.
564 #[serde(default)]
565 timeout_ms: Option<u64>,
566 /// Interactive mode - stream I/O instead of buffering.
567 #[serde(default)]
568 interactive: bool,
569 /// Allocate a pseudo-TTY for the command.
570 #[serde(default)]
571 tty: bool,
572 /// Background mode - spawn and return PID immediately without waiting.
573 #[serde(default)]
574 background: bool,
575 /// Data to pipe to the command's stdin.
576 #[serde(default)]
577 stdin_data: Option<String>,
578 },
579
580 /// Run a command in an image's rootfs.
581 ///
582 /// This prepares an overlay, chroots into it, and executes the command.
583 /// Returns stdout, stderr, and exit code when the command completes.
584 Run {
585 /// Image reference (must be pulled first).
586 image: String,
587 /// Command and arguments.
588 command: Vec<String>,
589 /// Environment variables.
590 #[serde(default)]
591 env: Vec<(String, String)>,
592 /// Working directory inside the rootfs.
593 workdir: Option<String>,
594 /// User inside the rootfs. If omitted, the OCI image default applies.
595 #[serde(default, skip_serializing_if = "Option::is_none")]
596 user: Option<String>,
597 /// Volume mounts to bind into the container.
598 /// Each tuple is (virtiofs_tag, container_path, read_only).
599 #[serde(default)]
600 mounts: Vec<(String, String, bool)>,
601 /// Timeout in milliseconds. If the command exceeds this duration,
602 /// it will be killed and return exit code 124.
603 #[serde(default)]
604 timeout_ms: Option<u64>,
605 /// Interactive mode - stream I/O instead of buffering.
606 /// When true, output is streamed via Stdout/Stderr responses,
607 /// and stdin can be sent via the Stdin request.
608 #[serde(default)]
609 interactive: bool,
610 /// Allocate a pseudo-TTY for the command.
611 /// Enables terminal features like colors, line editing, and signal handling.
612 #[serde(default)]
613 tty: bool,
614 /// Detached mode — start the container and return immediately with the
615 /// container ID. Only meaningful when `persistent_overlay_id` is set.
616 /// Returns a `Completed` response with `stdout` containing the container ID.
617 #[serde(default)]
618 detached: bool,
619 /// Run the workload as an unprivileged container: restricted capabilities,
620 /// read-only cgroup, and no extra tmpfs. The default (false) is "VM-grade"
621 /// — since the microVM is the isolation boundary, the workload gets a full
622 /// capability set and the mounts an init system needs (so any image, incl.
623 /// systemd, boots). Opt in for defense-in-depth when running untrusted code.
624 #[serde(default)]
625 unprivileged: bool,
626 /// If set, use a persistent overlay that survives across exec sessions.
627 /// The overlay is identified by this ID (typically the machine name)
628 /// and reused on subsequent runs. If not set, an ephemeral overlay is
629 /// created and destroyed after the run.
630 #[serde(default, skip_serializing_if = "Option::is_none")]
631 persistent_overlay_id: Option<String>,
632 /// Data to pipe to the command's stdin (non-interactive runs only).
633 /// The pipe is closed after writing, so the command sees EOF.
634 #[serde(default, skip_serializing_if = "Option::is_none")]
635 stdin_data: Option<String>,
636 /// Spawn the container and return immediately with the crun PID.
637 /// The container runs detached; stdout/stderr go to /dev/null.
638 /// Incompatible with `interactive` and `tty`.
639 #[serde(default)]
640 background: bool,
641 /// S3 volumes to mount into the workload container. The agent mounts
642 /// them between `crun create` and `crun start`, so the workload's first
643 /// instruction already sees them and its command is never rewritten.
644 #[serde(default, skip_serializing_if = "Vec::is_empty")]
645 s3_volumes: Vec<S3Volume>,
646 },
647
648 /// Send stdin data to a running interactive command.
649 Stdin {
650 /// Input data to send to the command's stdin.
651 #[serde(with = "base64_bytes")]
652 data: Vec<u8>,
653 },
654
655 /// Resize the PTY window (for TTY mode).
656 Resize {
657 /// New width in columns.
658 cols: u16,
659 /// New height in rows.
660 rows: u16,
661 },
662
663 // ========================================================================
664 // File I/O
665 // ========================================================================
666 /// Write a file inside the VM in a single message.
667 ///
668 /// Use only for files up to [`FILE_WRITE_SINGLE_SHOT_MAX`]. Larger
669 /// files must stream via [`Self::FileWriteBegin`] +
670 /// [`Self::FileWriteChunk`] to avoid exceeding [`MAX_FRAME_SIZE`]
671 /// after base64 + JSON inflation.
672 FileWrite {
673 /// Absolute path in the VM filesystem.
674 path: String,
675 /// File contents.
676 #[serde(with = "base64_bytes")]
677 data: Vec<u8>,
678 /// File mode (e.g., 0o644). None = default (0644).
679 #[serde(default)]
680 mode: Option<u32>,
681 /// Owner uid to apply after the write. None = leave as written (root).
682 #[serde(default)]
683 uid: Option<u32>,
684 /// Owner gid to apply after the write. None = leave as written (root).
685 #[serde(default)]
686 gid: Option<u32>,
687 },
688
689 /// Open a streaming file upload session on this connection.
690 ///
691 /// Must be followed by one or more [`Self::FileWriteChunk`]
692 /// requests. The final chunk sets `done: true` to finalize.
693 /// Dropping the connection (or sending any non-chunk request)
694 /// before `done` aborts the session and leaves no partial file
695 /// at `path`.
696 ///
697 /// Sessions are per-connection — one session at a time.
698 FileWriteBegin {
699 /// Absolute path in the VM filesystem.
700 path: String,
701 /// File mode (e.g., 0o644). None = default (0644).
702 #[serde(default)]
703 mode: Option<u32>,
704 /// Owner uid to apply on finalize. None = leave as written (root).
705 #[serde(default)]
706 uid: Option<u32>,
707 /// Owner gid to apply on finalize. None = leave as written (root).
708 #[serde(default)]
709 gid: Option<u32>,
710 /// Expected total size in bytes. Rejected if it exceeds
711 /// [`FILE_TRANSFER_MAX_TOTAL`]. The agent uses this for an
712 /// early-fail check only; the actual size written is the sum
713 /// of chunk byte lengths.
714 total_size: u64,
715 },
716
717 /// Append a chunk to the currently open streaming upload.
718 /// If `done` is true, the agent fsyncs and atomically renames the
719 /// staging file onto the target path.
720 FileWriteChunk {
721 /// Chunk bytes. Typically [`FILE_WRITE_CHUNK_SIZE`] except
722 /// for the last chunk.
723 #[serde(with = "base64_bytes")]
724 data: Vec<u8>,
725 /// True on the final chunk; closes and renames the staging
726 /// file. False on intermediate chunks.
727 done: bool,
728 },
729
730 /// Read a file from the VM.
731 FileRead {
732 /// Absolute path in the VM filesystem.
733 path: String,
734 },
735
736 /// List the entries of a guest directory.
737 ///
738 /// Kept separate from [`AgentRequest::FileRead`], which streams bytes: a
739 /// listing is small and structured, and a caller that has to discover what
740 /// exists would otherwise guess names and read a 404 for each miss.
741 ListDirectory {
742 /// Absolute directory path in the VM filesystem.
743 path: String,
744 },
745
746 /// Stream a tar archive of a guest directory without creating a guest-side
747 /// temporary file. Used by staged mounts to batch many small files across
748 /// vsock instead of paying one virtiofs round trip per file.
749 ArchiveDirectory {
750 /// Absolute directory path in the VM filesystem.
751 path: String,
752 },
753
754 /// Create (without starting) a Kubernetes pod container whose rootfs is a
755 /// virtiofs-shared host directory (containerd snapshotter output) and whose
756 /// process definition comes from the host's OCI config. The agent builds
757 /// its crun bundle around the shared rootfs; nothing runs until
758 /// `PodStart`. Part of the containerd shim v2 datapath
759 /// (docs/kubernetes-runtime.md).
760 PodCreate {
761 /// Container ID (containerd task id).
762 id: String,
763 /// Rootfs path relative to the sandbox's shared virtiofs mount. The
764 /// shim boots the sandbox VM with ONE shared dir and bind-mounts each
765 /// container's rootfs under it (virtiofs shares are fixed at boot, but
766 /// pod containers are created afterwards), so the guest resolves this
767 /// as `<sandbox-share-mount>/<rootfs_rel>`.
768 rootfs_rel: String,
769 /// The host OCI runtime spec (config.json bytes). The agent extracts
770 /// process/env/cwd/user/mounts/resources and grafts them onto its own
771 /// guest bundle template; host-specific namespaces/paths are ignored.
772 spec_json: String,
773 /// Allocate a PTY for the init process.
774 #[serde(default)]
775 tty: bool,
776 },
777
778 /// Start a pod container created by `PodCreate` (or an exec process
779 /// registered by `PodExec`), streaming its I/O on THIS connection:
780 /// `Started` → `Stdout`/`Stderr`... → `Exited`. Stdin arrives via `Stdin`
781 /// requests; PTY resize via `Resize`.
782 PodStart {
783 /// Container ID.
784 id: String,
785 /// Exec process to start instead of the init process.
786 #[serde(default, skip_serializing_if = "Option::is_none")]
787 exec_id: Option<String>,
788 },
789
790 /// Register an exec process for a running pod container. Started later by
791 /// `PodStart { exec_id }`.
792 PodExec {
793 /// Container ID.
794 id: String,
795 /// Exec process ID (unique within the container).
796 exec_id: String,
797 /// OCI Process JSON (containerd's ExecProcessRequest spec).
798 process_json: String,
799 /// Allocate a PTY for the exec process.
800 #[serde(default)]
801 tty: bool,
802 },
803
804 /// Signal a pod container's init process (or one exec process).
805 PodSignal {
806 /// Container ID.
807 id: String,
808 /// Exec process to signal instead of init.
809 #[serde(default, skip_serializing_if = "Option::is_none")]
810 exec_id: Option<String>,
811 /// Signal number (SIGKILL = 9, SIGTERM = 15, ...).
812 signal: u32,
813 /// Signal the whole container process group.
814 #[serde(default)]
815 all: bool,
816 },
817
818 /// List PIDs inside a pod container (guest view).
819 PodPids {
820 /// Container ID.
821 id: String,
822 },
823
824 /// Sample a pod container's resource usage (guest view). The agent reads the
825 /// container's process tree from /proc (there is no per-container cgroup); the
826 /// shim maps the reply into containerd's cgroups metrics for CRI stats.
827 PodStats {
828 /// Container ID.
829 id: String,
830 },
831
832 /// Remove a pod container's (or exec process's) guest resources after
833 /// exit: bundle, cgroup, PTY. Exit status was already streamed by
834 /// `PodStart`'s `Exited`.
835 PodDelete {
836 /// Container ID.
837 id: String,
838 /// Exec process to remove instead of the whole container.
839 #[serde(default, skip_serializing_if = "Option::is_none")]
840 exec_id: Option<String>,
841 },
842}
843
844impl AgentRequest {
845 /// A log-safe one-line summary of the request.
846 ///
847 /// This string is written to the machine's console log, which is exposed
848 /// over the logs API — so it must NEVER include credential- or data-bearing
849 /// fields: registry `auth`, `env` (which can carry host-resolved secrets),
850 /// `proxy` (may embed credentials), or `data` (file/stdin bytes). Only the
851 /// variant name plus a non-secret identifier (image) is emitted.
852 ///
853 /// The match is exhaustive with no catch-all on purpose: adding a new
854 /// variant forces a compile error here, so redaction is a deliberate
855 /// decision rather than an accidental leak in some future request type.
856 pub fn log_summary(&self) -> String {
857 match self {
858 AgentRequest::OnlineMemory {
859 start_address,
860 length_bytes,
861 } => {
862 format!("OnlineMemory {{ start_address: {start_address}, length_bytes: {length_bytes} }}")
863 }
864 AgentRequest::OnlineCpus { target_count } => {
865 format!("OnlineCpus {{ target_count: {target_count} }}")
866 }
867 AgentRequest::OfflineCpus { target_count } => {
868 format!("OfflineCpus {{ target_count: {target_count} }}")
869 }
870 AgentRequest::GrowFilesystem {
871 disk,
872 expected_bytes,
873 } => {
874 format!("GrowFilesystem {{ disk: {disk:?}, expected_bytes: {expected_bytes} }}")
875 }
876 AgentRequest::Ping => "Ping".into(),
877 AgentRequest::FsNotify { events } => format!("FsNotify {{ count: {} }}", events.len()),
878 AgentRequest::Pull { image, .. } => format!("Pull {{ image: {image} }}"),
879 AgentRequest::Query { image, .. } => format!("Query {{ image: {image} }}"),
880 AgentRequest::ListImages => "ListImages".into(),
881 AgentRequest::GarbageCollect { .. } => "GarbageCollect".into(),
882 AgentRequest::PrepareOverlay { .. } => "PrepareOverlay".into(),
883 AgentRequest::CleanupOverlay { .. } => "CleanupOverlay".into(),
884 AgentRequest::FormatStorage => "FormatStorage".into(),
885 AgentRequest::StorageStatus => "StorageStatus".into(),
886 AgentRequest::MemoryStatus => "MemoryStatus".into(),
887 AgentRequest::NetworkTest { .. } => "NetworkTest".into(),
888 AgentRequest::Shutdown { .. } => "Shutdown".into(),
889 AgentRequest::ExportLayer { .. } => "ExportLayer".into(),
890 AgentRequest::FlattenLayers { lowerdirs, .. } => {
891 format!("FlattenLayers {{ count: {} }}", lowerdirs.len())
892 }
893 AgentRequest::VmExec { .. } => "VmExec".into(),
894 AgentRequest::BranchpointWait { .. } => "BranchpointWait".into(),
895 AgentRequest::BranchpointArm => "BranchpointArm".into(),
896 AgentRequest::BranchpointPark => "BranchpointPark".into(),
897 AgentRequest::BranchpointRelease { .. } => "BranchpointRelease".into(),
898 AgentRequest::BranchpointActivate { .. } => "BranchpointActivate".into(),
899 AgentRequest::BranchpointWaitWorkerReady { .. } => "BranchpointWaitWorkerReady".into(),
900 AgentRequest::Run { image, .. } => format!("Run {{ image: {image} }}"),
901 AgentRequest::Stdin { .. } => "Stdin".into(),
902 AgentRequest::Resize { .. } => "Resize".into(),
903 AgentRequest::FileWrite { .. } => "FileWrite".into(),
904 AgentRequest::FileWriteBegin { .. } => "FileWriteBegin".into(),
905 AgentRequest::FileWriteChunk { .. } => "FileWriteChunk".into(),
906 AgentRequest::FileRead { .. } => "FileRead".into(),
907 AgentRequest::ListDirectory { .. } => "ListDirectory".into(),
908 AgentRequest::ArchiveDirectory { .. } => "ArchiveDirectory".into(),
909 // Pod requests: spec/process JSON may carry env secrets — emit ids only.
910 AgentRequest::PodCreate { id, .. } => format!("PodCreate {{ id: {id} }}"),
911 AgentRequest::PodStart { id, exec_id } => match exec_id {
912 Some(e) => format!("PodStart {{ id: {id}, exec: {e} }}"),
913 None => format!("PodStart {{ id: {id} }}"),
914 },
915 AgentRequest::PodExec { id, exec_id, .. } => {
916 format!("PodExec {{ id: {id}, exec: {exec_id} }}")
917 }
918 AgentRequest::PodSignal {
919 id, signal, all, ..
920 } => format!("PodSignal {{ id: {id}, signal: {signal}, all: {all} }}"),
921 AgentRequest::PodPids { id } => format!("PodPids {{ id: {id} }}"),
922 AgentRequest::PodStats { id } => format!("PodStats {{ id: {id} }}"),
923 AgentRequest::PodDelete { id, exec_id } => match exec_id {
924 Some(e) => format!("PodDelete {{ id: {id}, exec: {e} }}"),
925 None => format!("PodDelete {{ id: {id} }}"),
926 },
927 }
928 }
929}
930
931/// One entry of a [`AgentRequest::ListDirectory`] result.
932///
933/// Deliberately small: a name, what it is, and a size. A caller discovering a
934/// tree wants to know where to recurse and what is worth downloading, and
935/// anything richer invites the listing being used as a stat cache.
936#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
937pub struct DirectoryEntry {
938 /// Entry name, not a path: no separators, relative to the directory listed.
939 pub name: String,
940 /// `"file"`, `"dir"`, `"symlink"`, or `"other"` for sockets, devices and
941 /// fifos, which a caller can see but should not try to download.
942 pub kind: String,
943 /// Size in bytes. Zero for anything that is not a regular file.
944 #[serde(default)]
945 pub size: u64,
946}
947
948/// Agent response types.
949#[derive(Debug, Clone, Serialize, Deserialize)]
950#[serde(tag = "status", rename_all = "snake_case")]
951pub enum AgentResponse {
952 /// Operation completed successfully.
953 Ok {
954 /// Response data (varies by request type).
955 #[serde(default, skip_serializing_if = "Option::is_none")]
956 data: Option<serde_json::Value>,
957 },
958
959 /// Pong response to ping.
960 Pong {
961 /// Protocol version.
962 version: u32,
963 /// Optional agent features that can evolve independently of the base protocol.
964 #[serde(default, skip_serializing_if = "Vec::is_empty")]
965 capabilities: Vec<String>,
966 },
967
968 /// Progress update (for long operations like pull).
969 Progress {
970 /// Human-readable message.
971 message: String,
972 /// Completion percentage (0-100).
973 #[serde(default, skip_serializing_if = "Option::is_none")]
974 percent: Option<u8>,
975 /// Current layer being processed.
976 #[serde(default, skip_serializing_if = "Option::is_none")]
977 layer: Option<String>,
978 },
979
980 /// Operation failed.
981 Error {
982 /// Error message.
983 message: String,
984 /// Error code (for programmatic handling).
985 #[serde(default, skip_serializing_if = "Option::is_none")]
986 code: Option<String>,
987 },
988
989 /// Command execution completed (non-interactive mode).
990 Completed {
991 /// Exit code from the command.
992 exit_code: i32,
993 /// Standard output (may be truncated). `Vec<u8>` preserves binary
994 /// output (image bytes, tarballs, etc.) that would be truncated by
995 /// `String` at the first non-UTF-8 byte. Serialized as base64 JSON
996 /// string — the same format as the streaming `Stdout` variant.
997 #[serde(with = "base64_bytes")]
998 stdout: Vec<u8>,
999 /// Standard error (may be truncated).
1000 #[serde(with = "base64_bytes")]
1001 stderr: Vec<u8>,
1002 },
1003
1004 /// Command started (interactive mode).
1005 /// Indicates the command is running and ready to receive stdin.
1006 Started,
1007
1008 /// Stdout data from a running command (interactive mode).
1009 Stdout {
1010 /// Output data.
1011 #[serde(with = "base64_bytes")]
1012 data: Vec<u8>,
1013 },
1014
1015 /// Stderr data from a running command (interactive mode).
1016 Stderr {
1017 /// Error output data.
1018 #[serde(with = "base64_bytes")]
1019 data: Vec<u8>,
1020 },
1021
1022 /// Command exited (interactive mode).
1023 Exited {
1024 /// Exit code from the command.
1025 exit_code: i32,
1026 /// The container was terminated by the cgroup OOM killer. The shim
1027 /// turns this into a TaskOOM event so the CRI reports
1028 /// `reason=OOMKilled`. Only ever set on a pod container's init exit.
1029 #[serde(default)]
1030 oom: bool,
1031 },
1032
1033 /// PIDs inside a pod container (`PodPids` reply).
1034 Pids {
1035 /// Guest PIDs, container-init first when known.
1036 pids: Vec<u32>,
1037 },
1038
1039 /// Resource usage sample for a pod container (`PodStats` reply). Summed over
1040 /// the container's process tree read from /proc (no per-container cgroup).
1041 Stats {
1042 /// Cumulative CPU time of the process tree, in nanoseconds.
1043 cpu_usage_ns: u64,
1044 /// Resident memory of the process tree, in bytes.
1045 memory_bytes: u64,
1046 },
1047
1048 /// Streaming binary-data chunk.
1049 ///
1050 /// Used by every streaming download direction: the agent sends
1051 /// one or more `DataChunk` responses in sequence, with `done: true`
1052 /// on the final chunk. Current producers: `ExportLayer` and
1053 /// `FileRead`.
1054 ///
1055 /// Payload size per chunk should stay under
1056 /// [`LAYER_CHUNK_SIZE`] so the encoded frame (~1.33× after
1057 /// base64) fits inside [`MAX_FRAME_SIZE`] with JSON overhead to
1058 /// spare.
1059 DataChunk {
1060 /// Chunk bytes. Empty allowed on the final frame (common for
1061 /// EOF-on-clean-boundary cases).
1062 #[serde(with = "base64_bytes")]
1063 data: Vec<u8>,
1064 /// True on the final chunk of the stream.
1065 done: bool,
1066 },
1067}
1068
1069// ============================================================================
1070// Error Code Constants
1071// ============================================================================
1072//
1073// Standard error codes for AgentResponse::Error. Using constants ensures
1074// consistency across the codebase and makes error handling more reliable.
1075
1076/// Error codes for agent responses.
1077pub mod error_codes {
1078 /// Request payload was invalid or malformed.
1079 pub const INVALID_REQUEST: &str = "INVALID_REQUEST";
1080 /// Requested resource was not found.
1081 pub const NOT_FOUND: &str = "NOT_FOUND";
1082 /// Internal error during operation.
1083 pub const INTERNAL_ERROR: &str = "INTERNAL_ERROR";
1084 /// Image pull operation failed.
1085 pub const PULL_FAILED: &str = "PULL_FAILED";
1086 /// Image query operation failed.
1087 pub const QUERY_FAILED: &str = "QUERY_FAILED";
1088 /// Command execution failed.
1089 pub const RUN_FAILED: &str = "RUN_FAILED";
1090 /// Command execution failed in container.
1091 pub const EXEC_FAILED: &str = "EXEC_FAILED";
1092 /// Process spawn failed.
1093 pub const SPAWN_FAILED: &str = "SPAWN_FAILED";
1094 /// Mount operation failed.
1095 pub const MOUNT_FAILED: &str = "MOUNT_FAILED";
1096 /// File I/O operation failed.
1097 pub const FILE_IO_FAILED: &str = "FILE_IO_FAILED";
1098 /// Overlay filesystem operation failed.
1099 pub const OVERLAY_FAILED: &str = "OVERLAY_FAILED";
1100 /// Cleanup operation failed.
1101 pub const CLEANUP_FAILED: &str = "CLEANUP_FAILED";
1102 /// Storage format operation failed.
1103 pub const FORMAT_FAILED: &str = "FORMAT_FAILED";
1104 /// Storage status query failed.
1105 pub const STATUS_FAILED: &str = "STATUS_FAILED";
1106 /// List operation failed.
1107 pub const LIST_FAILED: &str = "LIST_FAILED";
1108 /// Garbage collection failed.
1109 pub const GC_FAILED: &str = "GC_FAILED";
1110 /// Container creation failed.
1111 pub const CREATE_FAILED: &str = "CREATE_FAILED";
1112 /// Container start failed.
1113 pub const START_FAILED: &str = "START_FAILED";
1114 /// Container stop failed.
1115 pub const STOP_FAILED: &str = "STOP_FAILED";
1116 /// Container delete failed.
1117 pub const DELETE_FAILED: &str = "DELETE_FAILED";
1118 /// Export operation failed.
1119 pub const EXPORT_FAILED: &str = "EXPORT_FAILED";
1120 /// Serialization error.
1121 pub const SERIALIZATION_ERROR: &str = "SERIALIZATION_ERROR";
1122 /// Message size exceeds maximum.
1123 pub const MESSAGE_TOO_LARGE: &str = "MESSAGE_TOO_LARGE";
1124 /// Process wait operation failed.
1125 pub const WAIT_FAILED: &str = "WAIT_FAILED";
1126}
1127
1128impl AgentResponse {
1129 /// Create an error response with the given message and code.
1130 ///
1131 /// # Example
1132 ///
1133 /// ```
1134 /// use smolvm_protocol::{AgentResponse, error_codes};
1135 ///
1136 /// let response = AgentResponse::error("image not found", error_codes::NOT_FOUND);
1137 /// ```
1138 pub fn error(message: impl Into<String>, code: &str) -> Self {
1139 AgentResponse::Error {
1140 message: message.into(),
1141 code: Some(code.to_string()),
1142 }
1143 }
1144
1145 /// Create an error response from a Result's error, with the given code.
1146 ///
1147 /// # Example
1148 ///
1149 /// ```ignore
1150 /// let response = some_operation()
1151 /// .map(|data| AgentResponse::ok_with_data(data))
1152 /// .unwrap_or_else(|e| AgentResponse::from_err(e, error_codes::PULL_FAILED));
1153 /// ```
1154 pub fn from_err<E: std::fmt::Display>(err: E, code: &str) -> Self {
1155 AgentResponse::Error {
1156 message: err.to_string(),
1157 code: Some(code.to_string()),
1158 }
1159 }
1160
1161 /// Create an Ok response with optional JSON data.
1162 pub fn ok(data: Option<serde_json::Value>) -> Self {
1163 AgentResponse::Ok { data }
1164 }
1165
1166 /// Create an Ok response with JSON-serializable data.
1167 ///
1168 /// Returns an error response if serialization fails.
1169 pub fn ok_with_data<T: serde::Serialize>(data: T) -> Self {
1170 match serde_json::to_value(data) {
1171 Ok(value) => AgentResponse::Ok { data: Some(value) },
1172 Err(e) => AgentResponse::error(
1173 format!("failed to serialize response: {}", e),
1174 error_codes::SERIALIZATION_ERROR,
1175 ),
1176 }
1177 }
1178
1179 /// Convert a Result into an AgentResponse.
1180 ///
1181 /// On success, serializes the value to JSON. On error, creates an error response.
1182 ///
1183 /// # Example
1184 ///
1185 /// ```ignore
1186 /// let response = AgentResponse::from_result(
1187 /// storage::pull_image(image),
1188 /// error_codes::PULL_FAILED,
1189 /// );
1190 /// ```
1191 pub fn from_result<T, E>(result: Result<T, E>, error_code: &str) -> Self
1192 where
1193 T: serde::Serialize,
1194 E: std::fmt::Display,
1195 {
1196 match result {
1197 Ok(data) => Self::ok_with_data(data),
1198 Err(e) => Self::from_err(e, error_code),
1199 }
1200 }
1201}
1202
1203/// Image information returned by Query/ListImages.
1204#[derive(Debug, Clone, Serialize, Deserialize)]
1205pub struct ImageInfo {
1206 /// Image reference.
1207 pub reference: String,
1208 /// Image digest (sha256:...).
1209 pub digest: String,
1210 /// Image size in bytes.
1211 pub size: u64,
1212 /// Creation timestamp (ISO 8601).
1213 pub created: Option<String>,
1214 /// Platform architecture.
1215 pub architecture: String,
1216 /// Platform OS.
1217 pub os: String,
1218 /// Number of layers.
1219 pub layer_count: usize,
1220 /// Layer digests in order.
1221 pub layers: Vec<String>,
1222 /// Image entrypoint (from OCI config).
1223 #[serde(default)]
1224 pub entrypoint: Vec<String>,
1225 /// Image default command (from OCI config).
1226 #[serde(default)]
1227 pub cmd: Vec<String>,
1228 /// Image environment variables (from OCI config).
1229 #[serde(default)]
1230 pub env: Vec<String>,
1231 /// Image working directory (from OCI config).
1232 #[serde(default)]
1233 pub workdir: Option<String>,
1234 /// Image default user (from OCI config).
1235 #[serde(default)]
1236 pub user: Option<String>,
1237}
1238
1239/// Overlay preparation result.
1240#[derive(Debug, Clone, Serialize, Deserialize)]
1241pub struct OverlayInfo {
1242 /// Path to the merged overlay rootfs.
1243 pub rootfs_path: String,
1244 /// Path to the upper (writable) directory.
1245 pub upper_path: String,
1246 /// Path to the work directory.
1247 pub work_path: String,
1248}
1249
1250/// The guest's own account of machine memory, from `/proc/meminfo`.
1251///
1252/// This is what a machine is actually using. The host-side `phys_footprint`
1253/// answers a different question — it counts compressed pages at their
1254/// uncompressed size and keeps counting memory the guest has stopped needing —
1255/// so it runs well above these figures and should not be read as consumption.
1256#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
1257#[serde(rename_all = "camelCase")]
1258pub struct MemoryStatus {
1259 /// Total usable RAM the guest sees, in bytes. Slightly below the machine's
1260 /// configured size: the kernel reserves some before the allocator sees it.
1261 pub total_bytes: u64,
1262 /// Memory available for new allocations without swapping, in bytes. The
1263 /// figure to report, since `free_bytes` excludes reclaimable page cache.
1264 pub available_bytes: u64,
1265 /// Free memory in bytes: never allocated, or released and not reused.
1266 pub free_bytes: u64,
1267 /// Page cache in bytes. Counted inside `available_bytes`.
1268 pub cached_bytes: u64,
1269 /// Swap configured in the guest, in bytes. Zero when the guest has none.
1270 pub swap_total_bytes: u64,
1271 /// Swap in use, in bytes.
1272 pub swap_used_bytes: u64,
1273}
1274
1275impl MemoryStatus {
1276 /// Memory the guest cannot hand back on demand.
1277 pub fn used_bytes(&self) -> u64 {
1278 self.total_bytes.saturating_sub(self.available_bytes)
1279 }
1280}
1281
1282/// Storage status information.
1283#[derive(Debug, Clone, Serialize, Deserialize)]
1284pub struct StorageStatus {
1285 /// Whether the storage is formatted and ready.
1286 pub ready: bool,
1287 /// Total size in bytes.
1288 pub total_bytes: u64,
1289 /// Used size in bytes.
1290 pub used_bytes: u64,
1291 /// Number of cached layers.
1292 pub layer_count: usize,
1293 /// Number of cached images.
1294 pub image_count: usize,
1295}
1296
1297/// Registry authentication credentials for pulling images.
1298///
1299/// `Debug` is hand-written to redact the password: this value is carried inside
1300/// `AgentRequest::Pull`, and any `{:?}` of that request (e.g. a tracing span)
1301/// would otherwise serialize the token verbatim into the machine's console log,
1302/// which is exposed over the logs API.
1303#[derive(Clone, Serialize, Deserialize)]
1304pub struct RegistryAuth {
1305 /// Username for authentication.
1306 pub username: String,
1307 /// Password or token for authentication.
1308 pub password: String,
1309}
1310
1311impl std::fmt::Debug for RegistryAuth {
1312 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
1313 f.debug_struct("RegistryAuth")
1314 .field("username", &self.username)
1315 .field("password", &"***")
1316 .finish()
1317 }
1318}
1319
1320// ============================================================================
1321// Workload VM Protocol (Command Execution)
1322// ============================================================================
1323
1324/// Messages from host to workload VM.
1325#[derive(Debug, Clone, Serialize, Deserialize)]
1326#[serde(tag = "type", rename_all = "snake_case")]
1327pub enum HostMessage {
1328 /// Authentication request.
1329 Auth {
1330 /// Authentication token (base64).
1331 token: String,
1332 /// Protocol version.
1333 protocol_version: u32,
1334 },
1335
1336 /// Run a command.
1337 Run {
1338 /// Request ID for correlating responses.
1339 request_id: u64,
1340 /// Command and arguments.
1341 command: Vec<String>,
1342 /// Environment variables.
1343 env: Vec<(String, String)>,
1344 /// Working directory.
1345 workdir: Option<String>,
1346 },
1347
1348 /// Execute a command in running VM.
1349 Exec {
1350 /// Request ID.
1351 request_id: u64,
1352 /// Command and arguments.
1353 command: Vec<String>,
1354 /// Allocate a TTY.
1355 tty: bool,
1356 },
1357
1358 /// Send a signal to a running command.
1359 Signal {
1360 /// Request ID of the command.
1361 request_id: u64,
1362 /// Signal number.
1363 signal: i32,
1364 },
1365
1366 /// Request graceful shutdown.
1367 Stop {
1368 /// Timeout in milliseconds.
1369 timeout_ms: u64,
1370 },
1371}
1372
1373/// Messages from workload VM to host.
1374#[derive(Debug, Clone, Serialize, Deserialize)]
1375#[serde(tag = "type", rename_all = "snake_case")]
1376pub enum GuestMessage {
1377 /// Authentication successful.
1378 AuthOk,
1379
1380 /// Authentication failed.
1381 AuthFailed,
1382
1383 /// VM is ready to receive commands.
1384 Ready,
1385
1386 /// Command started.
1387 Started {
1388 /// Request ID.
1389 request_id: u64,
1390 },
1391
1392 /// Stdout data from command.
1393 Stdout {
1394 /// Request ID.
1395 request_id: u64,
1396 /// Output data.
1397 #[serde(with = "base64_bytes")]
1398 data: Vec<u8>,
1399 /// Whether output was truncated.
1400 truncated: bool,
1401 },
1402
1403 /// Stderr data from command.
1404 Stderr {
1405 /// Request ID.
1406 request_id: u64,
1407 /// Output data.
1408 #[serde(with = "base64_bytes")]
1409 data: Vec<u8>,
1410 /// Whether output was truncated.
1411 truncated: bool,
1412 },
1413
1414 /// Command exited.
1415 Exit {
1416 /// Request ID.
1417 request_id: u64,
1418 /// Exit code.
1419 code: i32,
1420 /// Exit reason.
1421 reason: String,
1422 },
1423
1424 /// Error occurred.
1425 Error {
1426 /// Request ID (if applicable).
1427 request_id: Option<u64>,
1428 /// Error message.
1429 message: String,
1430 },
1431}
1432
1433// ============================================================================
1434// Wire Format Helpers
1435// ============================================================================
1436
1437/// Envelope that wraps any message with an optional trace ID for correlation.
1438///
1439/// On the wire, the trace_id is flattened into the JSON alongside the message
1440/// fields: `{"trace_id":"abc123","method":"ping"}`.
1441#[derive(Debug, Clone, Serialize, Deserialize)]
1442pub struct Envelope<T> {
1443 /// Trace ID for correlating host API requests to agent operations.
1444 #[serde(skip_serializing_if = "Option::is_none", default)]
1445 pub trace_id: Option<String>,
1446 /// The wrapped message.
1447 #[serde(flatten)]
1448 pub body: T,
1449}
1450
1451impl<T> Envelope<T> {
1452 /// Create an envelope with no trace ID.
1453 pub fn new(body: T) -> Self {
1454 Self {
1455 trace_id: None,
1456 body,
1457 }
1458 }
1459
1460 /// Create an envelope with an optional trace ID.
1461 pub fn with_trace_id(body: T, trace_id: Option<String>) -> Self {
1462 Self { trace_id, body }
1463 }
1464}
1465
1466/// Encode a message to wire format (length-prefixed JSON).
1467pub fn encode_message<T: Serialize>(msg: &T) -> Result<Vec<u8>, serde_json::Error> {
1468 let json = serde_json::to_vec(msg)?;
1469 let len = json.len() as u32;
1470
1471 let mut buf = Vec::with_capacity(4 + json.len());
1472 buf.extend_from_slice(&len.to_be_bytes());
1473 buf.extend_from_slice(&json);
1474
1475 Ok(buf)
1476}
1477
1478/// Decode a message from wire format.
1479pub fn decode_message<T: for<'de> Deserialize<'de>>(data: &[u8]) -> Result<T, DecodeError> {
1480 if data.len() < 4 {
1481 return Err(DecodeError::TooShort);
1482 }
1483
1484 let len = u32::from_be_bytes([data[0], data[1], data[2], data[3]]) as usize;
1485
1486 if len > MAX_FRAME_SIZE as usize {
1487 return Err(DecodeError::TooLarge(len));
1488 }
1489
1490 if data.len() < 4 + len {
1491 return Err(DecodeError::Incomplete {
1492 expected: len,
1493 got: data.len() - 4,
1494 });
1495 }
1496
1497 serde_json::from_slice(&data[4..4 + len]).map_err(DecodeError::Json)
1498}
1499
1500/// Error decoding a wire message.
1501#[derive(Debug)]
1502pub enum DecodeError {
1503 /// Data too short to contain length header.
1504 TooShort,
1505 /// Frame size exceeds maximum.
1506 TooLarge(usize),
1507 /// Incomplete frame.
1508 Incomplete {
1509 /// Expected length.
1510 expected: usize,
1511 /// Actual length.
1512 got: usize,
1513 },
1514 /// JSON parse error.
1515 Json(serde_json::Error),
1516}
1517
1518impl std::fmt::Display for DecodeError {
1519 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
1520 match self {
1521 DecodeError::TooShort => write!(f, "data too short for length header"),
1522 DecodeError::TooLarge(size) => write!(f, "frame too large: {} bytes", size),
1523 DecodeError::Incomplete { expected, got } => {
1524 write!(
1525 f,
1526 "incomplete frame: expected {} bytes, got {}",
1527 expected, got
1528 )
1529 }
1530 DecodeError::Json(e) => write!(f, "JSON decode error: {}", e),
1531 }
1532 }
1533}
1534
1535impl std::error::Error for DecodeError {}
1536
1537#[cfg(test)]
1538mod tests {
1539 #[test]
1540 fn memory_growth_roundtrips_numeric_geometry_only() {
1541 let request = super::AgentRequest::OnlineMemory {
1542 start_address: 1 << 30,
1543 length_bytes: 128 << 20,
1544 };
1545 let wire = serde_json::to_string(&request).unwrap();
1546 assert!(matches!(
1547 serde_json::from_str::<super::AgentRequest>(&wire).unwrap(),
1548 super::AgentRequest::OnlineMemory {
1549 start_address: 1073741824,
1550 length_bytes: 134217728
1551 }
1552 ));
1553 for invalid in [
1554 r#"{"method":"online_memory","start_address":-1,"length_bytes":4096}"#,
1555 r#"{"method":"online_memory","start_address":"/dev/mem","length_bytes":4096}"#,
1556 ] {
1557 assert!(serde_json::from_str::<super::AgentRequest>(invalid).is_err());
1558 }
1559 }
1560 #[test]
1561 fn cpu_growth_roundtrips_a_numeric_count_only() {
1562 let req = super::AgentRequest::OnlineCpus { target_count: 4 };
1563 let wire = serde_json::to_string(&req).unwrap();
1564 assert!(matches!(
1565 serde_json::from_str::<super::AgentRequest>(&wire).unwrap(),
1566 super::AgentRequest::OnlineCpus { target_count: 4 }
1567 ));
1568 assert!(serde_json::from_str::<super::AgentRequest>(
1569 r#"{"method":"online_cpus","target_count":256}"#
1570 )
1571 .is_err());
1572 }
1573 #[test]
1574 fn filesystem_growth_uses_only_managed_disk_names() {
1575 use super::{AgentRequest, ManagedDisk};
1576 let request = AgentRequest::GrowFilesystem {
1577 disk: ManagedDisk::Storage,
1578 expected_bytes: 2147483648,
1579 };
1580 let wire = serde_json::to_value(&request).unwrap();
1581 assert_eq!(wire["method"], "grow_filesystem");
1582 assert_eq!(wire["disk"], "storage");
1583 assert!(matches!(
1584 serde_json::from_value::<AgentRequest>(wire).unwrap(),
1585 AgentRequest::GrowFilesystem {
1586 disk: ManagedDisk::Storage,
1587 expected_bytes: 2147483648
1588 }
1589 ));
1590 assert!(serde_json::from_str::<AgentRequest>(
1591 r#"{"method":"grow_filesystem","disk":"/dev/vdc","expected_bytes":2147483648}"#
1592 )
1593 .is_err());
1594 }
1595
1596 use super::*;
1597
1598 #[test]
1599 fn file_write_without_owner_fields_still_parses() {
1600 // Requests from clients predating uid/gid must keep deserializing.
1601 let old = r#"{"method":"file_write","path":"/x","data":"aGk=","mode":420}"#;
1602 let req: AgentRequest = serde_json::from_str(old).unwrap();
1603 match req {
1604 AgentRequest::FileWrite { mode, uid, gid, .. } => {
1605 assert_eq!(mode, Some(420));
1606 assert_eq!(uid, None);
1607 assert_eq!(gid, None);
1608 }
1609 other => panic!("unexpected: {other:?}"),
1610 }
1611 }
1612
1613 #[test]
1614 fn flatten_layers_output_is_optional() {
1615 // A client predating streaming always names a guest path to stage the
1616 // archive at, and must keep deserializing.
1617 let staged =
1618 r#"{"method":"flatten_layers","lowerdirs":["/a","/b"],"output":"/storage/x.tar"}"#;
1619 let req: AgentRequest = serde_json::from_str(staged).unwrap();
1620 match req {
1621 AgentRequest::FlattenLayers { lowerdirs, output } => {
1622 assert_eq!(lowerdirs, vec!["/a".to_string(), "/b".to_string()]);
1623 assert_eq!(output, Some("/storage/x.tar".to_string()));
1624 }
1625 other => panic!("unexpected: {other:?}"),
1626 }
1627
1628 // Omitting it asks the agent to stream the archive back instead, which is
1629 // what keeps a large export from needing room for a second full copy.
1630 let streamed = r#"{"method":"flatten_layers","lowerdirs":["/a","/b"]}"#;
1631 let req: AgentRequest = serde_json::from_str(streamed).unwrap();
1632 match req {
1633 AgentRequest::FlattenLayers { output, .. } => assert_eq!(output, None),
1634 other => panic!("unexpected: {other:?}"),
1635 }
1636 }
1637
1638 #[test]
1639 fn test_encode_decode_roundtrip() {
1640 let req = AgentRequest::Pull {
1641 image: "alpine:latest".to_string(),
1642 oci_platform: Some("linux/arm64".to_string()),
1643 auth: None,
1644 proxy: None,
1645 no_proxy: None,
1646 };
1647
1648 let encoded = encode_message(&req).unwrap();
1649 let decoded: AgentRequest = decode_message(&encoded).unwrap();
1650
1651 let AgentRequest::Pull {
1652 image,
1653 oci_platform,
1654 auth,
1655 proxy,
1656 no_proxy,
1657 } = decoded
1658 else {
1659 panic!("expected Pull variant, got {:?}", decoded);
1660 };
1661 assert_eq!(image, "alpine:latest");
1662 assert_eq!(oci_platform, Some("linux/arm64".to_string()));
1663 assert!(auth.is_none());
1664 assert!(proxy.is_none());
1665 assert!(no_proxy.is_none());
1666 }
1667
1668 #[test]
1669 fn test_encode_decode_with_auth() {
1670 let req = AgentRequest::Pull {
1671 image: "ghcr.io/owner/repo:latest".to_string(),
1672 oci_platform: None,
1673 auth: Some(RegistryAuth {
1674 username: "testuser".to_string(),
1675 password: "testpass".to_string(),
1676 }),
1677 proxy: None,
1678 no_proxy: None,
1679 };
1680
1681 let encoded = encode_message(&req).unwrap();
1682 let decoded: AgentRequest = decode_message(&encoded).unwrap();
1683
1684 let AgentRequest::Pull {
1685 image,
1686 oci_platform,
1687 auth,
1688 proxy: _,
1689 no_proxy: _,
1690 } = decoded
1691 else {
1692 panic!("expected Pull variant, got {:?}", decoded);
1693 };
1694 assert_eq!(image, "ghcr.io/owner/repo:latest");
1695 assert!(oci_platform.is_none());
1696 let auth = auth.expect("auth should be Some");
1697 assert_eq!(auth.username, "testuser");
1698 assert_eq!(auth.password, "testpass");
1699 }
1700
1701 #[test]
1702 fn test_encode_decode_with_proxy() {
1703 let req = AgentRequest::Pull {
1704 image: "alpine:latest".to_string(),
1705 oci_platform: None,
1706 auth: None,
1707 proxy: Some("http://192.168.127.254:3128".to_string()),
1708 no_proxy: Some("127.0.0.1,localhost,.internal".to_string()),
1709 };
1710
1711 let encoded = encode_message(&req).unwrap();
1712 let decoded: AgentRequest = decode_message(&encoded).unwrap();
1713
1714 let AgentRequest::Pull {
1715 proxy, no_proxy, ..
1716 } = decoded
1717 else {
1718 panic!("expected Pull variant, got {:?}", decoded);
1719 };
1720 assert_eq!(proxy.as_deref(), Some("http://192.168.127.254:3128"));
1721 assert_eq!(no_proxy.as_deref(), Some("127.0.0.1,localhost,.internal"));
1722 }
1723
1724 #[test]
1725 fn test_decode_too_short() {
1726 let data = [0u8; 2];
1727 let result: Result<AgentRequest, _> = decode_message(&data);
1728 assert!(matches!(result, Err(DecodeError::TooShort)));
1729 }
1730
1731 #[test]
1732 fn test_decode_incomplete() {
1733 let mut data = vec![0, 0, 0, 100]; // claims 100 bytes
1734 data.extend_from_slice(b"{}"); // only 2 bytes of payload
1735 let result: Result<AgentRequest, _> = decode_message(&data);
1736 assert!(matches!(result, Err(DecodeError::Incomplete { .. })));
1737 }
1738
1739 #[test]
1740 fn test_agent_request_serialization() {
1741 let req = AgentRequest::Ping;
1742 let json = serde_json::to_string(&req).unwrap();
1743 assert!(json.contains("ping"));
1744
1745 let req = AgentRequest::PrepareOverlay {
1746 image: "ubuntu:22.04".to_string(),
1747 workload_id: "wl-123".to_string(),
1748 };
1749 let json = serde_json::to_string(&req).unwrap();
1750 assert!(json.contains("prepare_overlay"));
1751 }
1752
1753 #[test]
1754 fn test_agent_response_serialization() {
1755 let resp = AgentResponse::Pong {
1756 version: PROTOCOL_VERSION,
1757 capabilities: vec![forkpoint::TYPED_BRANCHPOINT_CAPABILITY.to_string()],
1758 };
1759 let json = serde_json::to_string(&resp).unwrap();
1760 assert!(json.contains("pong"));
1761 assert!(json.contains(forkpoint::TYPED_BRANCHPOINT_CAPABILITY));
1762
1763 let legacy: AgentResponse =
1764 serde_json::from_str(r#"{"status":"pong","version":1}"#).unwrap();
1765 assert!(matches!(
1766 legacy,
1767 AgentResponse::Pong {
1768 version: PROTOCOL_VERSION,
1769 capabilities
1770 } if capabilities.is_empty()
1771 ));
1772
1773 let resp = AgentResponse::Progress {
1774 message: "Pulling layer 1/3".to_string(),
1775 percent: Some(33),
1776 layer: Some("sha256:abc123".to_string()),
1777 };
1778 let json = serde_json::to_string(&resp).unwrap();
1779 assert!(json.contains("progress"));
1780 }
1781
1782 #[test]
1783 fn file_write_begin_roundtrips() {
1784 let req = AgentRequest::FileWriteBegin {
1785 path: "/tmp/target".into(),
1786 mode: Some(0o600),
1787 uid: Some(1000),
1788 gid: Some(1000),
1789 total_size: 123_456_789,
1790 };
1791 let bytes = encode_message(&req).unwrap();
1792 let back: AgentRequest = decode_message(&bytes).unwrap();
1793 match back {
1794 AgentRequest::FileWriteBegin {
1795 path,
1796 mode,
1797 uid,
1798 gid,
1799 total_size,
1800 } => {
1801 assert_eq!(path, "/tmp/target");
1802 assert_eq!(mode, Some(0o600));
1803 assert_eq!(uid, Some(1000));
1804 assert_eq!(gid, Some(1000));
1805 assert_eq!(total_size, 123_456_789);
1806 }
1807 _ => panic!("wrong variant"),
1808 }
1809 }
1810
1811 #[test]
1812 fn file_write_chunk_roundtrips_binary_data() {
1813 // Binary data (bytes outside UTF-8) must survive the base64
1814 // trip intact. If the encoding ever silently lossifies, this
1815 // fires.
1816 let payload: Vec<u8> = (0u8..=255).collect();
1817 let req = AgentRequest::FileWriteChunk {
1818 data: payload.clone(),
1819 done: true,
1820 };
1821 let bytes = encode_message(&req).unwrap();
1822 let back: AgentRequest = decode_message(&bytes).unwrap();
1823 match back {
1824 AgentRequest::FileWriteChunk { data, done } => {
1825 assert_eq!(data, payload);
1826 assert!(done);
1827 }
1828 _ => panic!("wrong variant"),
1829 }
1830 }
1831
1832 #[test]
1833 fn file_write_size_constants_are_frame_safe() {
1834 // Sanity: a single streaming chunk at FILE_WRITE_CHUNK_SIZE
1835 // must fit inside MAX_FRAME_SIZE after base64 (+ ~33%) and
1836 // JSON overhead. If anyone bumps CHUNK_SIZE past the limit,
1837 // this test fires before production does.
1838 let chunk_bytes = FILE_WRITE_CHUNK_SIZE as u64;
1839 let base64_bytes = chunk_bytes.div_ceil(3) * 4; // ceil(n/3)*4
1840 let json_overhead = 256u64; // method tag, done bool, quotes
1841 let total = base64_bytes + json_overhead;
1842 assert!(
1843 total < MAX_FRAME_SIZE as u64,
1844 "FILE_WRITE_CHUNK_SIZE of {} bytes would produce a frame \
1845 of ~{} bytes which exceeds MAX_FRAME_SIZE of {}",
1846 chunk_bytes,
1847 total,
1848 MAX_FRAME_SIZE
1849 );
1850 }
1851
1852 #[test]
1853 fn test_ports_constants() {
1854 assert_eq!(ports::WORKLOAD_CONTROL, 5000);
1855 assert_eq!(ports::WORKLOAD_LOGS, 5001);
1856 assert_eq!(ports::AGENT_CONTROL, 6000);
1857 assert_eq!(ports::SSH_AGENT, 6001);
1858 }
1859
1860 #[test]
1861 fn test_cid_constants() {
1862 assert_eq!(cid::HOST, 2);
1863 assert_eq!(cid::GUEST, 3);
1864 }
1865
1866 #[test]
1867 fn test_envelope_serialization_with_trace_id() {
1868 let req = AgentRequest::Ping;
1869 let envelope = Envelope::with_trace_id(&req, Some("abc123".to_string()));
1870 let json = serde_json::to_string(&envelope).unwrap();
1871
1872 // trace_id should be flattened alongside the method tag
1873 assert!(json.contains("\"trace_id\":\"abc123\""));
1874 assert!(json.contains("\"method\":\"ping\""));
1875
1876 // Deserialize back — Envelope<AgentRequest> with flatten
1877 let parsed: Envelope<AgentRequest> = serde_json::from_str(&json).unwrap();
1878 assert_eq!(parsed.trace_id.as_deref(), Some("abc123"));
1879 assert!(matches!(parsed.body, AgentRequest::Ping));
1880 }
1881
1882 #[test]
1883 fn test_envelope_without_trace_id() {
1884 let req = AgentRequest::Ping;
1885 let envelope = Envelope::new(&req);
1886 let json = serde_json::to_string(&envelope).unwrap();
1887
1888 // No trace_id field (skip_serializing_if = None)
1889 assert!(!json.contains("trace_id"));
1890 assert!(json.contains("\"method\":\"ping\""));
1891 }
1892
1893 #[test]
1894 fn test_envelope_backward_compat_bare_request() {
1895 // A bare AgentRequest (no Envelope) should fail to parse as Envelope
1896 // but succeed as bare AgentRequest — this is the agent's fallback path
1897 let bare_json = r#"{"method":"ping"}"#;
1898
1899 // Envelope parse should fail (no body field to flatten into)
1900 // Actually with flatten, this may work — let's verify
1901 let envelope_result = serde_json::from_str::<Envelope<AgentRequest>>(bare_json);
1902 let bare_result = serde_json::from_str::<AgentRequest>(bare_json);
1903
1904 // At least one must succeed for backward compat
1905 assert!(
1906 envelope_result.is_ok() || bare_result.is_ok(),
1907 "Neither Envelope nor bare parse succeeded"
1908 );
1909
1910 // Bare parse must always work
1911 assert!(bare_result.is_ok());
1912 assert!(matches!(bare_result.unwrap(), AgentRequest::Ping));
1913
1914 // If Envelope works, trace_id should be None
1915 if let Ok(env) = envelope_result {
1916 assert!(env.trace_id.is_none());
1917 }
1918 }
1919}