arcbox-agent 0.9.0

Guest agent for ArcBox VMs
//! ext4 metadata volume: format, mount, migrate, and bind the fsync-hot
//! boltdb metadata onto a small journaled ext4 disk.
//!
//! Container-start profiling (ABX-496) put ~90 % of fsyncs on these boltdb
//! files, and one fsync costs ~9.5 ms on btrfs vs ~1 ms on ext4 over the
//! same virtio-blk stack — so the hot metadata moves to ext4 while bulk data
//! (layers, blobs, volumes) stays on the compressed btrfs data volume. The
//! crash-safe migration state machine lives in `crate::metadata_migrate`;
//! design and failure policy: ../company/engineering/arcbox/plans/ext4-metadata-volume.md.

use std::path::Path;

use arcbox_constants::paths::{CONTAINERD_DATA_MOUNT_POINT, DOCKER_DATA_MOUNT_POINT};

use arcbox_constants::devices::DOCKER_METADATA_BLOCK_DEVICE;

use super::cmdline::declared_docker_metadata_device;
use crate::metadata_migrate::{
    EntryKind, Prepared, prepare_entry, verify_legacy_pair, verify_legacy_sources,
};

/// Mount point of the raw ext4 volume (`/run` is tmpfs, writable).
pub(super) const METADATA_MOUNT: &str = "/run/arcbox/metadata";
/// The formatter is baked into the EROFS rootfs (static e2fsprogs).
const MKFS_EXT4: &str = "/sbin/mkfs.ext4";

/// One fsync-hot metadata location: an entry on the volume bound over its
/// canonical btrfs-side path. The set is exactly the profiled hot set —
/// everything else (containers/, volumes/, builder/, trust/) stays on btrfs.
struct Mapping {
    /// Entry name inside the metadata volume.
    name: &'static str,
    /// Canonical path the runtime opens (bind target).
    target: String,
    kind: EntryKind,
}

fn mappings() -> Vec<Mapping> {
    vec![
        Mapping {
            name: "containerd-bolt",
            target: format!("{CONTAINERD_DATA_MOUNT_POINT}/io.containerd.metadata.v1.bolt"),
            kind: EntryKind::Dir,
        },
        // The snapshotter dir also holds snapshots/ (bulk layer data), so
        // only its boltdb is bound — a file bind is safe for bolt, which
        // writes in place and never renames its database file.
        Mapping {
            name: "snapshotter-metadata.db",
            target: format!(
                "{CONTAINERD_DATA_MOUNT_POINT}/io.containerd.snapshotter.v1.overlayfs/metadata.db"
            ),
            kind: EntryKind::File,
        },
        Mapping {
            name: "docker-network",
            target: format!("{DOCKER_DATA_MOUNT_POINT}/network"),
            kind: EntryKind::Dir,
        },
        Mapping {
            name: "docker-image",
            target: format!("{DOCKER_DATA_MOUNT_POINT}/image"),
            kind: EntryKind::Dir,
        },
        Mapping {
            name: "docker-buildkit",
            target: format!("{DOCKER_DATA_MOUNT_POINT}/buildkit"),
            kind: EntryKind::Dir,
        },
    ]
}

/// Mounts the ext4 metadata volume and binds the hot metadata dirs over
/// their btrfs-side paths. Must run after `ensure_data_mount` (targets live
/// on the data subvolumes) and before containerd/dockerd start (their boltdb
/// files must be closed while entries migrate).
///
/// Failure policy:
/// - no cmdline declaration and no default node (older daemon without the
///   third disk) → `Ok`, btrfs-only boot, zero probe delay;
/// - device declared but never appears, or present but unusable after an
///   attempted mount → `Err` — booting dockerd against the stale shadowed
///   btrfs state would fork it.
pub(super) fn ensure_metadata_mount() -> Result<String, String> {
    let maps = mappings();
    if maps.iter().all(|m| crate::mount::is_mounted(&m.target)) {
        super::storage_volume::finish_metadata_setup()?;
        return Ok("metadata binds already mounted".to_string());
    }

    let device = match declared_docker_metadata_device() {
        Some(device) => {
            if !wait_for_device(&device) {
                return Err(format!("declared metadata device {device} never appeared"));
            }
            device
        }
        // No declaration (older daemon). An already-present default node is
        // still honored so bespoke configs (e2e probes) can attach the disk
        // without a cmdline; otherwise run the btrfs-only layout.
        None if Path::new(DOCKER_METADATA_BLOCK_DEVICE).exists() => {
            DOCKER_METADATA_BLOCK_DEVICE.to_string()
        }
        None => {
            tracing::warn!("no metadata device declared or present; btrfs-only layout");
            return Ok("metadata volume skipped (no device)".to_string());
        }
    };

    let mut notes = Vec::new();

    let layout = super::storage_volume::authorize_legacy_migration(|| {
        let entries: Vec<_> = maps
            .iter()
            .map(|mapping| (Path::new(&mapping.target), mapping.kind))
            .collect();
        verify_legacy_sources(&entries)
    })?;
    super::storage_volume::ensure_filesystem(
        arcbox_storage::VolumeRole::Metadata,
        &device,
        |uuid| format_ext4(&device, uuid).map(|note| notes.push(note)),
    )?;

    mount_metadata(&device)?;

    if layout == arcbox_storage::StorageLayout::LegacyPair {
        let entries: Vec<_> = maps.iter().map(|mapping| mapping.name).collect();
        verify_legacy_pair(Path::new(METADATA_MOUNT), &entries)
            .map_err(|error| format!("runtime storage needs recovery: {error}"))?;
    }

    for mapping in &maps {
        if crate::mount::is_mounted(&mapping.target) {
            continue;
        }
        let volume_entry = Path::new(METADATA_MOUNT).join(mapping.name);
        if layout == arcbox_storage::StorageLayout::Paired
            && !volume_entry.try_exists().map_err(|error| {
                format!("inspect metadata entry {}: {error}", volume_entry.display())
            })?
        {
            return Err(format!(
                "runtime storage needs recovery: paired metadata entry {} is missing",
                volume_entry.display()
            ));
        }
        match prepare_entry(
            Path::new(METADATA_MOUNT),
            Path::new(&mapping.target),
            mapping.name,
            mapping.kind,
        ) {
            Ok(Prepared::Migrated) => notes.push(format!("migrated {}", mapping.target)),
            Ok(_) => {}
            Err(e) => return Err(format!("prepare {} failed: {e}", mapping.target)),
        }
        bind(&volume_entry, &mapping.target)?;
    }
    super::storage_volume::finish_metadata_setup()?;

    if notes.is_empty() {
        Ok("metadata volume mounted".to_string())
    } else {
        Ok(notes.join("; "))
    }
}

/// Waits up to 5 s for the VirtIO block device node (same budget and
/// rationale as the data-device wait in `btrfs.rs`).
fn wait_for_device(device: &str) -> bool {
    for attempt in 0..50 {
        if Path::new(device).exists() {
            if attempt > 0 {
                tracing::info!(device, attempt, "waited for metadata device");
            }
            return true;
        }
        std::thread::sleep(std::time::Duration::from_millis(100));
    }
    false
}

fn format_ext4(device: &str, uuid: uuid::Uuid) -> Result<String, String> {
    // Explicit feature list so the result is deterministic regardless of
    // any mke2fs.conf; fast_commit targets exactly the small-metadata-commit
    // fsync pattern boltdb produces, lazy init off pays the one-time cost at
    // format instead of trickling background writes into first boot.
    match std::process::Command::new(MKFS_EXT4)
        .args([
            "-F",
            "-U",
            &uuid.to_string(),
            "-t",
            "ext4",
            "-O",
            "has_journal,extent,huge_file,flex_bg,metadata_csum,64bit,dir_nlink,extra_isize,fast_commit",
            "-E",
            "lazy_itable_init=0,lazy_journal_init=0",
            "-L",
            "arcbox-meta",
            device,
        ])
        .status()
    {
        Ok(status) if status.success() => Ok(format!("formatted {device} as ext4")),
        Ok(status) => Err(format!(
            "mkfs.ext4 failed on {device} (exit={})",
            status.code().unwrap_or(-1)
        )),
        Err(e) => Err(format!("failed to execute mkfs.ext4: {e}")),
    }
}

/// Mounts the volume without running destructive repair during ordinary boot.
fn mount_metadata(device: &str) -> Result<(), String> {
    if crate::mount::is_mounted(METADATA_MOUNT) {
        return Ok(());
    }
    std::fs::create_dir_all(METADATA_MOUNT)
        .map_err(|e| format!("failed to create {METADATA_MOUNT}: {e}"))?;

    if try_mount(device) {
        return Ok(());
    }

    Err(format!(
        "runtime storage needs recovery: mount {device} on {METADATA_MOUNT} failed; preserve both images before offline repair"
    ))
}

fn try_mount(device: &str) -> bool {
    matches!(
        std::process::Command::new("/bin/busybox")
            .args(["mount", "-t", "ext4", "-o", "noatime", device, METADATA_MOUNT])
            .status(),
        Ok(status) if status.success()
    )
}

fn bind(source: &Path, target: &str) -> Result<(), String> {
    nix::mount::mount(
        Some(source),
        target,
        None::<&str>,
        nix::mount::MsFlags::MS_BIND,
        None::<&str>,
    )
    .map_err(|e| format!("bind {} -> {target} failed: {e}", source.display()))
}