Skip to main content

fakecloud_core/
container_net.rs

1//! Shared container-to-host networking resolution for service runtimes
2//! that spawn sibling containers (Lambda, ECS, RDS, ElastiCache).
3//!
4//! Captures the issue #1539 fix shape in one place so the four runtimes
5//! that shell out to `docker`/`podman` can't drift apart again:
6//!
7//! - **podman** ships `host.containers.internal` as a built-in container
8//!   DNS entry on every platform and must NOT receive
9//!   `--add-host host.docker.internal:host-gateway` — rootless podman's
10//!   gvproxy leaves the magic alias empty and the `create` fails with
11//!   "host containers internal IP address is empty".
12//! - **bare docker on Linux** has no `host-gateway` magic; the bridge
13//!   gateway IP has to be resolved from the daemon and injected explicitly.
14//! - **Docker Desktop on Mac/Windows** resolves the `host-gateway` magic
15//!   value to the host's IP.
16//! - when fakecloud itself runs in a container (`FAKECLOUD_IN_CONTAINER=1`,
17//!   baked into the published image), the sibling containers it spawns
18//!   publish their ports on the *host's* daemon — reachable from inside
19//!   fakecloud's container as `host.docker.internal:<port>`, not
20//!   `127.0.0.1:<port>`.
21
22/// Actionable remediation appended to every error raised when a container
23/// runtime (Docker/Podman) is required for an operation but none is
24/// available. Kept in one place so RDS, Lambda, ECS, and the server startup
25/// banner all surface the same fix steps and can't drift apart.
26pub const CONTAINER_RUNTIME_HINT: &str = "Install and start Docker or Podman, or set FAKECLOUD_CONTAINER_CLI to your container CLI path.";
27
28/// Auto-detect an available container CLI. Honors `FAKECLOUD_CONTAINER_CLI`
29/// as an explicit override (returns `None` if the override doesn't work),
30/// otherwise prefers `docker` then `podman`. Returns `None` when neither
31/// is usable.
32pub fn detect_container_cli() -> Option<String> {
33    if let Ok(cli) = std::env::var("FAKECLOUD_CONTAINER_CLI") {
34        return if cli_available(&cli) { Some(cli) } else { None };
35    }
36    if cli_available("docker") {
37        Some("docker".to_string())
38    } else if cli_available("podman") {
39        Some("podman".to_string())
40    } else {
41        None
42    }
43}
44
45/// How long to wait for `<cli> info` before giving up and treating the
46/// runtime as unavailable. A healthy daemon answers in well under a second;
47/// an unreachable or wedged daemon (stale `DOCKER_HOST`, Docker Desktop mid
48/// start, a broken socket) can leave the CLI blocked on connect *forever*,
49/// which would hang fakecloud startup and the test harness. Bounding the
50/// probe turns "daemon wedged" into "no runtime detected" instead of a hang.
51pub const CLI_PROBE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10);
52
53/// Process-global memo of `<cli> info` results, keyed by CLI name/path.
54///
55/// Container-runtime liveness is fixed for the life of a process, but every
56/// service runtime (Lambda, ECS, RDS, ElastiCache, EC2, MQ, MSK, ...) probes
57/// it independently at startup — a dozen-plus `detect_container_cli()` calls.
58/// Without a memo each probe re-runs `docker info`; when the daemon is wedged
59/// (see [`CLI_PROBE_TIMEOUT`]) those probes are serial 10s hangs that stack
60/// into minutes, wedging server startup and the conformance `*_probe` tests.
61/// Caching the first answer collapses that to a single probe.
62static CLI_AVAILABLE_CACHE: std::sync::OnceLock<
63    std::sync::Mutex<std::collections::HashMap<String, bool>>,
64> = std::sync::OnceLock::new();
65
66/// True when the CLI responds to `<cli> info` with success within
67/// [`CLI_PROBE_TIMEOUT`] — the same liveness probe every runtime used before
68/// this module existed, but bounded so an unreachable daemon can't hang the
69/// caller indefinitely (the CLI blocks on connect with no timeout of its own),
70/// and memoized per process so a dozen runtimes probing at startup don't each
71/// pay that bound.
72pub fn cli_available(cli: &str) -> bool {
73    let cache =
74        CLI_AVAILABLE_CACHE.get_or_init(|| std::sync::Mutex::new(std::collections::HashMap::new()));
75    if let Some(&cached) = cache.lock().unwrap().get(cli) {
76        return cached;
77    }
78    let result = probe_cli(cli);
79    cache.lock().unwrap().insert(cli.to_string(), result);
80    result
81}
82
83/// Run the bounded `<cli> info` liveness probe once (uncached).
84fn probe_cli(cli: &str) -> bool {
85    let child = spawn_bounded(
86        std::process::Command::new(cli)
87            .arg("info")
88            .stdout(std::process::Stdio::null())
89            .stderr(std::process::Stdio::null()),
90    );
91    let Ok(mut child) = child else {
92        return false;
93    };
94    wait_bounded_group(&mut child) && child.wait().map(|s| s.success()).unwrap_or(false)
95}
96
97/// Spawn a container-CLI command in a process group of its own (Unix), so a
98/// timed-out call can be torn down whole. `FAKECLOUD_CONTAINER_CLI` is
99/// routinely a wrapper -- `sh -c 'exec docker "$@"'`, a `podman-remote` shim --
100/// which makes the real command a *grandchild*: it survives `Child::kill`, goes
101/// on holding whatever pipes we handed it, and keeps running against a wedged
102/// daemon forever. Its own group makes it reachable by a single signal.
103/// Detaching these from terminal job control is fine: their lifetime is managed
104/// by deadline here, not by the shell fakecloud was started from.
105///
106/// stdin is /dev/null, and has to be: a new process group is a *background*
107/// one, so a child that reads the controlling terminal -- a `sudo` or
108/// credential-helper wrapper prompting, exactly the wrapper case above -- takes
109/// SIGTTIN, which stops it rather than ending it. `try_wait` is WNOHANG without
110/// WUNTRACED, so the loop below never sees a stopped child and the call burns
111/// the whole deadline before being killed. These calls are non-interactive
112/// anyway, so an immediate EOF is the right answer for them.
113fn spawn_bounded(cmd: &mut std::process::Command) -> std::io::Result<std::process::Child> {
114    cmd.stdin(std::process::Stdio::null());
115    #[cfg(unix)]
116    {
117        std::os::unix::process::CommandExt::process_group(cmd, 0);
118    }
119    cmd.spawn()
120}
121
122/// Wait for `child` up to [`CLI_PROBE_TIMEOUT`], killing it on expiry. Returns
123/// whether it exited on its own. Every container-CLI call goes through this:
124/// a liveness probe answering does not promise the next call will, and an
125/// unbounded one blocks the caller rather than just that command.
126///
127/// Only for a child from [`spawn_bounded`], which put it in a group of its own:
128/// the expiry kill hits that whole group, so a wrapper CLI's grandchildren die
129/// with it.
130fn wait_bounded_group(child: &mut std::process::Child) -> bool {
131    let deadline = std::time::Instant::now() + CLI_PROBE_TIMEOUT;
132    loop {
133        match child.try_wait() {
134            Ok(Some(_)) => return true,
135            Ok(None) => {}
136            Err(_) => return false,
137        }
138        if std::time::Instant::now() >= deadline {
139            // Daemon is wedged: kill the blocked call and report failure.
140            kill_expired(child);
141            let _ = child.wait();
142            return false;
143        }
144        std::thread::sleep(std::time::Duration::from_millis(25));
145    }
146}
147
148/// SIGKILL a timed-out child and its process group. [`spawn_bounded`] made the
149/// child its own group leader, so the group id is the child's pid, and the
150/// child is still unreaped here -- the pid cannot have been recycled and the
151/// signal cannot stray onto an unrelated group.
152#[cfg(unix)]
153fn kill_expired(child: &mut std::process::Child) {
154    // SAFETY: `kill` with a negative pid targets the process group of that
155    // id; any pid value is safe to pass.
156    let _ = unsafe { libc::kill(-(child.id() as libc::pid_t), libc::SIGKILL) };
157    let _ = child.kill();
158}
159
160/// Windows has no process-group signal (a job object would be needed), so the
161/// direct child is as far as the kill reaches; [`run_bounded`] still bounds the
162/// wait on its stdout reader so the caller can't be held by a surviving
163/// grandchild.
164#[cfg(not(unix))]
165fn kill_expired(child: &mut std::process::Child) {
166    let _ = child.kill();
167}
168
169/// Whether the stdout reader thread ended before the call returned.
170#[derive(Debug)]
171enum ReaderState {
172    /// The reader returned; its thread is gone.
173    Finished,
174    /// The reader is still blocked on the pipe because a write end we could not
175    /// close is held outside the child's process group. The thread outlives the
176    /// call; the caller does not wait for it.
177    Abandoned,
178}
179
180/// Floor on how long [`run_bounded`] waits for its stdout reader once the call
181/// is over (it also gets whatever is left of the call's own budget). Both exits
182/// close every write end we control -- the child exited, or its whole process
183/// group was killed -- which ends the blocked `read_to_end` at once, so this
184/// covers scheduling only. It exists so a write end held somewhere we cannot
185/// reach costs the caller a few hundred milliseconds instead of blocking it for
186/// good, which is what an unbounded join did.
187const READER_DRAIN_GRACE: std::time::Duration = std::time::Duration::from_millis(500);
188
189/// Run a container-CLI command and return its stdout, or `None` when it fails
190/// or outruns [`CLI_PROBE_TIMEOUT`].
191pub fn bounded_output(cli: &str, args: &[&str]) -> Option<String> {
192    run_bounded(cli, args).0
193}
194
195/// [`bounded_output`], plus whether its stdout reader finished -- so the
196/// timeout path's "no reader left behind" guarantee is unit-testable instead of
197/// only observable as a thread that never goes away.
198fn run_bounded(cli: &str, args: &[&str]) -> (Option<String>, ReaderState) {
199    let deadline = std::time::Instant::now() + CLI_PROBE_TIMEOUT;
200    let child = spawn_bounded(
201        std::process::Command::new(cli)
202            .args(args)
203            .stdout(std::process::Stdio::piped())
204            .stderr(std::process::Stdio::null()),
205    );
206    let Ok(mut child) = child else {
207        return (None, ReaderState::Finished);
208    };
209    let Some(mut stdout) = child.stdout.take() else {
210        kill_expired(&mut child);
211        let _ = child.wait();
212        return (None, ReaderState::Finished);
213    };
214    // Drain stdout while waiting. A child whose output outgrows the pipe
215    // buffer blocks on write until someone reads it, so waiting for exit
216    // first would deadlock until the deadline and then report the sweep as
217    // failed -- `docker ps -a` across a busy host is exactly that much output.
218    //
219    // The channel doubles as the reader's "I'm done" signal: the send is the
220    // last thing the thread does before dropping the pipe's read end, so a
221    // received buffer proves no reader is parked behind us. A `JoinHandle`
222    // can't say that without blocking, which on the timeout path is exactly
223    // what we must not do.
224    let (tx, rx) = std::sync::mpsc::channel();
225    std::thread::spawn(move || {
226        let mut buf = Vec::new();
227        let _ = std::io::Read::read_to_end(&mut stdout, &mut buf);
228        let _ = tx.send(buf);
229    });
230    // On expiry `wait_bounded_group` has killed the whole process group, so a
231    // wrapper CLI's grandchild releases the write end and the reader returns
232    // instead of blocking for the life of the process -- one leaked thread per
233    // call, on precisely the wedged-daemon path these bounds exist for.
234    let exited = wait_bounded_group(&mut child);
235    let status = child.wait().ok();
236    // Whatever is left of the call's own budget, and never less than the grace:
237    // a prompt call can afford to wait out a reader thread the scheduler hasn't
238    // run yet, a timed-out one gets only the grace, and either way the caller is
239    // back within CLI_PROBE_TIMEOUT plus that grace.
240    let grace = deadline
241        .saturating_duration_since(std::time::Instant::now())
242        .max(READER_DRAIN_GRACE);
243    let drained = rx.recv_timeout(grace).ok();
244    let output = match (exited, status, &drained) {
245        (true, Some(status), Some(buf)) if status.success() => {
246            Some(String::from_utf8_lossy(buf).into_owned())
247        }
248        _ => None,
249    };
250    let reader = if drained.is_some() {
251        ReaderState::Finished
252    } else {
253        ReaderState::Abandoned
254    };
255    (output, reader)
256}
257
258/// Run a container-CLI command for its effect only, bounded the same way.
259/// Returns whether it succeeded.
260pub fn bounded_status(cli: &str, args: &[&str]) -> bool {
261    let Ok(mut child) = spawn_bounded(
262        std::process::Command::new(cli)
263            .args(args)
264            .stdout(std::process::Stdio::null())
265            .stderr(std::process::Stdio::null()),
266    ) else {
267        return false;
268    };
269    wait_bounded_group(&mut child) && child.wait().map(|s| s.success()).unwrap_or(false)
270}
271
272/// True if the given PID is a live process on this host.
273///
274/// On Unix this is `kill(pid, 0)`: it returns 0 if the process exists
275/// (including zombies), or sets `errno` to `ESRCH` if not. On non-Unix
276/// platforms it conservatively returns `true`, so a caller never removes a
277/// resource it can't prove is orphaned.
278#[cfg(unix)]
279pub fn pid_alive(pid: u32) -> bool {
280    // SAFETY: `kill` with signal 0 is a liveness probe; it does not
281    // actually deliver a signal. Any PID value is safe to pass.
282    let rc = unsafe { libc::kill(pid as libc::pid_t, 0) };
283    if rc == 0 {
284        return true;
285    }
286    // errno == EPERM means the process exists but we can't signal it —
287    // still alive from our perspective.
288    std::io::Error::last_os_error().raw_os_error() == Some(libc::EPERM)
289}
290
291#[cfg(not(unix))]
292pub fn pid_alive(_pid: u32) -> bool {
293    true
294}
295
296/// Whether a container or network labelled `fakecloud-instance=<label>` was
297/// left behind by a fakecloud process that is gone. The label is
298/// `fakecloud-<pid>`; an object is orphaned only when that PID is neither the
299/// current process nor alive. Several fakecloud processes can share one
300/// daemon (parallel test servers, side-by-side installs), so an object owned
301/// by *another live* process is never an orphan. A label that doesn't parse is
302/// not treated as an orphan either -- nothing proves its owner is gone.
303pub fn owned_by_dead_process(label: &str, is_alive: impl Fn(u32) -> bool) -> bool {
304    let Some(pid) = label
305        .strip_prefix("fakecloud-")
306        .and_then(|p| p.parse::<u32>().ok())
307    else {
308        return false;
309    };
310    pid != std::process::id() && !is_alive(pid)
311}
312
313/// Which container engine a CLI actually drives. Decides the podman-only
314/// code paths: `host.containers.internal` without `--add-host`, and
315/// `--tls-verify=false` when pulling from fakecloud's plain-HTTP registry.
316#[derive(Debug, Clone, Copy, PartialEq, Eq)]
317pub enum ContainerEngine {
318    Docker,
319    Podman,
320}
321
322/// Process-global memo of [`is_podman`] results, keyed by CLI name/path. The
323/// engine behind a CLI is fixed for the life of the process, and the probe
324/// runs a subprocess, so every runtime constructor and image pull after the
325/// first reads the answer from here.
326static PODMAN_CACHE: std::sync::OnceLock<
327    std::sync::Mutex<std::collections::HashMap<String, bool>>,
328> = std::sync::OnceLock::new();
329
330fn podman_cache() -> &'static std::sync::Mutex<std::collections::HashMap<String, bool>> {
331    PODMAN_CACHE.get_or_init(|| std::sync::Mutex::new(std::collections::HashMap::new()))
332}
333
334/// True when `cli` drives podman -- including podman installed *as* `docker`.
335///
336/// The file name alone can't answer that: the `podman-docker` package
337/// (Fedora, RHEL, CentOS Stream, ...) ships a `docker` shim that execs podman,
338/// and a user may point `FAKECLOUD_CONTAINER_CLI` at any wrapper. So:
339///
340/// 1. fast path: a name containing `podman` (`podman`, `podman-remote`,
341///    `/opt/homebrew/bin/podman`) is podman without running anything;
342/// 2. otherwise ask the CLI: `<cli> --version` prints `podman version X.Y.Z`
343///    through the shim and `Docker version X.Y.Z, build ...` for Docker;
344/// 3. when that answers with something neither (a custom wrapper), look for
345///    podman-only fields in `<cli> info`.
346///
347/// Every call is bounded by [`CLI_PROBE_TIMEOUT`] (a wedged CLI costs one
348/// bound, then reads as Docker, the historical default) and the answer is
349/// memoized per CLI. Blocking: from async code use [`is_podman_async`].
350pub fn is_podman(cli: &str) -> bool {
351    if name_indicates_podman(cli) {
352        return true;
353    }
354    if let Some(&cached) = podman_cache().lock().unwrap().get(cli) {
355        return cached;
356    }
357    let podman = probe_engine(cli) == Some(ContainerEngine::Podman);
358    podman_cache()
359        .lock()
360        .unwrap()
361        .insert(cli.to_string(), podman);
362    podman
363}
364
365/// [`is_podman`] for async callers: a cached (or name-matched) answer returns
366/// at once, and a first-time probe runs on the blocking pool so it can't stall
367/// a runtime worker for up to [`CLI_PROBE_TIMEOUT`].
368pub async fn is_podman_async(cli: &str) -> bool {
369    if name_indicates_podman(cli) {
370        return true;
371    }
372    if let Some(&cached) = podman_cache().lock().unwrap().get(cli) {
373        return cached;
374    }
375    let owned = cli.to_string();
376    tokio::task::spawn_blocking(move || is_podman(&owned))
377        .await
378        .unwrap_or(false)
379}
380
381/// The name-only fast path: the file name component contains `podman`, so
382/// absolute paths and `podman-remote` register. Docker's CLI and the
383/// `podman-docker` shim are both named `docker` and fall through to the probe.
384fn name_indicates_podman(cli: &str) -> bool {
385    std::path::Path::new(cli)
386        .file_name()
387        .and_then(|n| n.to_str())
388        .map(|n| n.contains("podman"))
389        .unwrap_or(false)
390}
391
392/// Ask the CLI what it is (uncached). `None` when it can't tell -- the CLI
393/// failed, timed out, or answered with neither engine's markers.
394fn probe_engine(cli: &str) -> Option<ContainerEngine> {
395    // A failed or timed-out `--version` means the CLI isn't answering at
396    // all; don't spend a second bound on `info` against it.
397    let version = bounded_output(cli, &["--version"])?;
398    classify_version_output(&version).or_else(|| {
399        bounded_output(cli, &["info", "--format", "{{json .}}"])
400            .and_then(|info| classify_info_output(&info))
401    })
402}
403
404/// Classify `<cli> --version` output. Podman prints `podman version 5.2.0`
405/// (also through the `podman-docker` shim), Docker `Docker version 27.3.1,
406/// build ce12230`. Podman is checked first: the shim is podman whatever else
407/// the output mentions.
408pub fn classify_version_output(stdout: &str) -> Option<ContainerEngine> {
409    let lower = stdout.to_ascii_lowercase();
410    if lower.contains("podman") {
411        Some(ContainerEngine::Podman)
412    } else if lower.contains("docker") {
413        Some(ContainerEngine::Docker)
414    } else {
415        None
416    }
417}
418
419/// Classify `<cli> info --format '{{json .}}'` output by engine-specific
420/// fields: podman's host section carries `buildahVersion` / `ociRuntime`
421/// (camelCase), Docker's top level `ServerVersion`.
422pub fn classify_info_output(stdout: &str) -> Option<ContainerEngine> {
423    if stdout.contains("\"buildahVersion\"") || stdout.contains("\"ociRuntime\"") {
424        Some(ContainerEngine::Podman)
425    } else if stdout.contains("\"ServerVersion\"") {
426        Some(ContainerEngine::Docker)
427    } else {
428        None
429    }
430}
431
432/// Detect the Docker bridge gateway IP on Linux. Returns `None` if
433/// detection fails (caller falls back to the conventional `172.17.0.1`).
434///
435/// Goes through [`bounded_output`] like every other container-CLI call here:
436/// `network inspect` talks to the same daemon as the liveness probe, so a
437/// wedged one blocks it on connect forever. This runs inside runtime
438/// constructors on Linux, where an unbounded call hangs server startup outright
439/// -- the exact failure [`CLI_PROBE_TIMEOUT`] exists to prevent. On timeout the
440/// caller just takes the conventional fallback.
441pub fn detect_bridge_gateway(cli: &str) -> Option<String> {
442    let stdout = bounded_output(
443        cli,
444        &[
445            "network",
446            "inspect",
447            "bridge",
448            "--format",
449            "{{range .IPAM.Config}}{{.Gateway}}{{end}}",
450        ],
451    )?;
452    let gateway = stdout.trim().to_string();
453    if gateway.is_empty() || !gateway.contains('.') {
454        return None;
455    }
456    Some(gateway)
457}
458
459/// Resolved container-to-host networking for a given CLI. Built once at
460/// runtime construction and reused for every container spawn.
461#[derive(Debug, Clone)]
462pub struct HostNetworking {
463    /// DNS name a spawned container uses to reach fakecloud on the host.
464    /// `host.containers.internal` for podman, `host.docker.internal` for
465    /// docker.
466    pub host_alias: String,
467    /// `<alias>:<value>` argument for `--add-host`, injected into every
468    /// container `create`/`run`. `None` when the runtime provides the
469    /// alias natively (podman).
470    pub add_host_arg: Option<String>,
471    /// Address fakecloud uses to reach the *sibling* containers it just
472    /// spawned (readiness probes + advertised endpoints). `127.0.0.1`
473    /// when fakecloud runs on the host; `host.docker.internal` when
474    /// fakecloud is itself containerized (`FAKECLOUD_IN_CONTAINER=1`).
475    pub sibling_host: String,
476}
477
478impl HostNetworking {
479    /// Resolve networking for `cli`, reading `FAKECLOUD_IN_CONTAINER` from
480    /// the process environment.
481    pub fn detect(cli: &str) -> Self {
482        let (host_alias, mut add_host_arg) = resolve_host_alias(cli);
483        // A resolving `host.docker.internal` is only trustworthy evidence that
484        // the runtime provides the alias natively (and will inject it into
485        // sibling containers too) when fakecloud is itself containerized:
486        // Docker-Desktop-class runtimes inject the alias into CONTAINERS, never
487        // onto the host. On a bare native-Linux host a resolving alias is
488        // spurious (a hijacking NXDOMAIN resolver, a stray /etc/hosts entry, or
489        // a wildcard search domain), so suppressing the bridge --add-host there
490        // would break the host route sibling containers need. Gate the
491        // suppression on the in-container signal to avoid that regression.
492        let in_container = in_container_mode(std::env::var("FAKECLOUD_IN_CONTAINER").ok());
493        add_host_arg = preserve_native_host_alias(
494            add_host_arg,
495            in_container && host_alias_resolves(&host_alias),
496        );
497        let sibling_host =
498            resolve_sibling_host(&host_alias, std::env::var("FAKECLOUD_IN_CONTAINER").ok());
499        Self {
500            host_alias,
501            add_host_arg,
502            sibling_host,
503        }
504    }
505
506    /// Convenience: append the `--add-host <alias>:<value>` flag pair to a
507    /// growing argv vector when this runtime needs an explicit mapping.
508    /// No-op for podman.
509    pub fn push_add_host_args(&self, argv: &mut Vec<String>) {
510        if let Some(arg) = &self.add_host_arg {
511            argv.push("--add-host".to_string());
512            argv.push(arg.clone());
513        }
514    }
515}
516
517/// How long to wait for the blocking `getaddrinfo` in [`host_alias_resolves`]
518/// before giving up and returning `false`. `getaddrinfo` has no timeout of its
519/// own, and a slow or unreachable DNS server would otherwise block a runtime
520/// thread at startup (this runs inside runtime constructors under
521/// `#[tokio::main]`). Bounding it — same tradeoff as [`CLI_PROBE_TIMEOUT`] —
522/// turns "DNS wedged" into "alias doesn't resolve", the safe default that keeps
523/// the `--add-host` bridge mapping.
524pub const HOST_ALIAS_RESOLVE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(2);
525
526/// True when `host_alias` resolves via the process resolver. The `getaddrinfo`
527/// call is blocking with no timeout of its own, so it runs on a spawned thread
528/// bounded by [`HOST_ALIAS_RESOLVE_TIMEOUT`]; on timeout we return `false` (the
529/// safe default that keeps `--add-host`). A leaked resolver thread on timeout
530/// is acceptable — same tradeoff as [`probe_cli`].
531fn host_alias_resolves(host_alias: &str) -> bool {
532    let (tx, rx) = std::sync::mpsc::channel();
533    let alias = host_alias.to_string();
534    std::thread::spawn(move || {
535        let resolves = std::net::ToSocketAddrs::to_socket_addrs(&(alias.as_str(), 0)).is_ok();
536        let _ = tx.send(resolves);
537    });
538    rx.recv_timeout(HOST_ALIAS_RESOLVE_TIMEOUT).unwrap_or(false)
539}
540
541fn preserve_native_host_alias(
542    add_host_arg: Option<String>,
543    should_suppress: bool,
544) -> Option<String> {
545    if add_host_arg.is_some() && should_suppress {
546        // Suppress the injected `--add-host host.docker.internal:<vm-bridge-ip>`
547        // only when fakecloud is containerized AND the alias already resolves
548        // (see the gate in `detect`). In that case a Docker-Desktop-class
549        // runtime provides `host.docker.internal` natively inside every sibling
550        // container, pointing at the real host; injecting the VM bridge-gateway
551        // IP would shadow it and break the host route. On a bare host — where a
552        // hijacking resolver can make the alias resolve spuriously — the caller
553        // passes `false` here so native Linux docker keeps the bridge mapping
554        // it genuinely needs.
555        None
556    } else {
557        add_host_arg
558    }
559}
560
561/// Compute the `(host_alias, add_host_arg)` pair for a CLI. Pure except
562/// for the bridge-gateway daemon probe on Linux docker, so the macOS /
563/// podman branches are unit-testable without a daemon.
564pub fn resolve_host_alias(cli: &str) -> (String, Option<String>) {
565    if is_podman(cli) {
566        // Podman provides `host.containers.internal` natively on every
567        // supported platform; injecting `host-gateway` on macOS fails
568        // because rootless podman's gvproxy doesn't expose the magic alias.
569        ("host.containers.internal".to_string(), None)
570    } else if cfg!(target_os = "linux") {
571        // Bare docker on Linux: resolve the bridge gateway IP and add an
572        // explicit alias. `host.docker.internal:host-gateway` only works
573        // on Docker Desktop; native Linux docker has no such magic.
574        let ip = detect_bridge_gateway(cli).unwrap_or_else(|| "172.17.0.1".to_string());
575        (
576            "host.docker.internal".to_string(),
577            Some(format!("host.docker.internal:{ip}")),
578        )
579    } else {
580        // Docker Desktop on Mac/Windows: `host-gateway` is the magic alias
581        // that resolves to the host's IP.
582        (
583            "host.docker.internal".to_string(),
584            Some("host.docker.internal:host-gateway".to_string()),
585        )
586    }
587}
588
589/// Decide what address fakecloud uses to reach the sibling containers it
590/// just spawned. Pure helper so the env-var parsing can be tested without
591/// touching the process's real environment.
592///
593/// - `Some("1")` / `Some("true")` (case-insensitive) -> fakecloud is in a
594///   container; the siblings publish their ports on the host's daemon and
595///   are reachable at the same host alias the spawned containers use to
596///   reach fakecloud — `host.docker.internal` under docker,
597///   `host.containers.internal` under podman. Hardcoding
598///   `host.docker.internal` here broke podman, whose gvproxy network only
599///   resolves `host.containers.internal` (issue #1539 follow-up).
600/// - anything else, including `None` -> fakecloud runs on the host,
601///   siblings live on `127.0.0.1:<port>`.
602pub fn resolve_sibling_host(host_alias: &str, env_value: Option<String>) -> String {
603    if in_container_mode(env_value) {
604        host_alias.to_string()
605    } else {
606        "127.0.0.1".to_string()
607    }
608}
609
610/// Parse the `FAKECLOUD_IN_CONTAINER` signal: `Some("1")` or a case-insensitive
611/// `Some("true")` mean fakecloud is running inside a container; anything else,
612/// including `None`, means it runs on the host. Single source of truth for the
613/// parse so `detect`'s native-alias gate and `resolve_sibling_host` can't drift.
614fn in_container_mode(env_value: Option<String>) -> bool {
615    env_value
616        .map(|v| v == "1" || v.eq_ignore_ascii_case("true"))
617        .unwrap_or(false)
618}
619
620/// Hostnames fakecloud's bundled ECR/OCI registry can be addressed from a
621/// sibling container or the host, each at `server_port`.
622///
623/// A container-spawning service rewrites the image pull URI to the runtime's
624/// sibling host -- `host.docker.internal` under Docker, `host.containers.internal`
625/// under podman -- or leaves it `localhost` / `127.0.0.1` when fakecloud runs on
626/// the host (`localhost:<port>` is the documented local ECR endpoint, e.g.
627/// `localhost:4566`). The registry enforces auth, and the Docker/Podman CLI only
628/// attaches the `Authorization` header for hosts present in `config.json`, so the
629/// isolated pull config must list *every* alias or the pull gets a 401. The map
630/// previously omitted the podman alias, so image-based Lambda/ECS pulls failed
631/// under podman-in-a-container (bug-audit 2026-06-20, 0.B2). Authorize all of
632/// them with the same credential; centralized here so the two builders can't
633/// drift again.
634///
635/// A `FAKECLOUD_ECR_REGISTRY_HOST` override is the host the pull URI is
636/// rewritten to (see [`ecr_registry_host`]), so it is authorized as well.
637pub fn registry_auth_hosts(server_port: u16) -> Vec<String> {
638    registry_auth_hosts_with(
639        server_port,
640        std::env::var("FAKECLOUD_ECR_REGISTRY_HOST").ok(),
641    )
642}
643
644/// Pure half of [`registry_auth_hosts`]: the built-in aliases plus a
645/// non-blank registry-host override.
646pub fn registry_auth_hosts_with(server_port: u16, override_host: Option<String>) -> Vec<String> {
647    let mut hosts: Vec<String> = [
648        "localhost",
649        "127.0.0.1",
650        "host.docker.internal",
651        "host.containers.internal",
652    ]
653    .iter()
654    .map(|h| h.to_string())
655    .collect();
656    if let Some(h) = override_host
657        .map(|v| v.trim().to_string())
658        .filter(|v| !v.is_empty())
659    {
660        if !hosts.contains(&h) {
661            hosts.push(h);
662        }
663    }
664    hosts
665        .iter()
666        .map(|host| format!("{host}:{server_port}"))
667        .collect()
668}
669
670/// Host the container engine pulls fakecloud-ECR images from.
671///
672/// The pull is performed by the engine on the *host*, not by a sibling
673/// container, so the sibling host alias is the wrong address for it: Docker
674/// Desktop and OrbStack only accept a plain-HTTP registry over loopback, and
675/// `host.docker.internal` does not resolve on a Linux host at all. Defaults to
676/// `127.0.0.1`. Podman on macOS / Windows pulls inside its machine VM, where
677/// loopback is the VM and only `host.containers.internal` reaches the host.
678/// `FAKECLOUD_ECR_REGISTRY_HOST` overrides both (e.g. when a containerized
679/// fakecloud's port is published under another address).
680pub fn ecr_registry_host(cli: &str) -> String {
681    resolve_ecr_registry_host(
682        std::env::var("FAKECLOUD_ECR_REGISTRY_HOST").ok(),
683        is_podman(cli),
684        cfg!(target_os = "linux"),
685    )
686}
687
688/// Pure half of [`ecr_registry_host`].
689pub fn resolve_ecr_registry_host(env_value: Option<String>, podman: bool, linux: bool) -> String {
690    if let Some(v) = env_value
691        .map(|v| v.trim().to_string())
692        .filter(|v| !v.is_empty())
693    {
694        return v;
695    }
696    if podman && !linux {
697        "host.containers.internal".to_string()
698    } else {
699        "127.0.0.1".to_string()
700    }
701}
702
703/// Host port from `<cli> port <container> <port>` output. Docker prints one
704/// `<ip>:<port>` line per bound address family (`0.0.0.0:49153`,
705/// `[::]:49153`); podman prints the same shape. The first parseable port wins.
706pub fn parse_published_port(output: &str) -> Option<u16> {
707    output
708        .lines()
709        .filter_map(|l| l.trim().rsplit(':').next())
710        .find_map(|p| p.parse::<u16>().ok())
711}
712
713const LOOPBACK_NAMES: [&str; 2] = ["127.0.0.1", "localhost"];
714
715/// Rewrite loopback references in an environment value a sibling container
716/// will read, so they reach the host instead of the container itself.
717///
718/// Inside a container `127.0.0.1` / `localhost` is the container, so every
719/// endpoint fakecloud hands out on loopback -- the server URL, an RDS or
720/// ElastiCache endpoint address, an MSK bootstrap string -- has to name
721/// `target_host` (the host alias) instead. Rewritten:
722///
723/// - the whole value being a loopback host (`DB_HOST=127.0.0.1`);
724/// - a loopback host followed by `:<port>`, at a token boundary -- in URLs
725///   with or without userinfo (`postgres://u:p@localhost:5432/db`), bare
726///   `host:port` values and comma-separated lists
727///   (`127.0.0.1:9092,127.0.0.1:9094`);
728/// - a URL host without a port (`http://localhost/path`).
729///
730/// A port a service registered with
731/// [`crate::dataplane::register_container_port`] is swapped for its
732/// container-view port. Prose such as `EHLO localhost`, and names that merely
733/// contain a loopback name (`127.0.0.10`, `localhost.localdomain`), are left
734/// alone. When `target_host` is itself loopback (fakecloud and the workload
735/// share a network namespace) the value is returned unchanged.
736pub fn rewrite_loopback_value(value: &str, target_host: &str) -> String {
737    if LOOPBACK_NAMES.contains(&target_host) {
738        return value.to_string();
739    }
740    if LOOPBACK_NAMES.contains(&value.trim()) {
741        return value.replacen(value.trim(), target_host, 1);
742    }
743    let bytes = value.as_bytes();
744    let mut out = String::with_capacity(value.len());
745    let mut i = 0;
746    while i < value.len() {
747        let hit = LOOPBACK_NAMES
748            .iter()
749            .find(|name| value[i..].starts_with(**name) && loopback_at(value, i, name.len()));
750        let Some(name) = hit else {
751            let ch = value[i..].chars().next().unwrap_or_default();
752            out.push(ch);
753            i += ch.len_utf8().max(1);
754            continue;
755        };
756        out.push_str(target_host);
757        i += name.len();
758        // Swap a registered host port for its container-view port.
759        if bytes.get(i) == Some(&b':') {
760            let digits: String = value[i + 1..]
761                .chars()
762                .take_while(|c| c.is_ascii_digit())
763                .collect();
764            if let Some(mapped) = digits
765                .parse::<u16>()
766                .ok()
767                .and_then(crate::dataplane::container_port_for)
768            {
769                out.push(':');
770                out.push_str(&mapped.to_string());
771                i += 1 + digits.len();
772            }
773        }
774    }
775    out
776}
777
778/// Whether the loopback name at `value[i..i + len]` is a host reference: at a
779/// token boundary and followed by `:<digit>`, or a URL host (after `//` or
780/// `@`) followed by the end of the authority.
781fn loopback_at(value: &str, i: usize, len: usize) -> bool {
782    let before = value[..i].chars().next_back();
783    let boundary_before = match before {
784        None => true,
785        Some(c) => matches!(c, '/' | '@' | ',' | '=' | ';' | '(' | '"' | '\'') || c.is_whitespace(),
786    };
787    if !boundary_before {
788        return false;
789    }
790    let rest = &value[i + len..];
791    let mut after = rest.chars();
792    match after.next() {
793        Some(':') => after.next().is_some_and(|c| c.is_ascii_digit()),
794        next => {
795            let url_host = matches!(before, Some('@')) || value[..i].ends_with("//");
796            url_host && matches!(next, None | Some('/') | Some('?') | Some('#'))
797        }
798    }
799}
800
801#[cfg(test)]
802mod tests {
803    use super::*;
804
805    #[test]
806    fn cli_available_false_for_missing_binary() {
807        // A binary that doesn't exist fails to spawn -> unavailable, fast.
808        assert!(!cli_available("definitely-not-a-real-cli-binary-xyz-123"));
809    }
810
811    #[cfg(unix)]
812    #[test]
813    fn cli_available_bounds_a_hanging_probe() {
814        // A CLI whose `info` invocation blocks forever (like `docker info`
815        // against an unreachable daemon) must not hang the caller: the probe
816        // is killed at CLI_PROBE_TIMEOUT and reported unavailable. Regression
817        // test for the local-conformance-probe hang.
818        use std::io::Write;
819        use std::os::unix::fs::PermissionsExt;
820
821        let dir = std::env::temp_dir().join(format!("fc-clitest-{}", std::process::id()));
822        std::fs::create_dir_all(&dir).unwrap();
823        let script = dir.join("hangcli");
824        std::fs::write(&script, "#!/bin/sh\nsleep 600\n").unwrap();
825        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
826        std::io::stdout().flush().ok();
827
828        let start = std::time::Instant::now();
829        let available = cli_available(script.to_str().unwrap());
830        let elapsed = start.elapsed();
831
832        std::fs::remove_dir_all(&dir).ok();
833        assert!(!available, "a hanging probe must report unavailable");
834        assert!(
835            elapsed < CLI_PROBE_TIMEOUT + std::time::Duration::from_secs(5),
836            "probe took {elapsed:?}, expected it bounded near {CLI_PROBE_TIMEOUT:?}"
837        );
838    }
839
840    #[test]
841    fn registry_auth_hosts_includes_podman_alias() {
842        // The podman sibling alias (host.containers.internal) must be authorized
843        // or image-based Lambda/ECS pulls 401 under podman-in-a-container (0.B2).
844        let hosts = registry_auth_hosts(4566);
845        assert!(hosts.contains(&"localhost:4566".to_string()));
846        assert!(hosts.contains(&"127.0.0.1:4566".to_string()));
847        assert!(hosts.contains(&"host.docker.internal:4566".to_string()));
848        assert!(
849            hosts.contains(&"host.containers.internal:4566".to_string()),
850            "podman sibling alias must be authorized: {hosts:?}"
851        );
852    }
853
854    #[test]
855    fn name_fast_path_matches_podman_names() {
856        assert!(name_indicates_podman("podman"));
857        assert!(name_indicates_podman("podman-remote"));
858        assert!(name_indicates_podman("/opt/homebrew/bin/podman"));
859        assert!(name_indicates_podman("/usr/local/bin/podman-remote"));
860        assert!(!name_indicates_podman("docker"));
861        assert!(!name_indicates_podman("/usr/local/bin/docker"));
862        assert!(!name_indicates_podman("docker-credential-helper"));
863    }
864
865    #[test]
866    fn name_fast_path_needs_no_probe() {
867        // No binary by this name exists: a podman-named CLI is podman without
868        // running anything.
869        assert!(is_podman("/nonexistent-dir-fc-2599/podman"));
870        assert!(is_podman("podman-remote-definitely-missing-xyz"));
871    }
872
873    #[test]
874    fn version_output_classifies_the_engine() {
875        assert_eq!(
876            classify_version_output("podman version 5.2.0\n"),
877            Some(ContainerEngine::Podman)
878        );
879        assert_eq!(
880            classify_version_output("Docker version 27.3.1, build ce12230\n"),
881            Some(ContainerEngine::Docker)
882        );
883        // The podman-docker shim's notice mentions Docker; it is still podman.
884        assert_eq!(
885            classify_version_output(
886                "Emulate Docker CLI using podman. Create /etc/containers/nodocker to quiet msg.\npodman version 4.9.4\n"
887            ),
888            Some(ContainerEngine::Podman)
889        );
890        assert_eq!(classify_version_output("my-wrapper 1.0\n"), None);
891        assert_eq!(classify_version_output(""), None);
892    }
893
894    #[test]
895    fn info_output_classifies_the_engine() {
896        assert_eq!(
897            classify_info_output(
898                r#"{"host":{"buildahVersion":"1.37.0","ociRuntime":{"name":"crun"}}}"#
899            ),
900            Some(ContainerEngine::Podman)
901        );
902        assert_eq!(
903            classify_info_output(r#"{"ID":"abc","ServerVersion":"27.3.1","Driver":"overlay2"}"#),
904            Some(ContainerEngine::Docker)
905        );
906        assert_eq!(classify_info_output("{}"), None);
907    }
908
909    /// Write an executable fake CLI named `name` in a fresh temp dir, and run
910    /// it once so a concurrent fork can't leave it ETXTBSY for the probe.
911    #[cfg(unix)]
912    fn fake_cli(name: &str, body: &str) -> (tempfile::TempDir, String) {
913        use std::os::unix::fs::PermissionsExt;
914        let dir = tempfile::tempdir().unwrap();
915        let path = dir.path().join(name);
916        std::fs::write(&path, format!("#!/bin/sh\n{body}")).unwrap();
917        std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o755)).unwrap();
918        let mut attempts = 0;
919        loop {
920            match std::process::Command::new(&path).arg("warmup").output() {
921                Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy && attempts < 200 => {
922                    attempts += 1;
923                    std::thread::sleep(std::time::Duration::from_millis(5));
924                }
925                _ => break,
926            }
927        }
928        let cli = path.display().to_string();
929        (dir, cli)
930    }
931
932    #[cfg(unix)]
933    #[test]
934    fn podman_docker_shim_is_detected_as_podman() {
935        // What `podman-docker` installs: a `docker` that is podman.
936        let (_dir, cli) = fake_cli(
937            "docker",
938            "[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
939        );
940        assert!(is_podman(&cli));
941        assert_eq!(probe_engine(&cli), Some(ContainerEngine::Podman));
942    }
943
944    #[cfg(unix)]
945    #[test]
946    fn real_docker_cli_is_not_podman() {
947        let (_dir, cli) = fake_cli(
948            "docker",
949            "[ \"$1\" = --version ] && { echo 'Docker version 27.3.1, build ce12230'; exit 0; }\nexit 1\n",
950        );
951        assert!(!is_podman(&cli));
952    }
953
954    #[cfg(unix)]
955    #[test]
956    fn wrapper_with_opaque_version_falls_back_to_info() {
957        let (_dir, cli) = fake_cli(
958            "container-cli",
959            "case \"$1\" in\n  --version) echo 'wrapper 1.0' ;;\n  info) echo '{\"host\":{\"buildahVersion\":\"1.37.0\"}}' ;;\n  *) exit 1 ;;\nesac\n",
960        );
961        assert!(is_podman(&cli));
962    }
963
964    #[cfg(unix)]
965    #[test]
966    fn engine_probe_is_cached_per_cli() {
967        // Each run appends to a log; a second lookup must not run the CLI again.
968        let (dir, cli) = fake_cli(
969            "docker",
970            "echo \"$*\" >> \"$(dirname \"$0\")/calls.log\"\n[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
971        );
972        let log = dir.path().join("calls.log");
973        let _ = std::fs::remove_file(&log);
974        assert!(is_podman(&cli));
975        assert!(is_podman(&cli));
976        let calls = std::fs::read_to_string(&log).unwrap_or_default();
977        assert_eq!(calls.lines().collect::<Vec<_>>(), ["--version"]);
978    }
979
980    #[cfg(unix)]
981    #[tokio::test]
982    async fn async_probe_detects_the_shim() {
983        let (_dir, cli) = fake_cli(
984            "docker",
985            "[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
986        );
987        assert!(is_podman_async(&cli).await);
988    }
989
990    #[cfg(unix)]
991    #[test]
992    fn engine_probe_bounds_a_hanging_cli() {
993        // A CLI that never answers must not hang detection: one bounded call,
994        // no `info` fallback against it, and it reads as Docker.
995        // Only the probe's commands hang: `fake_cli`'s warmup run (no such
996        // argument) must return at once, or it eats the test's time budget.
997        let (_dir, cli) = fake_cli(
998            "docker",
999            "case \"$1\" in --version|info) sleep 600 ;; esac\nexit 1\n",
1000        );
1001        let start = std::time::Instant::now();
1002        assert!(!is_podman(&cli));
1003        let elapsed = start.elapsed();
1004        assert!(
1005            elapsed < CLI_PROBE_TIMEOUT + std::time::Duration::from_secs(5),
1006            "probe took {elapsed:?}, expected one bound near {CLI_PROBE_TIMEOUT:?}"
1007        );
1008    }
1009
1010    #[test]
1011    fn missing_cli_is_not_podman() {
1012        // `FAKECLOUD_CONTAINER_CLI=false`-style sentinels and missing binaries
1013        // read as not-podman without error.
1014        assert!(!is_podman("definitely-not-a-real-cli-binary-fc-2599"));
1015        assert!(!is_podman("false"));
1016    }
1017
1018    #[cfg(unix)]
1019    #[test]
1020    fn resolve_host_alias_treats_the_shim_as_podman() {
1021        let (_dir, cli) = fake_cli(
1022            "docker",
1023            "[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
1024        );
1025        let (alias, add_host) = resolve_host_alias(&cli);
1026        assert_eq!(alias, "host.containers.internal");
1027        assert_eq!(add_host, None);
1028    }
1029
1030    #[test]
1031    fn resolve_host_alias_podman_has_no_add_host() {
1032        let (alias, add_host) = resolve_host_alias("podman");
1033        assert_eq!(alias, "host.containers.internal");
1034        assert_eq!(add_host, None);
1035        let (alias, add_host) = resolve_host_alias("/opt/homebrew/bin/podman");
1036        assert_eq!(alias, "host.containers.internal");
1037        assert_eq!(add_host, None);
1038    }
1039
1040    #[test]
1041    #[cfg(unix)]
1042    fn resolve_host_alias_docker_emits_add_host() {
1043        // A fake Docker CLI, so a host whose `docker` is the podman-docker shim
1044        // can't flip this test.
1045        let (_dir, cli) = fake_cli(
1046            "docker",
1047            "[ \"$1\" = --version ] && { echo 'Docker version 27.3.1, build ce12230'; exit 0; }\nexit 1\n",
1048        );
1049        let (alias, add_host) = resolve_host_alias(&cli);
1050        assert_eq!(alias, "host.docker.internal");
1051        // On macOS this is host-gateway; on Linux it's a bridge IP. Either
1052        // way docker must get an explicit --add-host.
1053        assert!(add_host.is_some());
1054        assert!(add_host.unwrap().starts_with("host.docker.internal:"));
1055    }
1056
1057    #[test]
1058    fn native_host_alias_prevents_docker_add_host_override() {
1059        let add_host =
1060            preserve_native_host_alias(Some("host.docker.internal:host-gateway".to_string()), true);
1061
1062        assert_eq!(add_host, None);
1063    }
1064
1065    #[test]
1066    fn unresolved_host_alias_keeps_docker_add_host() {
1067        let add_host = preserve_native_host_alias(
1068            Some("host.docker.internal:host-gateway".to_string()),
1069            false,
1070        );
1071
1072        assert_eq!(
1073            add_host.as_deref(),
1074            Some("host.docker.internal:host-gateway")
1075        );
1076    }
1077
1078    #[test]
1079    fn absent_docker_add_host_remains_absent() {
1080        assert_eq!(preserve_native_host_alias(None, true), None);
1081        assert_eq!(preserve_native_host_alias(None, false), None);
1082    }
1083
1084    #[test]
1085    fn in_container_mode_parses_truthy_values() {
1086        assert!(in_container_mode(Some("1".to_string())));
1087        assert!(in_container_mode(Some("true".to_string())));
1088        assert!(in_container_mode(Some("True".to_string())));
1089        assert!(in_container_mode(Some("TRUE".to_string())));
1090    }
1091
1092    #[test]
1093    fn in_container_mode_rejects_falsey_and_absent() {
1094        assert!(!in_container_mode(None));
1095        assert!(!in_container_mode(Some(String::new())));
1096        assert!(!in_container_mode(Some("0".to_string())));
1097        assert!(!in_container_mode(Some("false".to_string())));
1098        assert!(!in_container_mode(Some("yes".to_string())));
1099    }
1100
1101    #[test]
1102    fn native_alias_gate_suppresses_only_in_container() {
1103        // The gate `detect` computes: `in_container && host_alias_resolves`.
1104        let add_host = || Some("host.docker.internal:172.17.0.1".to_string());
1105
1106        // In-container + resolves -> Desktop-class runtime provides the alias
1107        // natively in siblings; drop the shadowing bridge mapping.
1108        let in_container = true;
1109        let resolves = true;
1110        assert_eq!(
1111            preserve_native_host_alias(add_host(), in_container && resolves),
1112            None,
1113        );
1114
1115        // NOT in-container (bare host) + resolves -> the resolving alias is
1116        // spurious (hijacking resolver / stray hosts entry). Native Linux docker
1117        // needs the bridge mapping; must NOT drop it. Regression guard.
1118        let in_container = false;
1119        let resolves = true;
1120        assert_eq!(
1121            preserve_native_host_alias(add_host(), in_container && resolves).as_deref(),
1122            Some("host.docker.internal:172.17.0.1"),
1123        );
1124
1125        // In-container + does NOT resolve -> nothing native to preserve; keep
1126        // the injected mapping.
1127        let in_container = true;
1128        let resolves = false;
1129        assert_eq!(
1130            preserve_native_host_alias(add_host(), in_container && resolves).as_deref(),
1131            Some("host.docker.internal:172.17.0.1"),
1132        );
1133    }
1134
1135    #[test]
1136    fn resolve_sibling_host_defaults_to_loopback() {
1137        assert_eq!(
1138            resolve_sibling_host("host.docker.internal", None),
1139            "127.0.0.1"
1140        );
1141        assert_eq!(
1142            resolve_sibling_host("host.docker.internal", Some(String::new())),
1143            "127.0.0.1"
1144        );
1145        assert_eq!(
1146            resolve_sibling_host("host.docker.internal", Some("0".to_string())),
1147            "127.0.0.1"
1148        );
1149        assert_eq!(
1150            resolve_sibling_host("host.containers.internal", Some("false".to_string())),
1151            "127.0.0.1"
1152        );
1153    }
1154
1155    #[test]
1156    fn resolve_sibling_host_uses_host_alias_when_in_container() {
1157        // Docker: siblings reachable at host.docker.internal.
1158        assert_eq!(
1159            resolve_sibling_host("host.docker.internal", Some("1".to_string())),
1160            "host.docker.internal"
1161        );
1162        assert_eq!(
1163            resolve_sibling_host("host.docker.internal", Some("true".to_string())),
1164            "host.docker.internal"
1165        );
1166        assert_eq!(
1167            resolve_sibling_host("host.docker.internal", Some("TRUE".to_string())),
1168            "host.docker.internal"
1169        );
1170        // Podman: must use host.containers.internal, NOT host.docker.internal
1171        // (issue #1539 follow-up — gvproxy only resolves the containers alias).
1172        assert_eq!(
1173            resolve_sibling_host("host.containers.internal", Some("1".to_string())),
1174            "host.containers.internal"
1175        );
1176    }
1177
1178    #[test]
1179    fn detect_wires_sibling_host_to_podman_alias_in_container() {
1180        // Full path: a podman binary in a container must advertise siblings
1181        // at host.containers.internal. resolve_host_alias drives host_alias,
1182        // which resolve_sibling_host then reuses.
1183        let (alias, add_host) = resolve_host_alias("podman");
1184        assert_eq!(alias, "host.containers.internal");
1185        assert_eq!(add_host, None);
1186        assert_eq!(
1187            resolve_sibling_host(&alias, Some("1".to_string())),
1188            "host.containers.internal"
1189        );
1190    }
1191
1192    #[test]
1193    fn only_objects_of_a_dead_owner_are_orphans() {
1194        let me = std::process::id();
1195        let alive = |pid: u32| pid == 4242;
1196        // Another live fakecloud process: never an orphan.
1197        assert!(!owned_by_dead_process("fakecloud-4242", alive));
1198        // Its owner is gone: an orphan.
1199        assert!(owned_by_dead_process("fakecloud-777", alive));
1200        // The current process, even if the probe says otherwise.
1201        assert!(!owned_by_dead_process(&format!("fakecloud-{me}"), |_| {
1202            false
1203        }));
1204        // Nothing proves an unparseable owner is gone.
1205        for label in ["", "fakecloud-", "fakecloud-abc", "other-777"] {
1206            assert!(!owned_by_dead_process(label, alive), "{label:?}");
1207        }
1208    }
1209
1210    #[cfg(unix)]
1211    #[test]
1212    fn pid_alive_probes_real_processes() {
1213        assert!(pid_alive(std::process::id()));
1214        assert!(!pid_alive(u32::MAX - 1));
1215    }
1216
1217    #[test]
1218    fn push_add_host_args_noop_for_podman() {
1219        let net = HostNetworking {
1220            host_alias: "host.containers.internal".to_string(),
1221            add_host_arg: None,
1222            sibling_host: "127.0.0.1".to_string(),
1223        };
1224        let mut argv = vec!["create".to_string()];
1225        net.push_add_host_args(&mut argv);
1226        assert_eq!(argv, vec!["create".to_string()]);
1227    }
1228
1229    #[test]
1230    fn push_add_host_args_emits_for_docker() {
1231        let net = HostNetworking {
1232            host_alias: "host.docker.internal".to_string(),
1233            add_host_arg: Some("host.docker.internal:host-gateway".to_string()),
1234            sibling_host: "127.0.0.1".to_string(),
1235        };
1236        let mut argv = vec!["create".to_string()];
1237        net.push_add_host_args(&mut argv);
1238        assert_eq!(
1239            argv,
1240            vec![
1241                "create".to_string(),
1242                "--add-host".to_string(),
1243                "host.docker.internal:host-gateway".to_string(),
1244            ]
1245        );
1246    }
1247}
1248
1249#[cfg(test)]
1250mod bounded_cli_tests {
1251    use super::*;
1252
1253    /// A wedged daemon leaves the CLI blocked on connect forever. Every
1254    /// container call has to end at the bound instead of hanging its caller,
1255    /// which for the reaper means hanging server startup.
1256    #[test]
1257    fn a_hanging_cli_call_is_cut_off() {
1258        let start = std::time::Instant::now();
1259        let mut child = spawn_bounded(
1260            std::process::Command::new("sleep")
1261                .arg("600")
1262                .stdout(std::process::Stdio::null())
1263                .stderr(std::process::Stdio::null()),
1264        )
1265        .expect("sleep is available");
1266        assert!(!wait_bounded_group(&mut child));
1267        assert!(
1268            start.elapsed() < CLI_PROBE_TIMEOUT + std::time::Duration::from_secs(5),
1269            "the wait must end at the bound"
1270        );
1271    }
1272
1273    /// Output larger than a pipe buffer (64 KiB on Linux) must come back
1274    /// whole. Waiting for the child to exit before reading blocks it on write
1275    /// forever, so this used to burn the full timeout and report failure.
1276    #[test]
1277    fn output_larger_than_the_pipe_buffer_still_comes_back() {
1278        let start = std::time::Instant::now();
1279        // 200_000 bytes: comfortably past the buffer on every supported host.
1280        let out = bounded_output("sh", &["-c", "printf 'x%.0s' $(seq 1 200000)"])
1281            .expect("a large but prompt call must succeed");
1282        assert_eq!(out.len(), 200_000, "output was truncated");
1283        assert!(
1284            start.elapsed() < CLI_PROBE_TIMEOUT,
1285            "a prompt call must not reach the deadline"
1286        );
1287    }
1288
1289    #[test]
1290    fn a_prompt_cli_call_returns_its_output() {
1291        assert_eq!(
1292            bounded_output("echo", &["abc123"])
1293                .as_deref()
1294                .map(str::trim),
1295            Some("abc123")
1296        );
1297        assert!(bounded_status("true", &[]));
1298        assert!(!bounded_status("false", &[]));
1299    }
1300
1301    /// A bounded call gets an empty stdin, never fakecloud's own. The process
1302    /// group `spawn_bounded` creates is a *background* one, so a child that
1303    /// reads the controlling terminal -- a `sudo`/credential-helper wrapper
1304    /// prompting -- is stopped by SIGTTIN, which the WNOHANG `try_wait` loop
1305    /// cannot see: the call would burn the whole deadline and be killed.
1306    /// `cat` with no argument reads stdin to EOF, so it returns at once with
1307    /// nothing only when stdin is /dev/null.
1308    #[cfg(unix)]
1309    #[test]
1310    fn a_bounded_call_reads_an_empty_stdin() {
1311        let start = std::time::Instant::now();
1312        let out = bounded_output("cat", &[]).expect("a call reading stdin must not time out");
1313        assert!(out.is_empty(), "stdin must be empty, got {out:?}");
1314        assert!(
1315            start.elapsed() < CLI_PROBE_TIMEOUT,
1316            "a call reading stdin must not reach the deadline"
1317        );
1318    }
1319
1320    /// The happy path must still hand back the child's output *and* collect the
1321    /// reader, so the no-leak guarantee isn't bought by dropping output.
1322    #[test]
1323    fn a_prompt_cli_call_collects_its_reader() {
1324        let (output, reader) = run_bounded("echo", &["abc123"]);
1325        assert_eq!(output.as_deref().map(str::trim), Some("abc123"));
1326        assert!(
1327            matches!(reader, ReaderState::Finished),
1328            "reader was {reader:?}, expected it collected"
1329        );
1330    }
1331
1332    /// The bridge-gateway probe talks to the same daemon as the liveness probe,
1333    /// so it has to end at the same bound. It used to be a plain
1334    /// `Command::output()`, which against a wedged daemon hung whichever runtime
1335    /// constructor called it -- on Linux, every container-backed service at
1336    /// server startup. Timing out is not an error here: the caller falls back to
1337    /// the conventional `172.17.0.1`.
1338    #[cfg(unix)]
1339    #[test]
1340    fn a_hanging_bridge_gateway_probe_is_cut_off() {
1341        use std::os::unix::fs::PermissionsExt;
1342
1343        let dir = std::env::temp_dir().join(format!("fc-gwtest-{}", std::process::id()));
1344        std::fs::create_dir_all(&dir).unwrap();
1345        let script = dir.join("hangcli");
1346        std::fs::write(&script, "#!/bin/sh\nsleep 600\n").unwrap();
1347        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
1348
1349        let start = std::time::Instant::now();
1350        let gateway = detect_bridge_gateway(script.to_str().unwrap());
1351        let elapsed = start.elapsed();
1352
1353        std::fs::remove_dir_all(&dir).ok();
1354        assert_eq!(gateway, None, "a wedged daemon must report no gateway");
1355        assert!(
1356            elapsed < CLI_PROBE_TIMEOUT + READER_DRAIN_GRACE + std::time::Duration::from_secs(5),
1357            "the probe took {elapsed:?}, expected it bounded near {CLI_PROBE_TIMEOUT:?}"
1358        );
1359    }
1360
1361    /// The unavailable-CLI path keeps its shape: nothing to spawn, no gateway,
1362    /// and the caller's fallback stands.
1363    #[test]
1364    fn a_missing_cli_reports_no_bridge_gateway() {
1365        assert_eq!(
1366            detect_bridge_gateway("definitely-not-a-real-cli-binary-xyz-123"),
1367            None
1368        );
1369    }
1370
1371    /// A CLI that succeeds with no output -- an `inspect --format` over a bridge
1372    /// with no IPAM config -- still means "no gateway", not an empty
1373    /// `--add-host` value. Unchanged by the bounding; guarded so it stays that
1374    /// way.
1375    #[test]
1376    fn an_empty_gateway_is_rejected() {
1377        assert_eq!(detect_bridge_gateway("true"), None);
1378    }
1379
1380    /// `FAKECLOUD_CONTAINER_CLI` is routinely a wrapper (`sh -c 'exec docker
1381    /// "$@"'`, a `podman-remote` shim), which makes the real command a
1382    /// grandchild holding the stdout pipe. Killing only the direct child left
1383    /// the reader's `read_to_end` blocked forever -- a thread parked for the
1384    /// life of the process, once per call, on exactly the wedged-daemon path
1385    /// these bounds were added for (the server reaper calls this at startup).
1386    /// The reader reporting in is the evidence: EOF on that pipe is only
1387    /// possible once every write end is closed, so a collected buffer proves
1388    /// the grandchildren went down with the call.
1389    #[cfg(unix)]
1390    #[test]
1391    fn a_timed_out_wrapper_call_leaves_no_reader_behind() {
1392        let start = std::time::Instant::now();
1393        // A wrapper that outlives its own kill: the backgrounded sleep inherits
1394        // the stdout pipe and is not the process we spawned.
1395        let (output, reader) = run_bounded("sh", &["-c", "sleep 600 & sleep 600"]);
1396        assert_eq!(output, None, "a wedged call must report failure");
1397        assert!(
1398            matches!(reader, ReaderState::Finished),
1399            "reader was {reader:?}: the stdout reader must not outlive the call"
1400        );
1401        assert!(
1402            start.elapsed()
1403                < CLI_PROBE_TIMEOUT + READER_DRAIN_GRACE + std::time::Duration::from_secs(5),
1404            "the call must still end at the bound, took {:?}",
1405            start.elapsed()
1406        );
1407    }
1408}
1409
1410#[cfg(test)]
1411mod endpoint_tests {
1412    use super::*;
1413
1414    #[test]
1415    fn ecr_registry_host_defaults_to_loopback() {
1416        // The host daemon performs the pull; the sibling alias is wrong there.
1417        assert_eq!(resolve_ecr_registry_host(None, false, true), "127.0.0.1");
1418        assert_eq!(resolve_ecr_registry_host(None, false, false), "127.0.0.1");
1419        assert_eq!(
1420            resolve_ecr_registry_host(Some("  ".into()), false, true),
1421            "127.0.0.1"
1422        );
1423        // Podman on Linux pulls on the host; podman machine pulls in its VM.
1424        assert_eq!(resolve_ecr_registry_host(None, true, true), "127.0.0.1");
1425        assert_eq!(
1426            resolve_ecr_registry_host(None, true, false),
1427            "host.containers.internal"
1428        );
1429        assert_eq!(
1430            resolve_ecr_registry_host(Some("10.0.0.5".into()), true, false),
1431            "10.0.0.5"
1432        );
1433    }
1434
1435    #[test]
1436    fn registry_override_host_is_authorized() {
1437        // A pull rewritten to FAKECLOUD_ECR_REGISTRY_HOST must carry auth, or
1438        // the registry answers 401.
1439        let hosts = registry_auth_hosts_with(4566, Some("10.1.2.3".into()));
1440        assert!(hosts.contains(&"10.1.2.3:4566".to_string()), "{hosts:?}");
1441        assert!(hosts.contains(&"127.0.0.1:4566".to_string()));
1442        // No duplicates, blank override ignored.
1443        assert_eq!(
1444            registry_auth_hosts_with(4566, Some("127.0.0.1".into())).len(),
1445            4
1446        );
1447        assert_eq!(registry_auth_hosts_with(4566, Some(" ".into())).len(), 4);
1448    }
1449
1450    #[test]
1451    fn published_port_parses_docker_and_podman_output() {
1452        assert_eq!(
1453            parse_published_port("0.0.0.0:49153\n[::]:49153\n"),
1454            Some(49153)
1455        );
1456        assert_eq!(parse_published_port("[::]:5000"), Some(5000));
1457        assert_eq!(parse_published_port(""), None);
1458        assert_eq!(parse_published_port("garbage"), None);
1459    }
1460
1461    #[test]
1462    fn rewrite_loopback_urls_and_bare_endpoints() {
1463        let h = "host.docker.internal";
1464        assert_eq!(
1465            rewrite_loopback_value("http://localhost:4566", h),
1466            "http://host.docker.internal:4566"
1467        );
1468        assert_eq!(
1469            rewrite_loopback_value("https://127.0.0.1:4566/path", h),
1470            "https://host.docker.internal:4566/path"
1471        );
1472        // Bare host (an RDS / ElastiCache endpoint address).
1473        assert_eq!(rewrite_loopback_value("127.0.0.1", h), h);
1474        assert_eq!(rewrite_loopback_value("localhost", h), h);
1475        // Bare host:port and comma-separated broker lists.
1476        assert_eq!(
1477            rewrite_loopback_value("127.0.0.1:6379", h),
1478            "host.docker.internal:6379"
1479        );
1480        assert_eq!(
1481            rewrite_loopback_value("127.0.0.1:9092,localhost:9094", h),
1482            "host.docker.internal:9092,host.docker.internal:9094"
1483        );
1484        // Scheme-qualified DSNs with userinfo.
1485        assert_eq!(
1486            rewrite_loopback_value("postgres://u:p@localhost:5432/db", h),
1487            "postgres://u:p@host.docker.internal:5432/db"
1488        );
1489        assert_eq!(
1490            rewrite_loopback_value("redis://127.0.0.1:6379/0", h),
1491            "redis://host.docker.internal:6379/0"
1492        );
1493        // URL host without a port.
1494        assert_eq!(
1495            rewrite_loopback_value("http://localhost/health", h),
1496            "http://host.docker.internal/health"
1497        );
1498        // key=value lists.
1499        assert_eq!(
1500            rewrite_loopback_value("host=127.0.0.1:5432 user=x", h),
1501            "host=host.docker.internal:5432 user=x"
1502        );
1503    }
1504
1505    #[test]
1506    fn rewrite_loopback_leaves_prose_and_lookalikes_alone() {
1507        let h = "host.docker.internal";
1508        for v in [
1509            "EHLO localhost",
1510            "connect to localhost soon",
1511            "127.0.0.10:80",
1512            "localhost.localdomain:25",
1513            "mylocalhost:80",
1514            "",
1515        ] {
1516            assert_eq!(rewrite_loopback_value(v, h), v, "{v:?}");
1517        }
1518        // Workload sharing fakecloud's namespace: nothing to rewrite.
1519        assert_eq!(
1520            rewrite_loopback_value("http://localhost:4566", "127.0.0.1"),
1521            "http://localhost:4566"
1522        );
1523    }
1524
1525    #[test]
1526    fn rewrite_loopback_swaps_registered_container_ports() {
1527        // A broker whose host listener advertises loopback registers the
1528        // listener containers must use instead.
1529        crate::dataplane::register_container_port(41001, 41002);
1530        assert_eq!(
1531            rewrite_loopback_value("127.0.0.1:41001", "host.docker.internal"),
1532            "host.docker.internal:41002"
1533        );
1534        crate::dataplane::unregister_container_port(41001);
1535        assert_eq!(
1536            rewrite_loopback_value("127.0.0.1:41001", "host.docker.internal"),
1537            "host.docker.internal:41001"
1538        );
1539    }
1540}