fakecloud_core/container_net.rs
1//! Shared container-to-host networking resolution for service runtimes
2//! that spawn sibling containers (Lambda, ECS, RDS, ElastiCache).
3//!
4//! Captures the issue #1539 fix shape in one place so the four runtimes
5//! that shell out to `docker`/`podman` can't drift apart again:
6//!
7//! - **podman** ships `host.containers.internal` as a built-in container
8//! DNS entry on every platform and must NOT receive
9//! `--add-host host.docker.internal:host-gateway` — rootless podman's
10//! gvproxy leaves the magic alias empty and the `create` fails with
11//! "host containers internal IP address is empty".
12//! - **bare docker on Linux** has no `host-gateway` magic; the bridge
13//! gateway IP has to be resolved from the daemon and injected explicitly.
14//! - **Docker Desktop on Mac/Windows** resolves the `host-gateway` magic
15//! value to the host's IP.
16//! - when fakecloud itself runs in a container (`FAKECLOUD_IN_CONTAINER=1`,
17//! baked into the published image), the sibling containers it spawns
18//! publish their ports on the *host's* daemon — reachable from inside
19//! fakecloud's container as `host.docker.internal:<port>`, not
20//! `127.0.0.1:<port>`.
21
22/// Actionable remediation appended to every error raised when a container
23/// runtime (Docker/Podman) is required for an operation but none is
24/// available. Kept in one place so RDS, Lambda, ECS, and the server startup
25/// banner all surface the same fix steps and can't drift apart.
26pub const CONTAINER_RUNTIME_HINT: &str = "Install and start Docker or Podman, or set FAKECLOUD_CONTAINER_CLI to your container CLI path.";
27
28/// Auto-detect an available container CLI. Honors `FAKECLOUD_CONTAINER_CLI`
29/// as an explicit override (returns `None` if the override doesn't work),
30/// otherwise prefers `docker` then `podman`. Returns `None` when neither
31/// is usable.
32pub fn detect_container_cli() -> Option<String> {
33 if let Ok(cli) = std::env::var("FAKECLOUD_CONTAINER_CLI") {
34 return if cli_available(&cli) { Some(cli) } else { None };
35 }
36 if cli_available("docker") {
37 Some("docker".to_string())
38 } else if cli_available("podman") {
39 Some("podman".to_string())
40 } else {
41 None
42 }
43}
44
45/// How long to wait for `<cli> info` before giving up and treating the
46/// runtime as unavailable. A healthy daemon answers in well under a second;
47/// an unreachable or wedged daemon (stale `DOCKER_HOST`, Docker Desktop mid
48/// start, a broken socket) can leave the CLI blocked on connect *forever*,
49/// which would hang fakecloud startup and the test harness. Bounding the
50/// probe turns "daemon wedged" into "no runtime detected" instead of a hang.
51pub const CLI_PROBE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10);
52
53/// Process-global memo of `<cli> info` results, keyed by CLI name/path.
54///
55/// Container-runtime liveness is fixed for the life of a process, but every
56/// service runtime (Lambda, ECS, RDS, ElastiCache, EC2, MQ, MSK, ...) probes
57/// it independently at startup — a dozen-plus `detect_container_cli()` calls.
58/// Without a memo each probe re-runs `docker info`; when the daemon is wedged
59/// (see [`CLI_PROBE_TIMEOUT`]) those probes are serial 10s hangs that stack
60/// into minutes, wedging server startup and the conformance `*_probe` tests.
61/// Caching the first answer collapses that to a single probe.
62static CLI_AVAILABLE_CACHE: std::sync::OnceLock<
63 std::sync::Mutex<std::collections::HashMap<String, bool>>,
64> = std::sync::OnceLock::new();
65
66/// True when the CLI responds to `<cli> info` with success within
67/// [`CLI_PROBE_TIMEOUT`] — the same liveness probe every runtime used before
68/// this module existed, but bounded so an unreachable daemon can't hang the
69/// caller indefinitely (the CLI blocks on connect with no timeout of its own),
70/// and memoized per process so a dozen runtimes probing at startup don't each
71/// pay that bound.
72pub fn cli_available(cli: &str) -> bool {
73 let cache =
74 CLI_AVAILABLE_CACHE.get_or_init(|| std::sync::Mutex::new(std::collections::HashMap::new()));
75 if let Some(&cached) = cache.lock().unwrap().get(cli) {
76 return cached;
77 }
78 let result = probe_cli(cli);
79 cache.lock().unwrap().insert(cli.to_string(), result);
80 result
81}
82
83/// Run the bounded `<cli> info` liveness probe once (uncached).
84fn probe_cli(cli: &str) -> bool {
85 let child = spawn_bounded(
86 std::process::Command::new(cli)
87 .arg("info")
88 .stdout(std::process::Stdio::null())
89 .stderr(std::process::Stdio::null()),
90 );
91 let Ok(mut child) = child else {
92 return false;
93 };
94 wait_bounded_group(&mut child) && child.wait().map(|s| s.success()).unwrap_or(false)
95}
96
97/// Spawn a container-CLI command in a process group of its own (Unix), so a
98/// timed-out call can be torn down whole. `FAKECLOUD_CONTAINER_CLI` is
99/// routinely a wrapper -- `sh -c 'exec docker "$@"'`, a `podman-remote` shim --
100/// which makes the real command a *grandchild*: it survives `Child::kill`, goes
101/// on holding whatever pipes we handed it, and keeps running against a wedged
102/// daemon forever. Its own group makes it reachable by a single signal.
103/// Detaching these from terminal job control is fine: their lifetime is managed
104/// by deadline here, not by the shell fakecloud was started from.
105///
106/// stdin is /dev/null, and has to be: a new process group is a *background*
107/// one, so a child that reads the controlling terminal -- a `sudo` or
108/// credential-helper wrapper prompting, exactly the wrapper case above -- takes
109/// SIGTTIN, which stops it rather than ending it. `try_wait` is WNOHANG without
110/// WUNTRACED, so the loop below never sees a stopped child and the call burns
111/// the whole deadline before being killed. These calls are non-interactive
112/// anyway, so an immediate EOF is the right answer for them.
113fn spawn_bounded(cmd: &mut std::process::Command) -> std::io::Result<std::process::Child> {
114 cmd.stdin(std::process::Stdio::null());
115 #[cfg(unix)]
116 {
117 std::os::unix::process::CommandExt::process_group(cmd, 0);
118 }
119 cmd.spawn()
120}
121
122/// Wait for `child` up to [`CLI_PROBE_TIMEOUT`], killing it on expiry. Returns
123/// whether it exited on its own. Every container-CLI call goes through this:
124/// a liveness probe answering does not promise the next call will, and an
125/// unbounded one blocks the caller rather than just that command.
126///
127/// Only for a child from [`spawn_bounded`], which put it in a group of its own:
128/// the expiry kill hits that whole group, so a wrapper CLI's grandchildren die
129/// with it.
130fn wait_bounded_group(child: &mut std::process::Child) -> bool {
131 let deadline = std::time::Instant::now() + CLI_PROBE_TIMEOUT;
132 loop {
133 match child.try_wait() {
134 Ok(Some(_)) => return true,
135 Ok(None) => {}
136 Err(_) => return false,
137 }
138 if std::time::Instant::now() >= deadline {
139 // Daemon is wedged: kill the blocked call and report failure.
140 kill_expired(child);
141 let _ = child.wait();
142 return false;
143 }
144 std::thread::sleep(std::time::Duration::from_millis(25));
145 }
146}
147
148/// SIGKILL a timed-out child and its process group. [`spawn_bounded`] made the
149/// child its own group leader, so the group id is the child's pid, and the
150/// child is still unreaped here -- the pid cannot have been recycled and the
151/// signal cannot stray onto an unrelated group.
152#[cfg(unix)]
153fn kill_expired(child: &mut std::process::Child) {
154 // SAFETY: `kill` with a negative pid targets the process group of that
155 // id; any pid value is safe to pass.
156 let _ = unsafe { libc::kill(-(child.id() as libc::pid_t), libc::SIGKILL) };
157 let _ = child.kill();
158}
159
160/// Windows has no process-group signal (a job object would be needed), so the
161/// direct child is as far as the kill reaches; [`run_bounded`] still bounds the
162/// wait on its stdout reader so the caller can't be held by a surviving
163/// grandchild.
164#[cfg(not(unix))]
165fn kill_expired(child: &mut std::process::Child) {
166 let _ = child.kill();
167}
168
169/// Whether the stdout reader thread ended before the call returned.
170#[derive(Debug)]
171enum ReaderState {
172 /// The reader returned; its thread is gone.
173 Finished,
174 /// The reader is still blocked on the pipe because a write end we could not
175 /// close is held outside the child's process group. The thread outlives the
176 /// call; the caller does not wait for it.
177 Abandoned,
178}
179
180/// Floor on how long [`run_bounded`] waits for its stdout reader once the call
181/// is over (it also gets whatever is left of the call's own budget). Both exits
182/// close every write end we control -- the child exited, or its whole process
183/// group was killed -- which ends the blocked `read_to_end` at once, so this
184/// covers scheduling only. It exists so a write end held somewhere we cannot
185/// reach costs the caller a few hundred milliseconds instead of blocking it for
186/// good, which is what an unbounded join did.
187const READER_DRAIN_GRACE: std::time::Duration = std::time::Duration::from_millis(500);
188
189/// Run a container-CLI command and return its stdout, or `None` when it fails
190/// or outruns [`CLI_PROBE_TIMEOUT`].
191pub fn bounded_output(cli: &str, args: &[&str]) -> Option<String> {
192 run_bounded(cli, args).0
193}
194
195/// [`bounded_output`], plus whether its stdout reader finished -- so the
196/// timeout path's "no reader left behind" guarantee is unit-testable instead of
197/// only observable as a thread that never goes away.
198fn run_bounded(cli: &str, args: &[&str]) -> (Option<String>, ReaderState) {
199 let deadline = std::time::Instant::now() + CLI_PROBE_TIMEOUT;
200 let child = spawn_bounded(
201 std::process::Command::new(cli)
202 .args(args)
203 .stdout(std::process::Stdio::piped())
204 .stderr(std::process::Stdio::null()),
205 );
206 let Ok(mut child) = child else {
207 return (None, ReaderState::Finished);
208 };
209 let Some(mut stdout) = child.stdout.take() else {
210 kill_expired(&mut child);
211 let _ = child.wait();
212 return (None, ReaderState::Finished);
213 };
214 // Drain stdout while waiting. A child whose output outgrows the pipe
215 // buffer blocks on write until someone reads it, so waiting for exit
216 // first would deadlock until the deadline and then report the sweep as
217 // failed -- `docker ps -a` across a busy host is exactly that much output.
218 //
219 // The channel doubles as the reader's "I'm done" signal: the send is the
220 // last thing the thread does before dropping the pipe's read end, so a
221 // received buffer proves no reader is parked behind us. A `JoinHandle`
222 // can't say that without blocking, which on the timeout path is exactly
223 // what we must not do.
224 let (tx, rx) = std::sync::mpsc::channel();
225 std::thread::spawn(move || {
226 let mut buf = Vec::new();
227 let _ = std::io::Read::read_to_end(&mut stdout, &mut buf);
228 let _ = tx.send(buf);
229 });
230 // On expiry `wait_bounded_group` has killed the whole process group, so a
231 // wrapper CLI's grandchild releases the write end and the reader returns
232 // instead of blocking for the life of the process -- one leaked thread per
233 // call, on precisely the wedged-daemon path these bounds exist for.
234 let exited = wait_bounded_group(&mut child);
235 let status = child.wait().ok();
236 // Whatever is left of the call's own budget, and never less than the grace:
237 // a prompt call can afford to wait out a reader thread the scheduler hasn't
238 // run yet, a timed-out one gets only the grace, and either way the caller is
239 // back within CLI_PROBE_TIMEOUT plus that grace.
240 let grace = deadline
241 .saturating_duration_since(std::time::Instant::now())
242 .max(READER_DRAIN_GRACE);
243 let drained = rx.recv_timeout(grace).ok();
244 let output = match (exited, status, &drained) {
245 (true, Some(status), Some(buf)) if status.success() => {
246 Some(String::from_utf8_lossy(buf).into_owned())
247 }
248 _ => None,
249 };
250 let reader = if drained.is_some() {
251 ReaderState::Finished
252 } else {
253 ReaderState::Abandoned
254 };
255 (output, reader)
256}
257
258/// Run a container-CLI command for its effect only, bounded the same way.
259/// Returns whether it succeeded.
260pub fn bounded_status(cli: &str, args: &[&str]) -> bool {
261 let Ok(mut child) = spawn_bounded(
262 std::process::Command::new(cli)
263 .args(args)
264 .stdout(std::process::Stdio::null())
265 .stderr(std::process::Stdio::null()),
266 ) else {
267 return false;
268 };
269 wait_bounded_group(&mut child) && child.wait().map(|s| s.success()).unwrap_or(false)
270}
271
272/// True if the given PID is a live process on this host.
273///
274/// On Unix this is `kill(pid, 0)`: it returns 0 if the process exists
275/// (including zombies), or sets `errno` to `ESRCH` if not. On non-Unix
276/// platforms it conservatively returns `true`, so a caller never removes a
277/// resource it can't prove is orphaned.
278#[cfg(unix)]
279pub fn pid_alive(pid: u32) -> bool {
280 // SAFETY: `kill` with signal 0 is a liveness probe; it does not
281 // actually deliver a signal. Any PID value is safe to pass.
282 let rc = unsafe { libc::kill(pid as libc::pid_t, 0) };
283 if rc == 0 {
284 return true;
285 }
286 // errno == EPERM means the process exists but we can't signal it —
287 // still alive from our perspective.
288 std::io::Error::last_os_error().raw_os_error() == Some(libc::EPERM)
289}
290
291#[cfg(not(unix))]
292pub fn pid_alive(_pid: u32) -> bool {
293 true
294}
295
296/// Whether a container or network labelled `fakecloud-instance=<label>` was
297/// left behind by a fakecloud process that is gone. The label is
298/// `fakecloud-<pid>`; an object is orphaned only when that PID is neither the
299/// current process nor alive. Several fakecloud processes can share one
300/// daemon (parallel test servers, side-by-side installs), so an object owned
301/// by *another live* process is never an orphan. A label that doesn't parse is
302/// not treated as an orphan either -- nothing proves its owner is gone.
303pub fn owned_by_dead_process(label: &str, is_alive: impl Fn(u32) -> bool) -> bool {
304 let Some(pid) = label
305 .strip_prefix("fakecloud-")
306 .and_then(|p| p.parse::<u32>().ok())
307 else {
308 return false;
309 };
310 pid != std::process::id() && !is_alive(pid)
311}
312
313/// Which container engine a CLI actually drives. Decides the podman-only
314/// code paths: `host.containers.internal` without `--add-host`, and
315/// `--tls-verify=false` when pulling from fakecloud's plain-HTTP registry.
316#[derive(Debug, Clone, Copy, PartialEq, Eq)]
317pub enum ContainerEngine {
318 Docker,
319 Podman,
320}
321
322/// Process-global memo of [`is_podman`] results, keyed by CLI name/path. The
323/// engine behind a CLI is fixed for the life of the process, and the probe
324/// runs a subprocess, so every runtime constructor and image pull after the
325/// first reads the answer from here.
326static PODMAN_CACHE: std::sync::OnceLock<
327 std::sync::Mutex<std::collections::HashMap<String, bool>>,
328> = std::sync::OnceLock::new();
329
330fn podman_cache() -> &'static std::sync::Mutex<std::collections::HashMap<String, bool>> {
331 PODMAN_CACHE.get_or_init(|| std::sync::Mutex::new(std::collections::HashMap::new()))
332}
333
334/// True when `cli` drives podman -- including podman installed *as* `docker`.
335///
336/// The file name alone can't answer that: the `podman-docker` package
337/// (Fedora, RHEL, CentOS Stream, ...) ships a `docker` shim that execs podman,
338/// and a user may point `FAKECLOUD_CONTAINER_CLI` at any wrapper. So:
339///
340/// 1. fast path: a name containing `podman` (`podman`, `podman-remote`,
341/// `/opt/homebrew/bin/podman`) is podman without running anything;
342/// 2. otherwise ask the CLI: `<cli> --version` prints `podman version X.Y.Z`
343/// through the shim and `Docker version X.Y.Z, build ...` for Docker;
344/// 3. when that answers with something neither (a custom wrapper), look for
345/// podman-only fields in `<cli> info`.
346///
347/// Every call is bounded by [`CLI_PROBE_TIMEOUT`] (a wedged CLI costs one
348/// bound, then reads as Docker, the historical default) and the answer is
349/// memoized per CLI. Blocking: from async code use [`is_podman_async`].
350pub fn is_podman(cli: &str) -> bool {
351 if name_indicates_podman(cli) {
352 return true;
353 }
354 if let Some(&cached) = podman_cache().lock().unwrap().get(cli) {
355 return cached;
356 }
357 let podman = probe_engine(cli) == Some(ContainerEngine::Podman);
358 podman_cache()
359 .lock()
360 .unwrap()
361 .insert(cli.to_string(), podman);
362 podman
363}
364
365/// [`is_podman`] for async callers: a cached (or name-matched) answer returns
366/// at once, and a first-time probe runs on the blocking pool so it can't stall
367/// a runtime worker for up to [`CLI_PROBE_TIMEOUT`].
368pub async fn is_podman_async(cli: &str) -> bool {
369 if name_indicates_podman(cli) {
370 return true;
371 }
372 if let Some(&cached) = podman_cache().lock().unwrap().get(cli) {
373 return cached;
374 }
375 let owned = cli.to_string();
376 tokio::task::spawn_blocking(move || is_podman(&owned))
377 .await
378 .unwrap_or(false)
379}
380
381/// The name-only fast path: the file name component contains `podman`, so
382/// absolute paths and `podman-remote` register. Docker's CLI and the
383/// `podman-docker` shim are both named `docker` and fall through to the probe.
384fn name_indicates_podman(cli: &str) -> bool {
385 std::path::Path::new(cli)
386 .file_name()
387 .and_then(|n| n.to_str())
388 .map(|n| n.contains("podman"))
389 .unwrap_or(false)
390}
391
392/// Ask the CLI what it is (uncached). `None` when it can't tell -- the CLI
393/// failed, timed out, or answered with neither engine's markers.
394fn probe_engine(cli: &str) -> Option<ContainerEngine> {
395 // A failed or timed-out `--version` means the CLI isn't answering at
396 // all; don't spend a second bound on `info` against it.
397 let version = bounded_output(cli, &["--version"])?;
398 classify_version_output(&version).or_else(|| {
399 bounded_output(cli, &["info", "--format", "{{json .}}"])
400 .and_then(|info| classify_info_output(&info))
401 })
402}
403
404/// Classify `<cli> --version` output. Podman prints `podman version 5.2.0`
405/// (also through the `podman-docker` shim), Docker `Docker version 27.3.1,
406/// build ce12230`. Podman is checked first: the shim is podman whatever else
407/// the output mentions.
408pub fn classify_version_output(stdout: &str) -> Option<ContainerEngine> {
409 let lower = stdout.to_ascii_lowercase();
410 if lower.contains("podman") {
411 Some(ContainerEngine::Podman)
412 } else if lower.contains("docker") {
413 Some(ContainerEngine::Docker)
414 } else {
415 None
416 }
417}
418
419/// Classify `<cli> info --format '{{json .}}'` output by engine-specific
420/// fields: podman's host section carries `buildahVersion` / `ociRuntime`
421/// (camelCase), Docker's top level `ServerVersion`.
422pub fn classify_info_output(stdout: &str) -> Option<ContainerEngine> {
423 if stdout.contains("\"buildahVersion\"") || stdout.contains("\"ociRuntime\"") {
424 Some(ContainerEngine::Podman)
425 } else if stdout.contains("\"ServerVersion\"") {
426 Some(ContainerEngine::Docker)
427 } else {
428 None
429 }
430}
431
432/// Detect the Docker bridge gateway IP on Linux. Returns `None` if
433/// detection fails (caller falls back to the conventional `172.17.0.1`).
434///
435/// Goes through [`bounded_output`] like every other container-CLI call here:
436/// `network inspect` talks to the same daemon as the liveness probe, so a
437/// wedged one blocks it on connect forever. This runs inside runtime
438/// constructors on Linux, where an unbounded call hangs server startup outright
439/// -- the exact failure [`CLI_PROBE_TIMEOUT`] exists to prevent. On timeout the
440/// caller just takes the conventional fallback.
441pub fn detect_bridge_gateway(cli: &str) -> Option<String> {
442 let stdout = bounded_output(
443 cli,
444 &[
445 "network",
446 "inspect",
447 "bridge",
448 "--format",
449 "{{range .IPAM.Config}}{{.Gateway}}{{end}}",
450 ],
451 )?;
452 let gateway = stdout.trim().to_string();
453 if gateway.is_empty() || !gateway.contains('.') {
454 return None;
455 }
456 Some(gateway)
457}
458
459/// Resolved container-to-host networking for a given CLI. Built once at
460/// runtime construction and reused for every container spawn.
461#[derive(Debug, Clone)]
462pub struct HostNetworking {
463 /// DNS name a spawned container uses to reach fakecloud on the host.
464 /// `host.containers.internal` for podman, `host.docker.internal` for
465 /// docker.
466 pub host_alias: String,
467 /// `<alias>:<value>` argument for `--add-host`, injected into every
468 /// container `create`/`run`. `None` when the runtime provides the
469 /// alias natively (podman).
470 pub add_host_arg: Option<String>,
471 /// Address fakecloud uses to reach the *sibling* containers it just
472 /// spawned (readiness probes + advertised endpoints). `127.0.0.1`
473 /// when fakecloud runs on the host; `host.docker.internal` when
474 /// fakecloud is itself containerized (`FAKECLOUD_IN_CONTAINER=1`).
475 pub sibling_host: String,
476}
477
478impl HostNetworking {
479 /// Resolve networking for `cli`, reading `FAKECLOUD_IN_CONTAINER` from
480 /// the process environment.
481 pub fn detect(cli: &str) -> Self {
482 let (host_alias, mut add_host_arg) = resolve_host_alias(cli);
483 // A resolving `host.docker.internal` is only trustworthy evidence that
484 // the runtime provides the alias natively (and will inject it into
485 // sibling containers too) when fakecloud is itself containerized:
486 // Docker-Desktop-class runtimes inject the alias into CONTAINERS, never
487 // onto the host. On a bare native-Linux host a resolving alias is
488 // spurious (a hijacking NXDOMAIN resolver, a stray /etc/hosts entry, or
489 // a wildcard search domain), so suppressing the bridge --add-host there
490 // would break the host route sibling containers need. Gate the
491 // suppression on the in-container signal to avoid that regression.
492 let in_container = in_container_mode(std::env::var("FAKECLOUD_IN_CONTAINER").ok());
493 add_host_arg = preserve_native_host_alias(
494 add_host_arg,
495 in_container && host_alias_resolves(&host_alias),
496 );
497 let sibling_host =
498 resolve_sibling_host(&host_alias, std::env::var("FAKECLOUD_IN_CONTAINER").ok());
499 Self {
500 host_alias,
501 add_host_arg,
502 sibling_host,
503 }
504 }
505
506 /// Convenience: append the `--add-host <alias>:<value>` flag pair to a
507 /// growing argv vector when this runtime needs an explicit mapping.
508 /// No-op for podman.
509 pub fn push_add_host_args(&self, argv: &mut Vec<String>) {
510 if let Some(arg) = &self.add_host_arg {
511 argv.push("--add-host".to_string());
512 argv.push(arg.clone());
513 }
514 }
515}
516
517/// How long to wait for the blocking `getaddrinfo` in [`host_alias_resolves`]
518/// before giving up and returning `false`. `getaddrinfo` has no timeout of its
519/// own, and a slow or unreachable DNS server would otherwise block a runtime
520/// thread at startup (this runs inside runtime constructors under
521/// `#[tokio::main]`). Bounding it — same tradeoff as [`CLI_PROBE_TIMEOUT`] —
522/// turns "DNS wedged" into "alias doesn't resolve", the safe default that keeps
523/// the `--add-host` bridge mapping.
524pub const HOST_ALIAS_RESOLVE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(2);
525
526/// True when `host_alias` resolves via the process resolver. The `getaddrinfo`
527/// call is blocking with no timeout of its own, so it runs on a spawned thread
528/// bounded by [`HOST_ALIAS_RESOLVE_TIMEOUT`]; on timeout we return `false` (the
529/// safe default that keeps `--add-host`). A leaked resolver thread on timeout
530/// is acceptable — same tradeoff as [`probe_cli`].
531fn host_alias_resolves(host_alias: &str) -> bool {
532 let (tx, rx) = std::sync::mpsc::channel();
533 let alias = host_alias.to_string();
534 std::thread::spawn(move || {
535 let resolves = std::net::ToSocketAddrs::to_socket_addrs(&(alias.as_str(), 0)).is_ok();
536 let _ = tx.send(resolves);
537 });
538 rx.recv_timeout(HOST_ALIAS_RESOLVE_TIMEOUT).unwrap_or(false)
539}
540
541fn preserve_native_host_alias(
542 add_host_arg: Option<String>,
543 should_suppress: bool,
544) -> Option<String> {
545 if add_host_arg.is_some() && should_suppress {
546 // Suppress the injected `--add-host host.docker.internal:<vm-bridge-ip>`
547 // only when fakecloud is containerized AND the alias already resolves
548 // (see the gate in `detect`). In that case a Docker-Desktop-class
549 // runtime provides `host.docker.internal` natively inside every sibling
550 // container, pointing at the real host; injecting the VM bridge-gateway
551 // IP would shadow it and break the host route. On a bare host — where a
552 // hijacking resolver can make the alias resolve spuriously — the caller
553 // passes `false` here so native Linux docker keeps the bridge mapping
554 // it genuinely needs.
555 None
556 } else {
557 add_host_arg
558 }
559}
560
561/// Compute the `(host_alias, add_host_arg)` pair for a CLI. Pure except
562/// for the bridge-gateway daemon probe on Linux docker, so the macOS /
563/// podman branches are unit-testable without a daemon.
564pub fn resolve_host_alias(cli: &str) -> (String, Option<String>) {
565 if is_podman(cli) {
566 // Podman provides `host.containers.internal` natively on every
567 // supported platform; injecting `host-gateway` on macOS fails
568 // because rootless podman's gvproxy doesn't expose the magic alias.
569 ("host.containers.internal".to_string(), None)
570 } else if cfg!(target_os = "linux") {
571 // Bare docker on Linux: resolve the bridge gateway IP and add an
572 // explicit alias. `host.docker.internal:host-gateway` only works
573 // on Docker Desktop; native Linux docker has no such magic.
574 let ip = detect_bridge_gateway(cli).unwrap_or_else(|| "172.17.0.1".to_string());
575 (
576 "host.docker.internal".to_string(),
577 Some(format!("host.docker.internal:{ip}")),
578 )
579 } else {
580 // Docker Desktop on Mac/Windows: `host-gateway` is the magic alias
581 // that resolves to the host's IP.
582 (
583 "host.docker.internal".to_string(),
584 Some("host.docker.internal:host-gateway".to_string()),
585 )
586 }
587}
588
589/// Decide what address fakecloud uses to reach the sibling containers it
590/// just spawned. Pure helper so the env-var parsing can be tested without
591/// touching the process's real environment.
592///
593/// - `Some("1")` / `Some("true")` (case-insensitive) -> fakecloud is in a
594/// container; the siblings publish their ports on the host's daemon and
595/// are reachable at the same host alias the spawned containers use to
596/// reach fakecloud — `host.docker.internal` under docker,
597/// `host.containers.internal` under podman. Hardcoding
598/// `host.docker.internal` here broke podman, whose gvproxy network only
599/// resolves `host.containers.internal` (issue #1539 follow-up).
600/// - anything else, including `None` -> fakecloud runs on the host,
601/// siblings live on `127.0.0.1:<port>`.
602pub fn resolve_sibling_host(host_alias: &str, env_value: Option<String>) -> String {
603 if in_container_mode(env_value) {
604 host_alias.to_string()
605 } else {
606 "127.0.0.1".to_string()
607 }
608}
609
610/// Parse the `FAKECLOUD_IN_CONTAINER` signal: `Some("1")` or a case-insensitive
611/// `Some("true")` mean fakecloud is running inside a container; anything else,
612/// including `None`, means it runs on the host. Single source of truth for the
613/// parse so `detect`'s native-alias gate and `resolve_sibling_host` can't drift.
614fn in_container_mode(env_value: Option<String>) -> bool {
615 env_value
616 .map(|v| v == "1" || v.eq_ignore_ascii_case("true"))
617 .unwrap_or(false)
618}
619
620/// Hostnames fakecloud's bundled ECR/OCI registry can be addressed from a
621/// sibling container or the host, each at `server_port`.
622///
623/// A container-spawning service rewrites the image pull URI to the runtime's
624/// sibling host -- `host.docker.internal` under Docker, `host.containers.internal`
625/// under podman -- or leaves it `localhost` / `127.0.0.1` when fakecloud runs on
626/// the host (`localhost:<port>` is the documented local ECR endpoint, e.g.
627/// `localhost:4566`). The registry enforces auth, and the Docker/Podman CLI only
628/// attaches the `Authorization` header for hosts present in `config.json`, so the
629/// isolated pull config must list *every* alias or the pull gets a 401. The map
630/// previously omitted the podman alias, so image-based Lambda/ECS pulls failed
631/// under podman-in-a-container (bug-audit 2026-06-20, 0.B2). Authorize all of
632/// them with the same credential; centralized here so the two builders can't
633/// drift again.
634///
635/// A `FAKECLOUD_ECR_REGISTRY_HOST` override is the host the pull URI is
636/// rewritten to (see [`ecr_registry_host`]), so it is authorized as well.
637pub fn registry_auth_hosts(server_port: u16) -> Vec<String> {
638 registry_auth_hosts_with(
639 server_port,
640 std::env::var("FAKECLOUD_ECR_REGISTRY_HOST").ok(),
641 )
642}
643
644/// Pure half of [`registry_auth_hosts`]: the built-in aliases plus a
645/// non-blank registry-host override.
646pub fn registry_auth_hosts_with(server_port: u16, override_host: Option<String>) -> Vec<String> {
647 let mut hosts: Vec<String> = [
648 "localhost",
649 "127.0.0.1",
650 "host.docker.internal",
651 "host.containers.internal",
652 ]
653 .iter()
654 .map(|h| h.to_string())
655 .collect();
656 if let Some(h) = override_host
657 .map(|v| v.trim().to_string())
658 .filter(|v| !v.is_empty())
659 {
660 if !hosts.contains(&h) {
661 hosts.push(h);
662 }
663 }
664 hosts
665 .iter()
666 .map(|host| format!("{host}:{server_port}"))
667 .collect()
668}
669
670/// Host the container engine pulls fakecloud-ECR images from.
671///
672/// The pull is performed by the engine on the *host*, not by a sibling
673/// container, so the sibling host alias is the wrong address for it: Docker
674/// Desktop and OrbStack only accept a plain-HTTP registry over loopback, and
675/// `host.docker.internal` does not resolve on a Linux host at all. Defaults to
676/// `127.0.0.1`. Podman on macOS / Windows pulls inside its machine VM, where
677/// loopback is the VM and only `host.containers.internal` reaches the host.
678/// `FAKECLOUD_ECR_REGISTRY_HOST` overrides both (e.g. when a containerized
679/// fakecloud's port is published under another address).
680pub fn ecr_registry_host(cli: &str) -> String {
681 resolve_ecr_registry_host(
682 std::env::var("FAKECLOUD_ECR_REGISTRY_HOST").ok(),
683 is_podman(cli),
684 cfg!(target_os = "linux"),
685 )
686}
687
688/// Pure half of [`ecr_registry_host`].
689pub fn resolve_ecr_registry_host(env_value: Option<String>, podman: bool, linux: bool) -> String {
690 if let Some(v) = env_value
691 .map(|v| v.trim().to_string())
692 .filter(|v| !v.is_empty())
693 {
694 return v;
695 }
696 if podman && !linux {
697 "host.containers.internal".to_string()
698 } else {
699 "127.0.0.1".to_string()
700 }
701}
702
703/// Host port from `<cli> port <container> <port>` output. Docker prints one
704/// `<ip>:<port>` line per bound address family (`0.0.0.0:49153`,
705/// `[::]:49153`); podman prints the same shape. The first parseable port wins.
706pub fn parse_published_port(output: &str) -> Option<u16> {
707 output
708 .lines()
709 .filter_map(|l| l.trim().rsplit(':').next())
710 .find_map(|p| p.parse::<u16>().ok())
711}
712
713const LOOPBACK_NAMES: [&str; 2] = ["127.0.0.1", "localhost"];
714
715/// Rewrite loopback references in an environment value a sibling container
716/// will read, so they reach the host instead of the container itself.
717///
718/// Inside a container `127.0.0.1` / `localhost` is the container, so every
719/// endpoint fakecloud hands out on loopback -- the server URL, an RDS or
720/// ElastiCache endpoint address, an MSK bootstrap string -- has to name
721/// `target_host` (the host alias) instead. Rewritten:
722///
723/// - the whole value being a loopback host (`DB_HOST=127.0.0.1`);
724/// - a loopback host followed by `:<port>`, at a token boundary -- in URLs
725/// with or without userinfo (`postgres://u:p@localhost:5432/db`), bare
726/// `host:port` values and comma-separated lists
727/// (`127.0.0.1:9092,127.0.0.1:9094`);
728/// - a URL host without a port (`http://localhost/path`).
729///
730/// A port a service registered with
731/// [`crate::dataplane::register_container_port`] is swapped for its
732/// container-view port. Prose such as `EHLO localhost`, and names that merely
733/// contain a loopback name (`127.0.0.10`, `localhost.localdomain`), are left
734/// alone. When `target_host` is itself loopback (fakecloud and the workload
735/// share a network namespace) the value is returned unchanged.
736pub fn rewrite_loopback_value(value: &str, target_host: &str) -> String {
737 if LOOPBACK_NAMES.contains(&target_host) {
738 return value.to_string();
739 }
740 if LOOPBACK_NAMES.contains(&value.trim()) {
741 return value.replacen(value.trim(), target_host, 1);
742 }
743 let bytes = value.as_bytes();
744 let mut out = String::with_capacity(value.len());
745 let mut i = 0;
746 while i < value.len() {
747 let hit = LOOPBACK_NAMES
748 .iter()
749 .find(|name| value[i..].starts_with(**name) && loopback_at(value, i, name.len()));
750 let Some(name) = hit else {
751 let ch = value[i..].chars().next().unwrap_or_default();
752 out.push(ch);
753 i += ch.len_utf8().max(1);
754 continue;
755 };
756 out.push_str(target_host);
757 i += name.len();
758 // Swap a registered host port for its container-view port.
759 if bytes.get(i) == Some(&b':') {
760 let digits: String = value[i + 1..]
761 .chars()
762 .take_while(|c| c.is_ascii_digit())
763 .collect();
764 if let Some(mapped) = digits
765 .parse::<u16>()
766 .ok()
767 .and_then(crate::dataplane::container_port_for)
768 {
769 out.push(':');
770 out.push_str(&mapped.to_string());
771 i += 1 + digits.len();
772 }
773 }
774 }
775 out
776}
777
778/// Whether the loopback name at `value[i..i + len]` is a host reference: at a
779/// token boundary and followed by `:<digit>`, or a URL host (after `//` or
780/// `@`) followed by the end of the authority.
781fn loopback_at(value: &str, i: usize, len: usize) -> bool {
782 let before = value[..i].chars().next_back();
783 let boundary_before = match before {
784 None => true,
785 Some(c) => matches!(c, '/' | '@' | ',' | '=' | ';' | '(' | '"' | '\'') || c.is_whitespace(),
786 };
787 if !boundary_before {
788 return false;
789 }
790 let rest = &value[i + len..];
791 let mut after = rest.chars();
792 match after.next() {
793 Some(':') => after.next().is_some_and(|c| c.is_ascii_digit()),
794 next => {
795 let url_host = matches!(before, Some('@')) || value[..i].ends_with("//");
796 url_host && matches!(next, None | Some('/') | Some('?') | Some('#'))
797 }
798 }
799}
800
801#[cfg(test)]
802mod tests {
803 use super::*;
804
805 #[test]
806 fn cli_available_false_for_missing_binary() {
807 // A binary that doesn't exist fails to spawn -> unavailable, fast.
808 assert!(!cli_available("definitely-not-a-real-cli-binary-xyz-123"));
809 }
810
811 #[cfg(unix)]
812 #[test]
813 fn cli_available_bounds_a_hanging_probe() {
814 // A CLI whose `info` invocation blocks forever (like `docker info`
815 // against an unreachable daemon) must not hang the caller: the probe
816 // is killed at CLI_PROBE_TIMEOUT and reported unavailable. Regression
817 // test for the local-conformance-probe hang.
818 use std::io::Write;
819 use std::os::unix::fs::PermissionsExt;
820
821 let dir = std::env::temp_dir().join(format!("fc-clitest-{}", std::process::id()));
822 std::fs::create_dir_all(&dir).unwrap();
823 let script = dir.join("hangcli");
824 std::fs::write(&script, "#!/bin/sh\nsleep 600\n").unwrap();
825 std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
826 std::io::stdout().flush().ok();
827
828 let start = std::time::Instant::now();
829 let available = cli_available(script.to_str().unwrap());
830 let elapsed = start.elapsed();
831
832 std::fs::remove_dir_all(&dir).ok();
833 assert!(!available, "a hanging probe must report unavailable");
834 assert!(
835 elapsed < CLI_PROBE_TIMEOUT + std::time::Duration::from_secs(5),
836 "probe took {elapsed:?}, expected it bounded near {CLI_PROBE_TIMEOUT:?}"
837 );
838 }
839
840 #[test]
841 fn registry_auth_hosts_includes_podman_alias() {
842 // The podman sibling alias (host.containers.internal) must be authorized
843 // or image-based Lambda/ECS pulls 401 under podman-in-a-container (0.B2).
844 let hosts = registry_auth_hosts(4566);
845 assert!(hosts.contains(&"localhost:4566".to_string()));
846 assert!(hosts.contains(&"127.0.0.1:4566".to_string()));
847 assert!(hosts.contains(&"host.docker.internal:4566".to_string()));
848 assert!(
849 hosts.contains(&"host.containers.internal:4566".to_string()),
850 "podman sibling alias must be authorized: {hosts:?}"
851 );
852 }
853
854 #[test]
855 fn name_fast_path_matches_podman_names() {
856 assert!(name_indicates_podman("podman"));
857 assert!(name_indicates_podman("podman-remote"));
858 assert!(name_indicates_podman("/opt/homebrew/bin/podman"));
859 assert!(name_indicates_podman("/usr/local/bin/podman-remote"));
860 assert!(!name_indicates_podman("docker"));
861 assert!(!name_indicates_podman("/usr/local/bin/docker"));
862 assert!(!name_indicates_podman("docker-credential-helper"));
863 }
864
865 #[test]
866 fn name_fast_path_needs_no_probe() {
867 // No binary by this name exists: a podman-named CLI is podman without
868 // running anything.
869 assert!(is_podman("/nonexistent-dir-fc-2599/podman"));
870 assert!(is_podman("podman-remote-definitely-missing-xyz"));
871 }
872
873 #[test]
874 fn version_output_classifies_the_engine() {
875 assert_eq!(
876 classify_version_output("podman version 5.2.0\n"),
877 Some(ContainerEngine::Podman)
878 );
879 assert_eq!(
880 classify_version_output("Docker version 27.3.1, build ce12230\n"),
881 Some(ContainerEngine::Docker)
882 );
883 // The podman-docker shim's notice mentions Docker; it is still podman.
884 assert_eq!(
885 classify_version_output(
886 "Emulate Docker CLI using podman. Create /etc/containers/nodocker to quiet msg.\npodman version 4.9.4\n"
887 ),
888 Some(ContainerEngine::Podman)
889 );
890 assert_eq!(classify_version_output("my-wrapper 1.0\n"), None);
891 assert_eq!(classify_version_output(""), None);
892 }
893
894 #[test]
895 fn info_output_classifies_the_engine() {
896 assert_eq!(
897 classify_info_output(
898 r#"{"host":{"buildahVersion":"1.37.0","ociRuntime":{"name":"crun"}}}"#
899 ),
900 Some(ContainerEngine::Podman)
901 );
902 assert_eq!(
903 classify_info_output(r#"{"ID":"abc","ServerVersion":"27.3.1","Driver":"overlay2"}"#),
904 Some(ContainerEngine::Docker)
905 );
906 assert_eq!(classify_info_output("{}"), None);
907 }
908
909 /// Write an executable fake CLI named `name` in a fresh temp dir, and run
910 /// it once so a concurrent fork can't leave it ETXTBSY for the probe.
911 #[cfg(unix)]
912 fn fake_cli(name: &str, body: &str) -> (tempfile::TempDir, String) {
913 use std::os::unix::fs::PermissionsExt;
914 let dir = tempfile::tempdir().unwrap();
915 let path = dir.path().join(name);
916 std::fs::write(&path, format!("#!/bin/sh\n{body}")).unwrap();
917 std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o755)).unwrap();
918 let mut attempts = 0;
919 loop {
920 match std::process::Command::new(&path).arg("warmup").output() {
921 Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy && attempts < 200 => {
922 attempts += 1;
923 std::thread::sleep(std::time::Duration::from_millis(5));
924 }
925 _ => break,
926 }
927 }
928 let cli = path.display().to_string();
929 (dir, cli)
930 }
931
932 #[cfg(unix)]
933 #[test]
934 fn podman_docker_shim_is_detected_as_podman() {
935 // What `podman-docker` installs: a `docker` that is podman.
936 let (_dir, cli) = fake_cli(
937 "docker",
938 "[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
939 );
940 assert!(is_podman(&cli));
941 assert_eq!(probe_engine(&cli), Some(ContainerEngine::Podman));
942 }
943
944 #[cfg(unix)]
945 #[test]
946 fn real_docker_cli_is_not_podman() {
947 let (_dir, cli) = fake_cli(
948 "docker",
949 "[ \"$1\" = --version ] && { echo 'Docker version 27.3.1, build ce12230'; exit 0; }\nexit 1\n",
950 );
951 assert!(!is_podman(&cli));
952 }
953
954 #[cfg(unix)]
955 #[test]
956 fn wrapper_with_opaque_version_falls_back_to_info() {
957 let (_dir, cli) = fake_cli(
958 "container-cli",
959 "case \"$1\" in\n --version) echo 'wrapper 1.0' ;;\n info) echo '{\"host\":{\"buildahVersion\":\"1.37.0\"}}' ;;\n *) exit 1 ;;\nesac\n",
960 );
961 assert!(is_podman(&cli));
962 }
963
964 #[cfg(unix)]
965 #[test]
966 fn engine_probe_is_cached_per_cli() {
967 // Each run appends to a log; a second lookup must not run the CLI again.
968 let (dir, cli) = fake_cli(
969 "docker",
970 "echo \"$*\" >> \"$(dirname \"$0\")/calls.log\"\n[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
971 );
972 let log = dir.path().join("calls.log");
973 let _ = std::fs::remove_file(&log);
974 assert!(is_podman(&cli));
975 assert!(is_podman(&cli));
976 let calls = std::fs::read_to_string(&log).unwrap_or_default();
977 assert_eq!(calls.lines().collect::<Vec<_>>(), ["--version"]);
978 }
979
980 #[cfg(unix)]
981 #[tokio::test]
982 async fn async_probe_detects_the_shim() {
983 let (_dir, cli) = fake_cli(
984 "docker",
985 "[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
986 );
987 assert!(is_podman_async(&cli).await);
988 }
989
990 #[cfg(unix)]
991 #[test]
992 fn engine_probe_bounds_a_hanging_cli() {
993 // A CLI that never answers must not hang detection: one bounded call,
994 // no `info` fallback against it, and it reads as Docker.
995 // Only the probe's commands hang: `fake_cli`'s warmup run (no such
996 // argument) must return at once, or it eats the test's time budget.
997 let (_dir, cli) = fake_cli(
998 "docker",
999 "case \"$1\" in --version|info) sleep 600 ;; esac\nexit 1\n",
1000 );
1001 let start = std::time::Instant::now();
1002 assert!(!is_podman(&cli));
1003 let elapsed = start.elapsed();
1004 assert!(
1005 elapsed < CLI_PROBE_TIMEOUT + std::time::Duration::from_secs(5),
1006 "probe took {elapsed:?}, expected one bound near {CLI_PROBE_TIMEOUT:?}"
1007 );
1008 }
1009
1010 #[test]
1011 fn missing_cli_is_not_podman() {
1012 // `FAKECLOUD_CONTAINER_CLI=false`-style sentinels and missing binaries
1013 // read as not-podman without error.
1014 assert!(!is_podman("definitely-not-a-real-cli-binary-fc-2599"));
1015 assert!(!is_podman("false"));
1016 }
1017
1018 #[cfg(unix)]
1019 #[test]
1020 fn resolve_host_alias_treats_the_shim_as_podman() {
1021 let (_dir, cli) = fake_cli(
1022 "docker",
1023 "[ \"$1\" = --version ] && { echo 'podman version 5.2.0'; exit 0; }\nexit 1\n",
1024 );
1025 let (alias, add_host) = resolve_host_alias(&cli);
1026 assert_eq!(alias, "host.containers.internal");
1027 assert_eq!(add_host, None);
1028 }
1029
1030 #[test]
1031 fn resolve_host_alias_podman_has_no_add_host() {
1032 let (alias, add_host) = resolve_host_alias("podman");
1033 assert_eq!(alias, "host.containers.internal");
1034 assert_eq!(add_host, None);
1035 let (alias, add_host) = resolve_host_alias("/opt/homebrew/bin/podman");
1036 assert_eq!(alias, "host.containers.internal");
1037 assert_eq!(add_host, None);
1038 }
1039
1040 #[test]
1041 #[cfg(unix)]
1042 fn resolve_host_alias_docker_emits_add_host() {
1043 // A fake Docker CLI, so a host whose `docker` is the podman-docker shim
1044 // can't flip this test.
1045 let (_dir, cli) = fake_cli(
1046 "docker",
1047 "[ \"$1\" = --version ] && { echo 'Docker version 27.3.1, build ce12230'; exit 0; }\nexit 1\n",
1048 );
1049 let (alias, add_host) = resolve_host_alias(&cli);
1050 assert_eq!(alias, "host.docker.internal");
1051 // On macOS this is host-gateway; on Linux it's a bridge IP. Either
1052 // way docker must get an explicit --add-host.
1053 assert!(add_host.is_some());
1054 assert!(add_host.unwrap().starts_with("host.docker.internal:"));
1055 }
1056
1057 #[test]
1058 fn native_host_alias_prevents_docker_add_host_override() {
1059 let add_host =
1060 preserve_native_host_alias(Some("host.docker.internal:host-gateway".to_string()), true);
1061
1062 assert_eq!(add_host, None);
1063 }
1064
1065 #[test]
1066 fn unresolved_host_alias_keeps_docker_add_host() {
1067 let add_host = preserve_native_host_alias(
1068 Some("host.docker.internal:host-gateway".to_string()),
1069 false,
1070 );
1071
1072 assert_eq!(
1073 add_host.as_deref(),
1074 Some("host.docker.internal:host-gateway")
1075 );
1076 }
1077
1078 #[test]
1079 fn absent_docker_add_host_remains_absent() {
1080 assert_eq!(preserve_native_host_alias(None, true), None);
1081 assert_eq!(preserve_native_host_alias(None, false), None);
1082 }
1083
1084 #[test]
1085 fn in_container_mode_parses_truthy_values() {
1086 assert!(in_container_mode(Some("1".to_string())));
1087 assert!(in_container_mode(Some("true".to_string())));
1088 assert!(in_container_mode(Some("True".to_string())));
1089 assert!(in_container_mode(Some("TRUE".to_string())));
1090 }
1091
1092 #[test]
1093 fn in_container_mode_rejects_falsey_and_absent() {
1094 assert!(!in_container_mode(None));
1095 assert!(!in_container_mode(Some(String::new())));
1096 assert!(!in_container_mode(Some("0".to_string())));
1097 assert!(!in_container_mode(Some("false".to_string())));
1098 assert!(!in_container_mode(Some("yes".to_string())));
1099 }
1100
1101 #[test]
1102 fn native_alias_gate_suppresses_only_in_container() {
1103 // The gate `detect` computes: `in_container && host_alias_resolves`.
1104 let add_host = || Some("host.docker.internal:172.17.0.1".to_string());
1105
1106 // In-container + resolves -> Desktop-class runtime provides the alias
1107 // natively in siblings; drop the shadowing bridge mapping.
1108 let in_container = true;
1109 let resolves = true;
1110 assert_eq!(
1111 preserve_native_host_alias(add_host(), in_container && resolves),
1112 None,
1113 );
1114
1115 // NOT in-container (bare host) + resolves -> the resolving alias is
1116 // spurious (hijacking resolver / stray hosts entry). Native Linux docker
1117 // needs the bridge mapping; must NOT drop it. Regression guard.
1118 let in_container = false;
1119 let resolves = true;
1120 assert_eq!(
1121 preserve_native_host_alias(add_host(), in_container && resolves).as_deref(),
1122 Some("host.docker.internal:172.17.0.1"),
1123 );
1124
1125 // In-container + does NOT resolve -> nothing native to preserve; keep
1126 // the injected mapping.
1127 let in_container = true;
1128 let resolves = false;
1129 assert_eq!(
1130 preserve_native_host_alias(add_host(), in_container && resolves).as_deref(),
1131 Some("host.docker.internal:172.17.0.1"),
1132 );
1133 }
1134
1135 #[test]
1136 fn resolve_sibling_host_defaults_to_loopback() {
1137 assert_eq!(
1138 resolve_sibling_host("host.docker.internal", None),
1139 "127.0.0.1"
1140 );
1141 assert_eq!(
1142 resolve_sibling_host("host.docker.internal", Some(String::new())),
1143 "127.0.0.1"
1144 );
1145 assert_eq!(
1146 resolve_sibling_host("host.docker.internal", Some("0".to_string())),
1147 "127.0.0.1"
1148 );
1149 assert_eq!(
1150 resolve_sibling_host("host.containers.internal", Some("false".to_string())),
1151 "127.0.0.1"
1152 );
1153 }
1154
1155 #[test]
1156 fn resolve_sibling_host_uses_host_alias_when_in_container() {
1157 // Docker: siblings reachable at host.docker.internal.
1158 assert_eq!(
1159 resolve_sibling_host("host.docker.internal", Some("1".to_string())),
1160 "host.docker.internal"
1161 );
1162 assert_eq!(
1163 resolve_sibling_host("host.docker.internal", Some("true".to_string())),
1164 "host.docker.internal"
1165 );
1166 assert_eq!(
1167 resolve_sibling_host("host.docker.internal", Some("TRUE".to_string())),
1168 "host.docker.internal"
1169 );
1170 // Podman: must use host.containers.internal, NOT host.docker.internal
1171 // (issue #1539 follow-up — gvproxy only resolves the containers alias).
1172 assert_eq!(
1173 resolve_sibling_host("host.containers.internal", Some("1".to_string())),
1174 "host.containers.internal"
1175 );
1176 }
1177
1178 #[test]
1179 fn detect_wires_sibling_host_to_podman_alias_in_container() {
1180 // Full path: a podman binary in a container must advertise siblings
1181 // at host.containers.internal. resolve_host_alias drives host_alias,
1182 // which resolve_sibling_host then reuses.
1183 let (alias, add_host) = resolve_host_alias("podman");
1184 assert_eq!(alias, "host.containers.internal");
1185 assert_eq!(add_host, None);
1186 assert_eq!(
1187 resolve_sibling_host(&alias, Some("1".to_string())),
1188 "host.containers.internal"
1189 );
1190 }
1191
1192 #[test]
1193 fn only_objects_of_a_dead_owner_are_orphans() {
1194 let me = std::process::id();
1195 let alive = |pid: u32| pid == 4242;
1196 // Another live fakecloud process: never an orphan.
1197 assert!(!owned_by_dead_process("fakecloud-4242", alive));
1198 // Its owner is gone: an orphan.
1199 assert!(owned_by_dead_process("fakecloud-777", alive));
1200 // The current process, even if the probe says otherwise.
1201 assert!(!owned_by_dead_process(&format!("fakecloud-{me}"), |_| {
1202 false
1203 }));
1204 // Nothing proves an unparseable owner is gone.
1205 for label in ["", "fakecloud-", "fakecloud-abc", "other-777"] {
1206 assert!(!owned_by_dead_process(label, alive), "{label:?}");
1207 }
1208 }
1209
1210 #[cfg(unix)]
1211 #[test]
1212 fn pid_alive_probes_real_processes() {
1213 assert!(pid_alive(std::process::id()));
1214 assert!(!pid_alive(u32::MAX - 1));
1215 }
1216
1217 #[test]
1218 fn push_add_host_args_noop_for_podman() {
1219 let net = HostNetworking {
1220 host_alias: "host.containers.internal".to_string(),
1221 add_host_arg: None,
1222 sibling_host: "127.0.0.1".to_string(),
1223 };
1224 let mut argv = vec!["create".to_string()];
1225 net.push_add_host_args(&mut argv);
1226 assert_eq!(argv, vec!["create".to_string()]);
1227 }
1228
1229 #[test]
1230 fn push_add_host_args_emits_for_docker() {
1231 let net = HostNetworking {
1232 host_alias: "host.docker.internal".to_string(),
1233 add_host_arg: Some("host.docker.internal:host-gateway".to_string()),
1234 sibling_host: "127.0.0.1".to_string(),
1235 };
1236 let mut argv = vec!["create".to_string()];
1237 net.push_add_host_args(&mut argv);
1238 assert_eq!(
1239 argv,
1240 vec![
1241 "create".to_string(),
1242 "--add-host".to_string(),
1243 "host.docker.internal:host-gateway".to_string(),
1244 ]
1245 );
1246 }
1247}
1248
1249#[cfg(test)]
1250mod bounded_cli_tests {
1251 use super::*;
1252
1253 /// A wedged daemon leaves the CLI blocked on connect forever. Every
1254 /// container call has to end at the bound instead of hanging its caller,
1255 /// which for the reaper means hanging server startup.
1256 #[test]
1257 fn a_hanging_cli_call_is_cut_off() {
1258 let start = std::time::Instant::now();
1259 let mut child = spawn_bounded(
1260 std::process::Command::new("sleep")
1261 .arg("600")
1262 .stdout(std::process::Stdio::null())
1263 .stderr(std::process::Stdio::null()),
1264 )
1265 .expect("sleep is available");
1266 assert!(!wait_bounded_group(&mut child));
1267 assert!(
1268 start.elapsed() < CLI_PROBE_TIMEOUT + std::time::Duration::from_secs(5),
1269 "the wait must end at the bound"
1270 );
1271 }
1272
1273 /// Output larger than a pipe buffer (64 KiB on Linux) must come back
1274 /// whole. Waiting for the child to exit before reading blocks it on write
1275 /// forever, so this used to burn the full timeout and report failure.
1276 #[test]
1277 fn output_larger_than_the_pipe_buffer_still_comes_back() {
1278 let start = std::time::Instant::now();
1279 // 200_000 bytes: comfortably past the buffer on every supported host.
1280 let out = bounded_output("sh", &["-c", "printf 'x%.0s' $(seq 1 200000)"])
1281 .expect("a large but prompt call must succeed");
1282 assert_eq!(out.len(), 200_000, "output was truncated");
1283 assert!(
1284 start.elapsed() < CLI_PROBE_TIMEOUT,
1285 "a prompt call must not reach the deadline"
1286 );
1287 }
1288
1289 #[test]
1290 fn a_prompt_cli_call_returns_its_output() {
1291 assert_eq!(
1292 bounded_output("echo", &["abc123"])
1293 .as_deref()
1294 .map(str::trim),
1295 Some("abc123")
1296 );
1297 assert!(bounded_status("true", &[]));
1298 assert!(!bounded_status("false", &[]));
1299 }
1300
1301 /// A bounded call gets an empty stdin, never fakecloud's own. The process
1302 /// group `spawn_bounded` creates is a *background* one, so a child that
1303 /// reads the controlling terminal -- a `sudo`/credential-helper wrapper
1304 /// prompting -- is stopped by SIGTTIN, which the WNOHANG `try_wait` loop
1305 /// cannot see: the call would burn the whole deadline and be killed.
1306 /// `cat` with no argument reads stdin to EOF, so it returns at once with
1307 /// nothing only when stdin is /dev/null.
1308 #[cfg(unix)]
1309 #[test]
1310 fn a_bounded_call_reads_an_empty_stdin() {
1311 let start = std::time::Instant::now();
1312 let out = bounded_output("cat", &[]).expect("a call reading stdin must not time out");
1313 assert!(out.is_empty(), "stdin must be empty, got {out:?}");
1314 assert!(
1315 start.elapsed() < CLI_PROBE_TIMEOUT,
1316 "a call reading stdin must not reach the deadline"
1317 );
1318 }
1319
1320 /// The happy path must still hand back the child's output *and* collect the
1321 /// reader, so the no-leak guarantee isn't bought by dropping output.
1322 #[test]
1323 fn a_prompt_cli_call_collects_its_reader() {
1324 let (output, reader) = run_bounded("echo", &["abc123"]);
1325 assert_eq!(output.as_deref().map(str::trim), Some("abc123"));
1326 assert!(
1327 matches!(reader, ReaderState::Finished),
1328 "reader was {reader:?}, expected it collected"
1329 );
1330 }
1331
1332 /// The bridge-gateway probe talks to the same daemon as the liveness probe,
1333 /// so it has to end at the same bound. It used to be a plain
1334 /// `Command::output()`, which against a wedged daemon hung whichever runtime
1335 /// constructor called it -- on Linux, every container-backed service at
1336 /// server startup. Timing out is not an error here: the caller falls back to
1337 /// the conventional `172.17.0.1`.
1338 #[cfg(unix)]
1339 #[test]
1340 fn a_hanging_bridge_gateway_probe_is_cut_off() {
1341 use std::os::unix::fs::PermissionsExt;
1342
1343 let dir = std::env::temp_dir().join(format!("fc-gwtest-{}", std::process::id()));
1344 std::fs::create_dir_all(&dir).unwrap();
1345 let script = dir.join("hangcli");
1346 std::fs::write(&script, "#!/bin/sh\nsleep 600\n").unwrap();
1347 std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
1348
1349 let start = std::time::Instant::now();
1350 let gateway = detect_bridge_gateway(script.to_str().unwrap());
1351 let elapsed = start.elapsed();
1352
1353 std::fs::remove_dir_all(&dir).ok();
1354 assert_eq!(gateway, None, "a wedged daemon must report no gateway");
1355 assert!(
1356 elapsed < CLI_PROBE_TIMEOUT + READER_DRAIN_GRACE + std::time::Duration::from_secs(5),
1357 "the probe took {elapsed:?}, expected it bounded near {CLI_PROBE_TIMEOUT:?}"
1358 );
1359 }
1360
1361 /// The unavailable-CLI path keeps its shape: nothing to spawn, no gateway,
1362 /// and the caller's fallback stands.
1363 #[test]
1364 fn a_missing_cli_reports_no_bridge_gateway() {
1365 assert_eq!(
1366 detect_bridge_gateway("definitely-not-a-real-cli-binary-xyz-123"),
1367 None
1368 );
1369 }
1370
1371 /// A CLI that succeeds with no output -- an `inspect --format` over a bridge
1372 /// with no IPAM config -- still means "no gateway", not an empty
1373 /// `--add-host` value. Unchanged by the bounding; guarded so it stays that
1374 /// way.
1375 #[test]
1376 fn an_empty_gateway_is_rejected() {
1377 assert_eq!(detect_bridge_gateway("true"), None);
1378 }
1379
1380 /// `FAKECLOUD_CONTAINER_CLI` is routinely a wrapper (`sh -c 'exec docker
1381 /// "$@"'`, a `podman-remote` shim), which makes the real command a
1382 /// grandchild holding the stdout pipe. Killing only the direct child left
1383 /// the reader's `read_to_end` blocked forever -- a thread parked for the
1384 /// life of the process, once per call, on exactly the wedged-daemon path
1385 /// these bounds were added for (the server reaper calls this at startup).
1386 /// The reader reporting in is the evidence: EOF on that pipe is only
1387 /// possible once every write end is closed, so a collected buffer proves
1388 /// the grandchildren went down with the call.
1389 #[cfg(unix)]
1390 #[test]
1391 fn a_timed_out_wrapper_call_leaves_no_reader_behind() {
1392 let start = std::time::Instant::now();
1393 // A wrapper that outlives its own kill: the backgrounded sleep inherits
1394 // the stdout pipe and is not the process we spawned.
1395 let (output, reader) = run_bounded("sh", &["-c", "sleep 600 & sleep 600"]);
1396 assert_eq!(output, None, "a wedged call must report failure");
1397 assert!(
1398 matches!(reader, ReaderState::Finished),
1399 "reader was {reader:?}: the stdout reader must not outlive the call"
1400 );
1401 assert!(
1402 start.elapsed()
1403 < CLI_PROBE_TIMEOUT + READER_DRAIN_GRACE + std::time::Duration::from_secs(5),
1404 "the call must still end at the bound, took {:?}",
1405 start.elapsed()
1406 );
1407 }
1408}
1409
1410#[cfg(test)]
1411mod endpoint_tests {
1412 use super::*;
1413
1414 #[test]
1415 fn ecr_registry_host_defaults_to_loopback() {
1416 // The host daemon performs the pull; the sibling alias is wrong there.
1417 assert_eq!(resolve_ecr_registry_host(None, false, true), "127.0.0.1");
1418 assert_eq!(resolve_ecr_registry_host(None, false, false), "127.0.0.1");
1419 assert_eq!(
1420 resolve_ecr_registry_host(Some(" ".into()), false, true),
1421 "127.0.0.1"
1422 );
1423 // Podman on Linux pulls on the host; podman machine pulls in its VM.
1424 assert_eq!(resolve_ecr_registry_host(None, true, true), "127.0.0.1");
1425 assert_eq!(
1426 resolve_ecr_registry_host(None, true, false),
1427 "host.containers.internal"
1428 );
1429 assert_eq!(
1430 resolve_ecr_registry_host(Some("10.0.0.5".into()), true, false),
1431 "10.0.0.5"
1432 );
1433 }
1434
1435 #[test]
1436 fn registry_override_host_is_authorized() {
1437 // A pull rewritten to FAKECLOUD_ECR_REGISTRY_HOST must carry auth, or
1438 // the registry answers 401.
1439 let hosts = registry_auth_hosts_with(4566, Some("10.1.2.3".into()));
1440 assert!(hosts.contains(&"10.1.2.3:4566".to_string()), "{hosts:?}");
1441 assert!(hosts.contains(&"127.0.0.1:4566".to_string()));
1442 // No duplicates, blank override ignored.
1443 assert_eq!(
1444 registry_auth_hosts_with(4566, Some("127.0.0.1".into())).len(),
1445 4
1446 );
1447 assert_eq!(registry_auth_hosts_with(4566, Some(" ".into())).len(), 4);
1448 }
1449
1450 #[test]
1451 fn published_port_parses_docker_and_podman_output() {
1452 assert_eq!(
1453 parse_published_port("0.0.0.0:49153\n[::]:49153\n"),
1454 Some(49153)
1455 );
1456 assert_eq!(parse_published_port("[::]:5000"), Some(5000));
1457 assert_eq!(parse_published_port(""), None);
1458 assert_eq!(parse_published_port("garbage"), None);
1459 }
1460
1461 #[test]
1462 fn rewrite_loopback_urls_and_bare_endpoints() {
1463 let h = "host.docker.internal";
1464 assert_eq!(
1465 rewrite_loopback_value("http://localhost:4566", h),
1466 "http://host.docker.internal:4566"
1467 );
1468 assert_eq!(
1469 rewrite_loopback_value("https://127.0.0.1:4566/path", h),
1470 "https://host.docker.internal:4566/path"
1471 );
1472 // Bare host (an RDS / ElastiCache endpoint address).
1473 assert_eq!(rewrite_loopback_value("127.0.0.1", h), h);
1474 assert_eq!(rewrite_loopback_value("localhost", h), h);
1475 // Bare host:port and comma-separated broker lists.
1476 assert_eq!(
1477 rewrite_loopback_value("127.0.0.1:6379", h),
1478 "host.docker.internal:6379"
1479 );
1480 assert_eq!(
1481 rewrite_loopback_value("127.0.0.1:9092,localhost:9094", h),
1482 "host.docker.internal:9092,host.docker.internal:9094"
1483 );
1484 // Scheme-qualified DSNs with userinfo.
1485 assert_eq!(
1486 rewrite_loopback_value("postgres://u:p@localhost:5432/db", h),
1487 "postgres://u:p@host.docker.internal:5432/db"
1488 );
1489 assert_eq!(
1490 rewrite_loopback_value("redis://127.0.0.1:6379/0", h),
1491 "redis://host.docker.internal:6379/0"
1492 );
1493 // URL host without a port.
1494 assert_eq!(
1495 rewrite_loopback_value("http://localhost/health", h),
1496 "http://host.docker.internal/health"
1497 );
1498 // key=value lists.
1499 assert_eq!(
1500 rewrite_loopback_value("host=127.0.0.1:5432 user=x", h),
1501 "host=host.docker.internal:5432 user=x"
1502 );
1503 }
1504
1505 #[test]
1506 fn rewrite_loopback_leaves_prose_and_lookalikes_alone() {
1507 let h = "host.docker.internal";
1508 for v in [
1509 "EHLO localhost",
1510 "connect to localhost soon",
1511 "127.0.0.10:80",
1512 "localhost.localdomain:25",
1513 "mylocalhost:80",
1514 "",
1515 ] {
1516 assert_eq!(rewrite_loopback_value(v, h), v, "{v:?}");
1517 }
1518 // Workload sharing fakecloud's namespace: nothing to rewrite.
1519 assert_eq!(
1520 rewrite_loopback_value("http://localhost:4566", "127.0.0.1"),
1521 "http://localhost:4566"
1522 );
1523 }
1524
1525 #[test]
1526 fn rewrite_loopback_swaps_registered_container_ports() {
1527 // A broker whose host listener advertises loopback registers the
1528 // listener containers must use instead.
1529 crate::dataplane::register_container_port(41001, 41002);
1530 assert_eq!(
1531 rewrite_loopback_value("127.0.0.1:41001", "host.docker.internal"),
1532 "host.docker.internal:41002"
1533 );
1534 crate::dataplane::unregister_container_port(41001);
1535 assert_eq!(
1536 rewrite_loopback_value("127.0.0.1:41001", "host.docker.internal"),
1537 "host.docker.internal:41001"
1538 );
1539 }
1540}