Skip to main content

trusty_memory/commands/
single_instance.rs

1//! Single-instance guard for the trusty-memory daemon.
2//!
3//! Why: macOS launchd `KeepAlive { SuccessfulExit: false }` respawns the daemon
4//! whenever it exits with a non-zero code. A second instance that fails to bind
5//! exits non-zero, launchd reads that as a crash and spawns another copy, and
6//! the resulting zombie herd (69 observed in the wild) exhausts file
7//! descriptors on top of the existing fd-limit bug.
8//!
9//! The fix: before binding, probe the socket. If something is already serving
10//! it, exit **0**. Launchd treats exit-0 as a clean shutdown and does not
11//! respawn, which collapses the herd on the next invocation without touching
12//! the launchd config.
13//!
14//! #6286 changed what is probed, not the decision. It used to read the
15//! `http_addr` discovery file and GET `/health` at whatever address that named
16//! — a file that goes stale after a SIGKILL, which is why the probe had to
17//! tolerate one pointing at a dead port. The socket path is derived rather than
18//! published, so there is nothing to be stale and the probe is a bare connect
19//! through `trusty_common::uds::socket_is_serving`.
20//!
21//! What: [`single_instance_check`] (async, for real daemon startups) and
22//! [`StartupAction`] (a pure enum, so the decision is unit-testable without
23//! I/O).
24//!
25//! Test: `startup_action_*` for the decision, `single_instance_check_*` for the
26//! probe.
27
28use std::path::Path;
29use std::time::Duration;
30
31/// How long a liveness connect may take before the path is called dead.
32///
33/// A local socket accepts or refuses in microseconds; this is headroom for a
34/// loaded machine, not a latency budget. It matches trusty-analyze's
35/// `daemon_guard::PROBE_TIMEOUT` so the two daemons wait the same.
36const PROBE_TIMEOUT: Duration = Duration::from_millis(500);
37
38/// What the daemon startup should do after the single-instance check.
39///
40/// Why: separating the decision from the I/O lets us unit-test the logic
41/// with injected probe results rather than spinning up real TCP listeners.
42/// What: three variants covering the full decision tree.
43/// Test: `startup_action_from_probe_result_*` tests in this module.
44#[derive(Debug, Clone, PartialEq, Eq)]
45pub enum StartupAction {
46    /// Proceed to bind the socket and start serving.
47    Proceed,
48    /// Another healthy instance is already running — exit 0 cleanly so
49    /// launchd does not respawn.
50    ExitAlreadyRunning,
51    /// A probe attempt failed with an unexpected error — propagate as a startup
52    /// failure so the operator sees a real error in the launchd log. Launchd
53    /// respawns, correctly, because this is a genuine failure.
54    ///
55    /// Nothing constructs this today: `socket_is_serving` answers a bool, so
56    /// every failure it can have is "nothing is serving". The variant stays
57    /// because the caller in `main.rs` branches on it and a future probe that
58    /// can distinguish a permission error from an absence has somewhere to
59    /// report it.
60    Fail(String),
61}
62
63/// Decide what to do based on the result of a liveness probe.
64///
65/// Why: the single-instance check reduces to "did the health probe succeed?".
66/// Encoding the decision as a pure function (rather than embedding it in the
67/// async probe body) makes the logic unit-testable without actual network I/O.
68/// What: `probe_ok = true` → [`StartupAction::ExitAlreadyRunning`];
69/// `probe_ok = false` → [`StartupAction::Proceed`].
70/// Test: `startup_action_from_probe_result_when_alive`,
71///       `startup_action_from_probe_result_when_dead`.
72pub fn startup_action_from_probe_result(probe_ok: bool) -> StartupAction {
73    if probe_ok {
74        StartupAction::ExitAlreadyRunning
75    } else {
76        StartupAction::Proceed
77    }
78}
79
80/// Perform the single-instance check at daemon startup.
81///
82/// Why: launchd respawns any non-zero exit, so a second instance that fails to
83/// bind causes an endless respawn storm. Exiting 0 when another healthy
84/// instance is detected short-circuits it.
85///
86/// What: a bare connect to `socket`. It deliberately does NOT call
87/// `memory.health`: the question is whether the endpoint is live, and a daemon
88/// that is up but degraded must not be reported absent and spawned on top of
89/// itself. An absent or dead socket returns [`StartupAction::Proceed`], so a
90/// cold start is never blocked.
91///
92/// Test: `single_instance_check_proceeds_when_nothing_is_serving`,
93/// `single_instance_check_exits_when_something_is_serving`.
94pub async fn single_instance_check(socket: &Path) -> StartupAction {
95    let probe_ok = trusty_common::uds::socket_is_serving(socket, PROBE_TIMEOUT).await;
96    startup_action_from_probe_result(probe_ok)
97}
98
99/// Single-instance check with up to `max_retries` additional probes.
100///
101/// Why (issue #1152, Tier 3): a single probe can miss a daemon that is
102/// mid-boot — it has not bound the socket yet. Retrying with a short sleep lets
103/// a slow-boot daemon be detected and this caller exit 0, rather than
104/// proceeding to open redb and triggering `DatabaseAlreadyOpen`.
105/// What: calls `single_instance_check` repeatedly up to `1 + max_retries`
106/// times, sleeping `delay_ms` between each call, stopping on the first
107/// non-`Proceed` result. Returns the final `StartupAction`.
108/// Test: covered by the unit tests for `startup_action_from_probe_result`;
109/// the retry path is exercised by the integration guard in `main.rs`.
110pub async fn single_instance_check_retried(
111    socket: &Path,
112    max_retries: u8,
113    delay_ms: u64,
114) -> StartupAction {
115    let mut action = single_instance_check(socket).await;
116    let mut retries = max_retries;
117    while action == StartupAction::Proceed && retries > 0 {
118        retries -= 1;
119        tokio::time::sleep(std::time::Duration::from_millis(delay_ms)).await;
120        action = single_instance_check(socket).await;
121    }
122    action
123}
124
125#[cfg(test)]
126mod tests {
127    use super::*;
128
129    /// Why: when the health probe returns `Some(url)` (daemon is alive),
130    /// the startup action must be `ExitAlreadyRunning` so the caller can
131    /// exit 0 and stop the launchd respawn storm.
132    /// What: asserts the mapping for `probe_ok = true`.
133    /// Test: itself (pure function, no I/O).
134    #[test]
135    fn startup_action_from_probe_result_when_alive() {
136        assert_eq!(
137            startup_action_from_probe_result(true),
138            StartupAction::ExitAlreadyRunning,
139            "alive probe → ExitAlreadyRunning"
140        );
141    }
142
143    /// Why: when the health probe returns `None` (addr file missing, stale,
144    /// or daemon not responding), the startup action must be `Proceed` so the
145    /// daemon continues with its normal bind sequence.
146    /// What: asserts the mapping for `probe_ok = false`.
147    /// Test: itself (pure function, no I/O).
148    #[test]
149    fn startup_action_from_probe_result_when_dead() {
150        assert_eq!(
151            startup_action_from_probe_result(false),
152            StartupAction::Proceed,
153            "dead/absent probe → Proceed"
154        );
155    }
156
157    /// Why: an absent socket means no daemon is running, and the guard must
158    /// let the cold start proceed. A guard that reported "already running" for
159    /// a path with nothing on it would stop the daemon ever starting.
160    /// Test: itself.
161    #[tokio::test]
162    async fn single_instance_check_proceeds_when_nothing_is_serving() {
163        let tmp = tempfile::tempdir().expect("tempdir");
164        let action = single_instance_check(&tmp.path().join("absent.sock")).await;
165        assert_eq!(
166            action,
167            StartupAction::Proceed,
168            "an absent socket must never block a cold start"
169        );
170    }
171
172    /// Why: this is the branch that collapses the launchd respawn herd. A live
173    /// socket must produce `ExitAlreadyRunning` so the second instance exits 0
174    /// rather than failing its bind and being respawned.
175    /// Test: itself.
176    #[tokio::test(flavor = "multi_thread")]
177    async fn single_instance_check_exits_when_something_is_serving() {
178        let tmp = tempfile::tempdir().expect("tempdir");
179        let socket = tmp.path().join("sockets").join("trusty-memory.sock");
180        let listener = trusty_common::uds::bind_hardened(&socket).expect("bind");
181        tokio::spawn(async move { while listener.accept().await.is_ok() {} });
182
183        assert_eq!(
184            single_instance_check(&socket).await,
185            StartupAction::ExitAlreadyRunning,
186            "a live socket must stop a second instance from binding"
187        );
188    }
189}