Skip to main content

trusty_memory/commands/
single_instance.rs

1//! Single-instance guard for the trusty-memory daemon.
2//!
3//! Why: macOS launchd `KeepAlive { SuccessfulExit: false }` respawns the daemon
4//! whenever it exits with a non-zero code. A second instance that fails to bind
5//! exits non-zero, launchd reads that as a crash and spawns another copy, and
6//! the resulting zombie herd (69 observed in the wild) exhausts file
7//! descriptors on top of the existing fd-limit bug.
8//!
9//! The fix: before binding, probe the socket. If something is already serving
10//! it, exit **0**. Launchd treats exit-0 as a clean shutdown and does not
11//! respawn, which collapses the herd on the next invocation without touching
12//! the launchd config.
13//!
14//! #6286 changed what is probed, not the decision. It used to read the
15//! `http_addr` discovery file and GET `/health` at whatever address that named
16//! — a file that goes stale after a SIGKILL, which is why the probe had to
17//! tolerate one pointing at a dead port. The socket path is derived rather than
18//! published, so there is nothing to be stale and the probe is a bare connect
19//! through `trusty_common::uds::socket_is_serving`.
20//!
21//! What: [`single_instance_check`] (async, for real daemon startups) and
22//! [`StartupAction`] (a pure enum, so the decision is unit-testable without
23//! I/O).
24//!
25//! Test: `startup_action_*` for the decision, `single_instance_check_*` for the
26//! probe.
27
28use std::path::Path;
29use std::time::Duration;
30
31/// How long a liveness connect may take before the path is called dead.
32///
33/// A local socket accepts or refuses in microseconds; this is headroom for a
34/// loaded machine, not a latency budget. It matches trusty-analyze's
35/// `daemon_guard::PROBE_TIMEOUT` so the two daemons wait the same.
36const PROBE_TIMEOUT: Duration = Duration::from_millis(500);
37
38/// What the daemon startup should do after the single-instance check.
39///
40/// Why: separating the decision from the I/O lets us unit-test the logic
41/// with injected probe results rather than spinning up real TCP listeners.
42/// What: three variants covering the full decision tree.
43/// Test: `startup_action_from_probe_result_*` tests in this module.
44#[derive(Debug, Clone, PartialEq, Eq)]
45pub enum StartupAction {
46    /// Proceed to bind the socket and start serving.
47    Proceed,
48    /// Another healthy instance is already running — exit 0 cleanly so
49    /// launchd does not respawn.
50    ExitAlreadyRunning,
51    /// A probe attempt failed with an unexpected error — propagate as a startup
52    /// failure so the operator sees a real error in the launchd log. Launchd
53    /// respawns, correctly, because this is a genuine failure.
54    ///
55    /// Nothing constructs this today: `socket_is_serving` answers a bool, so
56    /// every failure it can have is "nothing is serving". The variant stays
57    /// because the caller in `main.rs` branches on it and a future probe that
58    /// can distinguish a permission error from an absence has somewhere to
59    /// report it.
60    Fail(String),
61}
62
63/// Decide what to do based on the result of a liveness probe.
64///
65/// Why: the single-instance check reduces to "did the health probe succeed?".
66/// Encoding the decision as a pure function (rather than embedding it in the
67/// async probe body) makes the logic unit-testable without actual network I/O.
68/// What: `probe_ok = true` → [`StartupAction::ExitAlreadyRunning`];
69/// `probe_ok = false` → [`StartupAction::Proceed`].
70/// Test: `startup_action_from_probe_result_when_alive`,
71///       `startup_action_from_probe_result_when_dead`.
72pub fn startup_action_from_probe_result(probe_ok: bool) -> StartupAction {
73    if probe_ok {
74        StartupAction::ExitAlreadyRunning
75    } else {
76        StartupAction::Proceed
77    }
78}
79
80/// Whether `socket` is the path a launchd-managed trusty-memory would serve
81/// (#6619).
82///
83/// Why: every #6619 guard turns on this one question, and getting it wrong in
84/// either direction is a bug — too broad refuses a sandboxed daemon that launchd
85/// never managed, too narrow lets the orphan back onto the production path. One
86/// definition, used by both the caller-side spawn guard and the callee-side bind
87/// refusal.
88/// What: `socket` must be what [`crate::socket_path`] resolves AND the process
89/// must not be running under a `TRUSTY_DATA_DIR_OVERRIDE` sandbox — the plist
90/// sets no such override, so a process that has one is by construction not the
91/// unit launchd runs.
92/// Test: `production_socket_is_the_resolved_path`,
93/// `production_socket_is_false_under_a_data_dir_override`.
94#[must_use]
95pub fn is_production_socket(socket: &Path) -> bool {
96    if std::env::var_os(trusty_common::DATA_DIR_OVERRIDE_ENV).is_some() {
97        return false;
98    }
99    crate::socket_path().is_ok_and(|p| p == socket)
100}
101
102/// Refuse the production socket when this process is not the launchd unit that
103/// owns it (#6619).
104///
105/// Why: the spawn guard stops a bridge from racing launchd, but only a bridge
106/// that runs the fixed code. This is the callee-side half, and it is the one
107/// that holds for a daemon started any other way — by hand, by an older bridge,
108/// by a script. Binding the production socket unsupervised is what left an
109/// orphan without the plist's `FASTEMBED_CACHE_DIR` owning the path while
110/// launchd's own instance exited 0 reporting success.
111///
112/// What: three inputs, and the refusal needs all three.
113///
114/// - `is_production_socket` — a test or sandbox socket is never launchd's.
115/// - `unit_registered` — [`trusty_common::launchd_claim`]'s verdict.
116/// - `launchd_runs_us` — a POSITIVE answer from launchd that it does NOT run
117///   this PID (`LaunchdSupervision::NotSupervised`), never
118///   [`trusty_common::supervision::LaunchdSupervision::Unknown`].
119///
120/// 🔴 The `Unknown` exclusion is deliberate and is the difference between a
121/// guard and an outage. Both signals here read launchd, so a host where
122/// `launchctl` cannot be queried would report "unit registered" from the plist
123/// on disk and "not supervised" from a failed query — and refuse to start the
124/// launchd-supervised daemon on every boot. Refusing only on a positive
125/// `NotSupervised` fails closed against the observed defect (an unsupervised
126/// spawn on a healthy host) and open against an unreadable launchd.
127///
128/// Test: `bind_refused_for_an_unsupervised_process_on_a_registered_socket`,
129/// `bind_permitted_for_the_launchd_unit_itself`,
130/// `bind_permitted_when_no_unit_is_registered`,
131/// `bind_permitted_on_a_socket_launchd_does_not_own`,
132/// `bind_permitted_when_launchd_cannot_be_asked`.
133#[must_use]
134pub fn production_bind_refusal(
135    label: &str,
136    is_production_socket: bool,
137    unit_registered: bool,
138    supervision: &trusty_common::supervision::LaunchdSupervision,
139) -> Option<String> {
140    use trusty_common::supervision::LaunchdSupervision;
141
142    let positively_unsupervised = matches!(supervision, LaunchdSupervision::NotSupervised);
143    if !(is_production_socket && unit_registered && positively_unsupervised) {
144        return None;
145    }
146    Some(format!(
147        "refusing to bind the trusty-memory production socket: launchd unit \
148         {label} is registered for it and launchd does not run this process. An \
149         unsupervised daemon here starts without the plist's EnvironmentVariables \
150         and launchd's own instance then exits 0 reporting success (#6619). Start \
151         it with `launchctl kickstart -k gui/$(id -u)/{label}`, or point this \
152         process at a different socket"
153    ))
154}
155
156/// Perform the single-instance check at daemon startup.
157///
158/// Why: launchd respawns any non-zero exit, so a second instance that fails to
159/// bind causes an endless respawn storm. Exiting 0 when another healthy
160/// instance is detected short-circuits it.
161///
162/// What: a bare connect to `socket`. It deliberately does NOT call
163/// `memory.health`: the question is whether the endpoint is live, and a daemon
164/// that is up but degraded must not be reported absent and spawned on top of
165/// itself. An absent or dead socket returns [`StartupAction::Proceed`], so a
166/// cold start is never blocked.
167///
168/// Test: `single_instance_check_proceeds_when_nothing_is_serving`,
169/// `single_instance_check_exits_when_something_is_serving`.
170pub async fn single_instance_check(socket: &Path) -> StartupAction {
171    let probe_ok = trusty_common::uds::socket_is_serving(socket, PROBE_TIMEOUT).await;
172    startup_action_from_probe_result(probe_ok)
173}
174
175/// Single-instance check with up to `max_retries` additional probes.
176///
177/// Why (issue #1152, Tier 3): a single probe can miss a daemon that is
178/// mid-boot — it has not bound the socket yet. Retrying with a short sleep lets
179/// a slow-boot daemon be detected and this caller exit 0, rather than
180/// proceeding to open redb and triggering `DatabaseAlreadyOpen`.
181/// What: calls `single_instance_check` repeatedly up to `1 + max_retries`
182/// times, sleeping `delay_ms` between each call, stopping on the first
183/// non-`Proceed` result. Returns the final `StartupAction`.
184/// Test: covered by the unit tests for `startup_action_from_probe_result`;
185/// the retry path is exercised by the integration guard in `main.rs`.
186pub async fn single_instance_check_retried(
187    socket: &Path,
188    max_retries: u8,
189    delay_ms: u64,
190) -> StartupAction {
191    let mut action = single_instance_check(socket).await;
192    let mut retries = max_retries;
193    while action == StartupAction::Proceed && retries > 0 {
194        retries -= 1;
195        tokio::time::sleep(std::time::Duration::from_millis(delay_ms)).await;
196        action = single_instance_check(socket).await;
197    }
198    action
199}
200
201#[cfg(test)]
202mod tests {
203    use super::*;
204
205    /// Why: when the health probe returns `Some(url)` (daemon is alive),
206    /// the startup action must be `ExitAlreadyRunning` so the caller can
207    /// exit 0 and stop the launchd respawn storm.
208    /// What: asserts the mapping for `probe_ok = true`.
209    /// Test: itself (pure function, no I/O).
210    #[test]
211    fn startup_action_from_probe_result_when_alive() {
212        assert_eq!(
213            startup_action_from_probe_result(true),
214            StartupAction::ExitAlreadyRunning,
215            "alive probe → ExitAlreadyRunning"
216        );
217    }
218
219    /// Why: when the health probe returns `None` (addr file missing, stale,
220    /// or daemon not responding), the startup action must be `Proceed` so the
221    /// daemon continues with its normal bind sequence.
222    /// What: asserts the mapping for `probe_ok = false`.
223    /// Test: itself (pure function, no I/O).
224    #[test]
225    fn startup_action_from_probe_result_when_dead() {
226        assert_eq!(
227            startup_action_from_probe_result(false),
228            StartupAction::Proceed,
229            "dead/absent probe → Proceed"
230        );
231    }
232
233    /// Why: an absent socket means no daemon is running, and the guard must
234    /// let the cold start proceed. A guard that reported "already running" for
235    /// a path with nothing on it would stop the daemon ever starting.
236    /// Test: itself.
237    #[tokio::test]
238    async fn single_instance_check_proceeds_when_nothing_is_serving() {
239        let tmp = tempfile::tempdir().expect("tempdir");
240        let action = single_instance_check(&tmp.path().join("absent.sock")).await;
241        assert_eq!(
242            action,
243            StartupAction::Proceed,
244            "an absent socket must never block a cold start"
245        );
246    }
247
248    /// Why: this is the branch that collapses the launchd respawn herd. A live
249    /// socket must produce `ExitAlreadyRunning` so the second instance exits 0
250    /// rather than failing its bind and being respawned.
251    /// Test: itself.
252    #[tokio::test(flavor = "multi_thread")]
253    async fn single_instance_check_exits_when_something_is_serving() {
254        let tmp = tempfile::tempdir().expect("tempdir");
255        let socket = tmp.path().join("sockets").join("trusty-memory.sock");
256        let listener = trusty_common::uds::bind_hardened(&socket).expect("bind");
257        tokio::spawn(async move { while listener.accept().await.is_ok() {} });
258
259        assert_eq!(
260            single_instance_check(&socket).await,
261            StartupAction::ExitAlreadyRunning,
262            "a live socket must stop a second instance from binding"
263        );
264    }
265
266    use trusty_common::supervision::LaunchdSupervision;
267
268    /// The unit that owns the production socket.
269    const LABEL: &str = "com.trusty.memory";
270
271    /// Why (#6619): the observed orphan. A bridge-spawned daemon bound the
272    /// production socket while `com.trusty.memory` was mid-restart, without the
273    /// plist's `FASTEMBED_CACHE_DIR`, and launchd's own instance then exited 0
274    /// reporting success. The callee-side refusal is what holds even for a
275    /// daemon started by something this fix did not change.
276    /// What: production socket + registered unit + a positive "launchd does not
277    /// run this PID" refuses, naming the unit.
278    /// Test: itself.
279    #[test]
280    fn bind_refused_for_an_unsupervised_process_on_a_registered_socket() {
281        let refusal =
282            production_bind_refusal(LABEL, true, true, &LaunchdSupervision::NotSupervised)
283                .expect("an unsupervised bind of a registered socket must be refused");
284        assert!(refusal.contains(LABEL), "the unit must be named: {refusal}");
285    }
286
287    /// Why: the unit launchd runs IS the legitimate owner. Refusing it would
288    /// stop trusty-memory starting on every supervised host — an outage caused
289    /// by the guard.
290    /// What: a `Supervised` answer permits the bind.
291    /// Test: itself.
292    #[test]
293    fn bind_permitted_for_the_launchd_unit_itself() {
294        assert_eq!(
295            production_bind_refusal(
296                LABEL,
297                true,
298                true,
299                &LaunchdSupervision::Supervised(LABEL.to_owned())
300            ),
301            None
302        );
303    }
304
305    /// Why: a dev machine that never installed the service must keep starting
306    /// the daemon by hand.
307    /// What: no registered unit permits the bind.
308    /// Test: itself.
309    #[test]
310    fn bind_permitted_when_no_unit_is_registered() {
311        assert_eq!(
312            production_bind_refusal(LABEL, true, false, &LaunchdSupervision::NotSupervised),
313            None
314        );
315    }
316
317    /// Why: a sandboxed daemon under `TRUSTY_DATA_DIR_OVERRIDE` serves a path
318    /// launchd never manages, so the guard must not reach it — this is what
319    /// keeps the crate's own test daemons startable on an installed host.
320    /// What: a non-production socket permits the bind even with a unit
321    /// registered.
322    /// Test: itself.
323    #[test]
324    fn bind_permitted_on_a_socket_launchd_does_not_own() {
325        assert_eq!(
326            production_bind_refusal(LABEL, false, true, &LaunchdSupervision::NotSupervised),
327            None
328        );
329    }
330
331    /// Why: `Unknown` means launchd could not be ASKED, and both of this
332    /// guard's other signals read launchd too. Refusing on it would take the
333    /// daemon down on every host whose `launchctl list` is unreadable — trading
334    /// an orphan for an outage.
335    /// What: `Unknown` permits, unlike `NotSupervised`.
336    /// Test: itself.
337    #[test]
338    fn bind_permitted_when_launchd_cannot_be_asked() {
339        assert_eq!(
340            production_bind_refusal(
341                LABEL,
342                true,
343                true,
344                &LaunchdSupervision::Unknown("launchctl timed out".to_owned())
345            ),
346            None,
347            "an unanswerable launchd is not evidence of an orphan"
348        );
349    }
350
351    /// Why: every #6619 guard turns on this predicate, so the path it accepts
352    /// must be the one the daemon actually resolves — not a re-derived guess.
353    /// What: the resolved socket is production; a sibling path is not.
354    /// Test: itself.
355    #[test]
356    fn production_socket_is_the_resolved_path() {
357        let Ok(resolved) = crate::socket_path() else {
358            return; // no home directory in this environment; nothing to assert
359        };
360        if std::env::var_os(trusty_common::DATA_DIR_OVERRIDE_ENV).is_some() {
361            return; // a sibling test set the override; covered below instead
362        }
363        assert!(is_production_socket(&resolved));
364        assert!(!is_production_socket(Path::new("/tmp/not-the-daemon.sock")));
365    }
366
367    /// Why: a `TRUSTY_DATA_DIR_OVERRIDE` sandbox is by construction not the unit
368    /// launchd runs — the plist sets no override — so the guard must not reach
369    /// it whatever launchd has registered.
370    /// What: with the override set, even the resolved socket is not production.
371    /// Test: itself.
372    #[test]
373    #[serial_test::serial]
374    fn production_socket_is_false_under_a_data_dir_override() {
375        let tmp = tempfile::tempdir().expect("tempdir");
376        let previous = std::env::var_os(trusty_common::DATA_DIR_OVERRIDE_ENV);
377        // SAFETY: serialised by `#[serial]`; no concurrent env access here.
378        unsafe { std::env::set_var(trusty_common::DATA_DIR_OVERRIDE_ENV, tmp.path()) };
379
380        let resolved = crate::socket_path();
381        let verdict = resolved.as_ref().map(|p| is_production_socket(p));
382
383        // SAFETY: same as above.
384        unsafe {
385            match previous {
386                Some(v) => std::env::set_var(trusty_common::DATA_DIR_OVERRIDE_ENV, v),
387                None => std::env::remove_var(trusty_common::DATA_DIR_OVERRIDE_ENV),
388            }
389        }
390
391        assert_eq!(
392            verdict.ok(),
393            Some(false),
394            "a sandboxed socket is never launchd's production path"
395        );
396    }
397}