Skip to main content

rlmctl_core/guard/
systemd.rs

1//! Blocking systemd user-bus (`org.freedesktop.systemd1`) client used on the
2//! freeze-guard's storm path as an alternative to fork+exec'ing `systemctl`.
3//!
4//! The connection is established once at daemon startup (pre-storm, when a
5//! slow session-bus handshake doesn't matter) and reused for every call. Each
6//! individual call still gets its own hard timeout: zbus's own blocking calls
7//! default to a 25s method-call timeout, which is far too long for a path that
8//! exists specifically to react before the system locks up. We enforce our own
9//! deadline by running the call on a spawned thread and waiting on it via
10//! `mpsc::Receiver::recv_timeout`. If the deadline passes first we return
11//! `Err` immediately; the spawned thread is left to finish (or never finish,
12//! if the bus is wedged) in the background and its result is dropped. This is
13//! a bounded leak (one thread per timed-out call), accepted because the
14//! storm path always falls back to raw cgroup writes on `Err`, so a wedged
15//! D-Bus call never blocks the daemon itself.
16
17use std::sync::mpsc;
18use std::thread;
19use std::time::Duration;
20
21use common::{Error, Result};
22use zbus::blocking::Connection;
23
24const DESTINATION: &str = "org.freedesktop.systemd1";
25const PATH: &str = "/org/freedesktop/systemd1";
26const INTERFACE: &str = "org.freedesktop.systemd1.Manager";
27
28/// Longest wait for the session-bus handshake at startup. Startup recovery
29/// runs after the connect, so a wedged bus must not hold a frozen app.
30pub const CONNECT_TIMEOUT: Duration = Duration::from_secs(2);
31
32/// Blocking client for the systemd user-session D-Bus manager.
33pub struct SystemdUser {
34    conn: Connection,
35}
36
37impl SystemdUser {
38    /// Connect to the session bus at daemon startup. Returns `None` if no
39    /// session bus is available (e.g. headless, no `DBUS_SESSION_BUS_ADDRESS`)
40    /// or it does not answer within [`CONNECT_TIMEOUT`]. Callers then fall
41    /// back to raw cgroup operations everywhere.
42    pub fn connect() -> Option<Self> {
43        connect_within(CONNECT_TIMEOUT, || {
44            Connection::session().map_err(|e| Error::Cgroup(format!("session bus: {e}")))
45        })
46    }
47
48    /// `FreezeUnit(name)`. Returns `Err` on failure OR timeout; the caller
49    /// falls back to a raw cgroup freeze.
50    pub fn freeze_unit(&self, unit: &str, timeout: Duration) -> Result<()> {
51        let unit = unit.to_string();
52        self.call_with_timeout(timeout, move |conn| {
53            conn.call_method(
54                Some(DESTINATION),
55                PATH,
56                Some(INTERFACE),
57                "FreezeUnit",
58                &(unit.as_str(),),
59            )
60            .map(|_| ())
61            .map_err(|e| Error::Cgroup(format!("FreezeUnit({unit}) failed: {e}")))
62        })
63    }
64
65    /// `ThawUnit(name)`. Returns `Err` on failure OR timeout; the caller falls
66    /// back to a raw cgroup thaw.
67    pub fn thaw_unit(&self, unit: &str, timeout: Duration) -> Result<()> {
68        let unit = unit.to_string();
69        self.call_with_timeout(timeout, move |conn| {
70            conn.call_method(
71                Some(DESTINATION),
72                PATH,
73                Some(INTERFACE),
74                "ThawUnit",
75                &(unit.as_str(),),
76            )
77            .map(|_| ())
78            .map_err(|e| Error::Cgroup(format!("ThawUnit({unit}) failed: {e}")))
79        })
80    }
81
82    /// Run `f` against a clone of the connection on a spawned thread, and
83    /// enforce `timeout` ourselves rather than trusting zbus's own (25s)
84    /// default. See [`run_with_timeout`] for the timeout mechanics.
85    fn call_with_timeout(
86        &self,
87        timeout: Duration,
88        f: impl FnOnce(&Connection) -> Result<()> + Send + 'static,
89    ) -> Result<()> {
90        let conn = self.conn.clone();
91        run_with_timeout(timeout, move || f(&conn))
92    }
93}
94
95/// Run the connect `f` under `timeout`; `None` on failure or timeout.
96fn connect_within(
97    timeout: Duration,
98    f: impl FnOnce() -> Result<Connection> + Send + 'static,
99) -> Option<SystemdUser> {
100    run_with_timeout(timeout, f)
101        .ok()
102        .map(|conn| SystemdUser { conn })
103}
104
105/// Run `f` on a spawned thread and wait for it via
106/// `mpsc::Receiver::recv_timeout(timeout)` instead of trusting the callee's
107/// own notion of a deadline. On timeout, returns `Err` immediately; the
108/// spawned thread is detached and left to finish (or never finish) in the
109/// background, with its eventual result silently dropped on send. This is a
110/// bounded leak (one thread per timed-out call), accepted because callers
111/// always treat `Err` as "fall back to raw cgroup ops", so a wedged call
112/// never blocks the daemon itself.
113fn run_with_timeout<T: Send + 'static>(
114    timeout: Duration,
115    f: impl FnOnce() -> Result<T> + Send + 'static,
116) -> Result<T> {
117    let (tx, rx) = mpsc::channel();
118    thread::spawn(move || {
119        // Ignore send errors: the receiver may already have timed out and
120        // been dropped, in which case there's nothing left to notify.
121        let _ = tx.send(f());
122    });
123    rx.recv_timeout(timeout).unwrap_or_else(|_| {
124        Err(Error::Cgroup(format!(
125            "D-Bus call timed out after {timeout:?}"
126        )))
127    })
128}
129
130#[cfg(test)]
131mod tests {
132    use super::*;
133
134    /// Pure test of the timeout wrapper: a closure that outlives the deadline
135    /// must cause `run_with_timeout` to return `Err` promptly, without
136    /// requiring a real session bus.
137    #[test]
138    fn run_with_timeout_returns_err_on_timeout() {
139        let result: Result<()> = run_with_timeout(Duration::from_millis(50), || {
140            thread::sleep(Duration::from_secs(5));
141            Ok(())
142        });
143        assert!(result.is_err(), "expected a timeout error");
144    }
145
146    /// A closure that finishes well within the deadline should return its
147    /// own `Ok` value through unchanged.
148    #[test]
149    fn run_with_timeout_returns_ok_when_fast() {
150        let result: Result<u32> = run_with_timeout(Duration::from_secs(2), || Ok(42));
151        assert_eq!(result.unwrap(), 42);
152    }
153
154    /// A session bus that never answers must not hold up startup.
155    #[test]
156    fn connect_gives_up_on_a_slow_bus() {
157        let start = std::time::Instant::now();
158        let conn = connect_within(Duration::from_millis(50), || {
159            thread::sleep(Duration::from_secs(5));
160            Err(Error::Cgroup("never".into()))
161        });
162        assert!(conn.is_none());
163        assert!(start.elapsed() < Duration::from_secs(2));
164    }
165
166    #[test]
167    #[ignore = "requires a session bus and a running user unit; run manually"]
168    fn freeze_thaw_transient_unit_roundtrip() {
169        // systemd-run --user --unit=rlm-dbus-test sleep 30 must be running.
170        let s = SystemdUser::connect().expect("session bus");
171        s.freeze_unit("rlm-dbus-test.service", Duration::from_secs(2))
172            .expect("freeze");
173        s.thaw_unit("rlm-dbus-test.service", Duration::from_secs(2))
174            .expect("thaw");
175    }
176}