Skip to main content

weida_runtime/
unix.rs

1//! `AF_UNIX` bind hygiene and the kernel's answer to *who is on the other
2//! end*.
3//!
4//! A filesystem endpoint is not a port: it has an owner, a mode, a path
5//! length the kernel truncates at, and a node that outlives the process that
6//! bound it. Every one of those is a hazard with a known answer, and the
7//! answers are the same for any protocol that offers an `ipc://`-style
8//! transport — weida's `weida+unix://` and ZeroMQ's `ipc://` differ in what
9//! they write on the socket, not in how they bind it
10//! ([decisions/0010](../../../docs/decisions/0010-local-transport.md) §4.5,
11//! `docs/research/ipc.md` §1.2, §1.5, §7).
12
13use std::os::unix::fs::PermissionsExt;
14use std::path::{Path, PathBuf};
15
16use tokio::net::{UnixListener, UnixStream};
17use weida_core::{Error, LocalPrincipal, MAX_SOCKET_PATH_BYTES};
18
19/// Mode of a bound socket file.
20///
21/// Set explicitly after bind, never inherited: a new socket file gets every
22/// permission bit `umask` does not mask, so the default is whatever the
23/// process happened to inherit (`docs/research/ipc.md` §1.2, [0010 §4.5]).
24const SOCKET_MODE: u32 = 0o600;
25
26/// A bound `AF_UNIX` socket file: the node exists as long as this value
27/// does.
28///
29/// Hold it beside the [`UnixListener`] it was bound with. Dropping it removes
30/// the node, which is what keeps the *next* bind of the same path out of the
31/// unlink-then-bind race below.
32#[derive(Debug)]
33pub struct BoundUnixSocket {
34    path: PathBuf,
35}
36
37impl BoundUnixSocket {
38    /// Binds `path` with the hygiene a filesystem endpoint needs, replacing a
39    /// stale socket file left by a crash.
40    ///
41    /// Four things happen here, and each answers a documented hazard:
42    ///
43    /// 1. **The path budget.** `sun_path` is 108 bytes on Linux and 104 on
44    ///    macOS *including* the terminator, and the kernel truncates rather
45    ///    than failing, so an over-long path binds something other than what
46    ///    was asked for. It is refused instead
47    ///    (`docs/research/ipc.md` §1.1, §2.1). libzmq's `ipc://` publishes
48    ///    the same limit as "113 characters including the prefix" and leaves
49    ///    the rest to the caller (`docs/research/zeromq.md` §11).
50    /// 2. **The socket-type check.** Closing a socket does not remove its
51    ///    node, so a crash leaves one and `bind()` then fails with
52    ///    `EADDRINUSE`. A stale node is a socket nobody is listening on;
53    ///    anything else at that path is not ours to remove, and removing it
54    ///    anyway is how a bind deletes a caller's data.
55    /// 3. **Unlink, then bind.** The usual answer to the stale node, and it
56    ///    opens a substitution race that is closed only "unless directory
57    ///    ownership and permissions prevent endpoint substitution"
58    ///    (`docs/research/ipc.md` §1.2, §7). **The directory is therefore
59    ///    load-bearing**: the caller MUST place the socket in a directory it
60    ///    owns and that no other user may write. libzmq's `ipc://` has this
61    ///    hazard too and answers none of it — a local process can steal a
62    ///    bound endpoint.
63    /// 4. **The mode, explicitly.** `0600` set *after* bind, because a socket
64    ///    file is created with whatever `umask` allows, which is whatever the
65    ///    process happened to inherit.
66    pub fn bind(path: &Path) -> Result<(BoundUnixSocket, UnixListener), Error> {
67        if path.as_os_str().len() > MAX_SOCKET_PATH_BYTES {
68            return Err(Error::InvalidAddress(format!(
69                "socket path exceeds this platform's {}-byte sun_path budget: {}",
70                MAX_SOCKET_PATH_BYTES,
71                path.display()
72            )));
73        }
74        // A stale node is a socket nobody is listening on. Anything else at
75        // that path is not ours to remove.
76        match std::fs::metadata(path) {
77            Ok(meta) if is_socket(&meta) => std::fs::remove_file(path).map_err(Error::Io)?,
78            Ok(_) => {
79                return Err(Error::InvalidAddress(format!(
80                    "{} exists and is not a socket",
81                    path.display()
82                )));
83            }
84            Err(_) => {}
85        }
86        let listener = UnixListener::bind(path).map_err(Error::Io)?;
87        std::fs::set_permissions(path, std::fs::Permissions::from_mode(SOCKET_MODE))
88            .map_err(Error::Io)?;
89        Ok((
90            BoundUnixSocket {
91                path: path.to_path_buf(),
92            },
93            listener,
94        ))
95    }
96
97    /// The path this socket is bound at.
98    pub fn path(&self) -> &Path {
99        &self.path
100    }
101}
102
103impl Drop for BoundUnixSocket {
104    fn drop(&mut self) {
105        // Leaving the node behind is what forces the next bind into
106        // unlink-then-bind; removing it on the way out keeps the common case
107        // free of that race.
108        let _ = std::fs::remove_file(&self.path);
109    }
110}
111
112fn is_socket(meta: &std::fs::Metadata) -> bool {
113    use std::os::unix::fs::FileTypeExt;
114    meta.file_type().is_socket()
115}
116
117/// The credentials the kernel attributes to the peer of `stream`.
118///
119/// This is the whole of a local peer's identity: there is no key, no
120/// certificate and nothing the peer asserts about itself — the kernel says
121/// what it is, which is why it can be trusted at all
122/// ([decisions/0010](../../../docs/decisions/0010-local-transport.md) §4.4,
123/// §4.5).
124///
125/// Taken at accept or connect time, which is when `SO_PEERCRED` captures
126/// them; they are **not** re-read per message, and re-reading would not help:
127/// the values are a snapshot of the peer as it was when the socket was
128/// created (`docs/research/ipc.md` §1.5). The pid is an `Option` because
129/// macOS's `LOCAL_PEERCRED` reports none, and because a pid is an
130/// observation — it may be reused — rather than an identity.
131///
132/// A protocol that uses this as an authorization input should say so
133/// explicitly: libzmq's `ZMQ_IPC_FILTER_UID`/`_GID`/`_PID` did exactly this
134/// and are deprecated in favour of ZAP, which is a decision about *where*
135/// authorization lives and not about whether the kernel's answer is true
136/// (`docs/research/zeromq.md` §10).
137pub fn peer_credentials(stream: &UnixStream) -> Result<LocalPrincipal, Error> {
138    let cred = stream.peer_cred().map_err(Error::Io)?;
139    Ok(LocalPrincipal {
140        uid: cred.uid(),
141        gid: cred.gid(),
142        // macOS reports no PID at all, and a PID is an observation even where
143        // it exists [0010 §4.4].
144        pid: cred.pid().map(|pid| pid as u32),
145    })
146}
147
148#[cfg(test)]
149mod tests {
150    use super::*;
151
152    /// A private directory per test, which is also what the substitution race
153    /// of `bind` requires of a caller.
154    fn dir(tag: &str) -> PathBuf {
155        let dir = std::env::temp_dir().join(format!("weida-runtime-{}-{tag}", std::process::id()));
156        std::fs::create_dir_all(&dir).expect("test directory");
157        dir
158    }
159
160    /// Claim: the mode is what this function sets, not what `umask` allowed.
161    #[tokio::test]
162    async fn a_bound_socket_is_private_to_its_owner() {
163        let path = dir("mode").join("s");
164        let (bound, _listener) = BoundUnixSocket::bind(&path).expect("bind");
165        let mode = std::fs::metadata(&path)
166            .expect("metadata")
167            .permissions()
168            .mode();
169        assert_eq!(mode & 0o777, SOCKET_MODE, "mode {mode:o}");
170        assert_eq!(bound.path(), path.as_path());
171    }
172
173    /// Claim: the node goes away with the binding, so the next bind of the
174    /// same path is not an unlink-then-bind at all.
175    #[tokio::test]
176    async fn the_node_is_removed_on_drop() {
177        let path = dir("drop").join("s");
178        let (bound, listener) = BoundUnixSocket::bind(&path).expect("bind");
179        assert!(path.exists());
180        drop(listener);
181        drop(bound);
182        assert!(!path.exists(), "the socket node outlived its binding");
183    }
184
185    /// Claim: a stale node — a socket file with nobody listening, which is
186    /// what a crash leaves — is replaced rather than reported as
187    /// `EADDRINUSE`.
188    #[tokio::test]
189    async fn a_stale_socket_is_replaced() {
190        let path = dir("stale").join("s");
191        {
192            let (bound, listener) = BoundUnixSocket::bind(&path).expect("first bind");
193            // Forget the guard the way a crash does: the node stays behind.
194            std::mem::forget(bound);
195            drop(listener);
196        }
197        assert!(path.exists(), "the stale node must still be there");
198        let (_bound, _listener) = BoundUnixSocket::bind(&path).expect("bind over the stale node");
199    }
200
201    /// Claim: anything at that path that is not a socket is refused, because
202    /// unlinking it would delete somebody's file.
203    #[tokio::test]
204    async fn a_regular_file_is_not_unlinked() {
205        let path = dir("file").join("s");
206        std::fs::write(&path, b"not a socket").expect("write");
207        let err = BoundUnixSocket::bind(&path).unwrap_err();
208        assert!(matches!(err, Error::InvalidAddress(_)), "{err:?}");
209        assert_eq!(std::fs::read(&path).expect("read"), b"not a socket");
210    }
211
212    /// Claim: a path the kernel would truncate is refused before it binds
213    /// something other than what was asked for.
214    #[tokio::test]
215    async fn an_over_long_path_is_refused() {
216        let path = dir("long").join("x".repeat(MAX_SOCKET_PATH_BYTES + 1));
217        let err = BoundUnixSocket::bind(&path).unwrap_err();
218        assert!(matches!(err, Error::InvalidAddress(_)), "{err:?}");
219    }
220
221    /// Claim: the credentials are the kernel's, and on a socket between two
222    /// halves of this process they are this process's own.
223    #[tokio::test]
224    async fn peer_credentials_are_the_kernels_answer() {
225        let (a, _b) = UnixStream::pair().expect("socket pair");
226        let principal = peer_credentials(&a).expect("credentials");
227        assert_eq!(principal.uid, unsafe_free_uid());
228        if let Some(pid) = principal.pid {
229            assert_eq!(pid, std::process::id());
230        }
231    }
232
233    /// This process's uid without `unsafe`: the effective uid of the owner of
234    /// a file this process just created.
235    fn unsafe_free_uid() -> u32 {
236        use std::os::unix::fs::MetadataExt;
237        let path = dir("uid").join("owned");
238        std::fs::write(&path, b"").expect("write");
239        let uid = std::fs::metadata(&path).expect("metadata").uid();
240        let _ = std::fs::remove_file(&path);
241        uid
242    }
243}