weida_runtime/unix.rs
1//! `AF_UNIX` bind hygiene and the kernel's answer to *who is on the other
2//! end*.
3//!
4//! A filesystem endpoint is not a port: it has an owner, a mode, a path
5//! length the kernel truncates at, and a node that outlives the process that
6//! bound it. Every one of those is a hazard with a known answer, and the
7//! answers are the same for any protocol that offers an `ipc://`-style
8//! transport — weida's `weida+unix://` and ZeroMQ's `ipc://` differ in what
9//! they write on the socket, not in how they bind it
10//! ([decisions/0010](../../../docs/decisions/0010-local-transport.md) §4.5,
11//! `docs/research/ipc.md` §1.2, §1.5, §7).
12
13use std::os::unix::fs::PermissionsExt;
14use std::path::{Path, PathBuf};
15
16use tokio::net::{UnixListener, UnixStream};
17use weida_core::{Error, LocalPrincipal, MAX_SOCKET_PATH_BYTES};
18
19/// Mode of a bound socket file.
20///
21/// Set explicitly after bind, never inherited: a new socket file gets every
22/// permission bit `umask` does not mask, so the default is whatever the
23/// process happened to inherit (`docs/research/ipc.md` §1.2, [0010 §4.5]).
24const SOCKET_MODE: u32 = 0o600;
25
26/// A bound `AF_UNIX` socket file: the node exists as long as this value
27/// does.
28///
29/// Hold it beside the [`UnixListener`] it was bound with. Dropping it removes
30/// the node, which is what keeps the *next* bind of the same path out of the
31/// unlink-then-bind race below.
32#[derive(Debug)]
33pub struct BoundUnixSocket {
34 path: PathBuf,
35}
36
37impl BoundUnixSocket {
38 /// Binds `path` with the hygiene a filesystem endpoint needs, replacing a
39 /// stale socket file left by a crash.
40 ///
41 /// Four things happen here, and each answers a documented hazard:
42 ///
43 /// 1. **The path budget.** `sun_path` is 108 bytes on Linux and 104 on
44 /// macOS *including* the terminator, and the kernel truncates rather
45 /// than failing, so an over-long path binds something other than what
46 /// was asked for. It is refused instead
47 /// (`docs/research/ipc.md` §1.1, §2.1). libzmq's `ipc://` publishes
48 /// the same limit as "113 characters including the prefix" and leaves
49 /// the rest to the caller (`docs/research/zeromq.md` §11).
50 /// 2. **The socket-type check.** Closing a socket does not remove its
51 /// node, so a crash leaves one and `bind()` then fails with
52 /// `EADDRINUSE`. A stale node is a socket nobody is listening on;
53 /// anything else at that path is not ours to remove, and removing it
54 /// anyway is how a bind deletes a caller's data.
55 /// 3. **Unlink, then bind.** The usual answer to the stale node, and it
56 /// opens a substitution race that is closed only "unless directory
57 /// ownership and permissions prevent endpoint substitution"
58 /// (`docs/research/ipc.md` §1.2, §7). **The directory is therefore
59 /// load-bearing**: the caller MUST place the socket in a directory it
60 /// owns and that no other user may write. libzmq's `ipc://` has this
61 /// hazard too and answers none of it — a local process can steal a
62 /// bound endpoint.
63 /// 4. **The mode, explicitly.** `0600` set *after* bind, because a socket
64 /// file is created with whatever `umask` allows, which is whatever the
65 /// process happened to inherit.
66 pub fn bind(path: &Path) -> Result<(BoundUnixSocket, UnixListener), Error> {
67 if path.as_os_str().len() > MAX_SOCKET_PATH_BYTES {
68 return Err(Error::InvalidAddress(format!(
69 "socket path exceeds this platform's {}-byte sun_path budget: {}",
70 MAX_SOCKET_PATH_BYTES,
71 path.display()
72 )));
73 }
74 // A stale node is a socket nobody is listening on. Anything else at
75 // that path is not ours to remove.
76 match std::fs::metadata(path) {
77 Ok(meta) if is_socket(&meta) => std::fs::remove_file(path).map_err(Error::Io)?,
78 Ok(_) => {
79 return Err(Error::InvalidAddress(format!(
80 "{} exists and is not a socket",
81 path.display()
82 )));
83 }
84 Err(_) => {}
85 }
86 let listener = UnixListener::bind(path).map_err(Error::Io)?;
87 std::fs::set_permissions(path, std::fs::Permissions::from_mode(SOCKET_MODE))
88 .map_err(Error::Io)?;
89 Ok((
90 BoundUnixSocket {
91 path: path.to_path_buf(),
92 },
93 listener,
94 ))
95 }
96
97 /// The path this socket is bound at.
98 pub fn path(&self) -> &Path {
99 &self.path
100 }
101}
102
103impl Drop for BoundUnixSocket {
104 fn drop(&mut self) {
105 // Leaving the node behind is what forces the next bind into
106 // unlink-then-bind; removing it on the way out keeps the common case
107 // free of that race.
108 let _ = std::fs::remove_file(&self.path);
109 }
110}
111
112fn is_socket(meta: &std::fs::Metadata) -> bool {
113 use std::os::unix::fs::FileTypeExt;
114 meta.file_type().is_socket()
115}
116
117/// The credentials the kernel attributes to the peer of `stream`.
118///
119/// This is the whole of a local peer's identity: there is no key, no
120/// certificate and nothing the peer asserts about itself — the kernel says
121/// what it is, which is why it can be trusted at all
122/// ([decisions/0010](../../../docs/decisions/0010-local-transport.md) §4.4,
123/// §4.5).
124///
125/// Taken at accept or connect time, which is when `SO_PEERCRED` captures
126/// them; they are **not** re-read per message, and re-reading would not help:
127/// the values are a snapshot of the peer as it was when the socket was
128/// created (`docs/research/ipc.md` §1.5). The pid is an `Option` because
129/// macOS's `LOCAL_PEERCRED` reports none, and because a pid is an
130/// observation — it may be reused — rather than an identity.
131///
132/// A protocol that uses this as an authorization input should say so
133/// explicitly: libzmq's `ZMQ_IPC_FILTER_UID`/`_GID`/`_PID` did exactly this
134/// and are deprecated in favour of ZAP, which is a decision about *where*
135/// authorization lives and not about whether the kernel's answer is true
136/// (`docs/research/zeromq.md` §10).
137pub fn peer_credentials(stream: &UnixStream) -> Result<LocalPrincipal, Error> {
138 let cred = stream.peer_cred().map_err(Error::Io)?;
139 Ok(LocalPrincipal {
140 uid: cred.uid(),
141 gid: cred.gid(),
142 // macOS reports no PID at all, and a PID is an observation even where
143 // it exists [0010 §4.4].
144 pid: cred.pid().map(|pid| pid as u32),
145 })
146}
147
148#[cfg(test)]
149mod tests {
150 use super::*;
151
152 /// A private directory per test, which is also what the substitution race
153 /// of `bind` requires of a caller.
154 fn dir(tag: &str) -> PathBuf {
155 let dir = std::env::temp_dir().join(format!("weida-runtime-{}-{tag}", std::process::id()));
156 std::fs::create_dir_all(&dir).expect("test directory");
157 dir
158 }
159
160 /// Claim: the mode is what this function sets, not what `umask` allowed.
161 #[tokio::test]
162 async fn a_bound_socket_is_private_to_its_owner() {
163 let path = dir("mode").join("s");
164 let (bound, _listener) = BoundUnixSocket::bind(&path).expect("bind");
165 let mode = std::fs::metadata(&path)
166 .expect("metadata")
167 .permissions()
168 .mode();
169 assert_eq!(mode & 0o777, SOCKET_MODE, "mode {mode:o}");
170 assert_eq!(bound.path(), path.as_path());
171 }
172
173 /// Claim: the node goes away with the binding, so the next bind of the
174 /// same path is not an unlink-then-bind at all.
175 #[tokio::test]
176 async fn the_node_is_removed_on_drop() {
177 let path = dir("drop").join("s");
178 let (bound, listener) = BoundUnixSocket::bind(&path).expect("bind");
179 assert!(path.exists());
180 drop(listener);
181 drop(bound);
182 assert!(!path.exists(), "the socket node outlived its binding");
183 }
184
185 /// Claim: a stale node — a socket file with nobody listening, which is
186 /// what a crash leaves — is replaced rather than reported as
187 /// `EADDRINUSE`.
188 #[tokio::test]
189 async fn a_stale_socket_is_replaced() {
190 let path = dir("stale").join("s");
191 {
192 let (bound, listener) = BoundUnixSocket::bind(&path).expect("first bind");
193 // Forget the guard the way a crash does: the node stays behind.
194 std::mem::forget(bound);
195 drop(listener);
196 }
197 assert!(path.exists(), "the stale node must still be there");
198 let (_bound, _listener) = BoundUnixSocket::bind(&path).expect("bind over the stale node");
199 }
200
201 /// Claim: anything at that path that is not a socket is refused, because
202 /// unlinking it would delete somebody's file.
203 #[tokio::test]
204 async fn a_regular_file_is_not_unlinked() {
205 let path = dir("file").join("s");
206 std::fs::write(&path, b"not a socket").expect("write");
207 let err = BoundUnixSocket::bind(&path).unwrap_err();
208 assert!(matches!(err, Error::InvalidAddress(_)), "{err:?}");
209 assert_eq!(std::fs::read(&path).expect("read"), b"not a socket");
210 }
211
212 /// Claim: a path the kernel would truncate is refused before it binds
213 /// something other than what was asked for.
214 #[tokio::test]
215 async fn an_over_long_path_is_refused() {
216 let path = dir("long").join("x".repeat(MAX_SOCKET_PATH_BYTES + 1));
217 let err = BoundUnixSocket::bind(&path).unwrap_err();
218 assert!(matches!(err, Error::InvalidAddress(_)), "{err:?}");
219 }
220
221 /// Claim: the credentials are the kernel's, and on a socket between two
222 /// halves of this process they are this process's own.
223 #[tokio::test]
224 async fn peer_credentials_are_the_kernels_answer() {
225 let (a, _b) = UnixStream::pair().expect("socket pair");
226 let principal = peer_credentials(&a).expect("credentials");
227 assert_eq!(principal.uid, unsafe_free_uid());
228 if let Some(pid) = principal.pid {
229 assert_eq!(pid, std::process::id());
230 }
231 }
232
233 /// This process's uid without `unsafe`: the effective uid of the owner of
234 /// a file this process just created.
235 fn unsafe_free_uid() -> u32 {
236 use std::os::unix::fs::MetadataExt;
237 let path = dir("uid").join("owned");
238 std::fs::write(&path, b"").expect("write");
239 let uid = std::fs::metadata(&path).expect("metadata").uid();
240 let _ = std::fs::remove_file(&path);
241 uid
242 }
243}