Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    let path = std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture");
134    // Without the fixture every macOS spawn is refused, and the tests that
135    // spawn fail later as a module in state Failed, which names the wrong
136    // cause. `cargo test -p subc-daemon --lib` alone does not build it.
137    assert!(
138        path.exists(),
139        "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140         --bins --features test-support` or `cargo test -p subc-daemon` first",
141        path.display()
142    );
143    path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148    use std::io::Read;
149    let mut probe = std::process::Command::new(path)
150        .args(["__disclaim-exec", "--probe"])
151        .stdin(Stdio::null())
152        .stdout(Stdio::piped())
153        .stderr(Stdio::piped())
154        .spawn()
155        .map_err(|error| {
156            format!(
157                "privacy trampoline probe failed for {}: {error}",
158                path.display()
159            )
160        })?;
161    let deadline = std::time::Instant::now() + Duration::from_secs(5);
162    let status = loop {
163        match probe.try_wait() {
164            Ok(Some(status)) => break status,
165            Ok(None) if std::time::Instant::now() < deadline => {
166                std::thread::sleep(Duration::from_millis(5))
167            }
168            result => {
169                let _ = probe.kill();
170                let _ = probe.wait();
171                return Err(format!(
172                    "privacy trampoline probe failed or timed out for {}: {result:?}",
173                    path.display()
174                ));
175            }
176        }
177    };
178    let mut answer = String::new();
179    if let Some(stdout) = probe.stdout.take() {
180        let _ = stdout.take(256).read_to_string(&mut answer);
181    }
182    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183        return Ok(());
184    }
185    let mut diagnostic = String::new();
186    if let Some(stderr) = probe.stderr.take() {
187        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188    }
189    let cause = diagnostic
190        .trim()
191        .strip_prefix("ck-subc: own privacy identity refused: ")
192        .unwrap_or("binary does not implement the privacy trampoline protocol");
193    Err(format!(
194        "{cause}: probe of {} exited {status}",
195        path.display()
196    ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201    spec: &ModuleSpec,
202    roster: &ChildRoster,
203) -> Result<
204    (
205        Command,
206        Option<PrivacyExec>,
207        subc_os::privacy_identity::ExecAcknowledgement,
208    ),
209    SuperviseError,
210> {
211    let failure = |cause: String| {
212        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213        SuperviseError::Spawn {
214            program: spec.program.clone(),
215            source: io::Error::other(cause),
216            cgroup_path: None,
217        }
218    };
219    let trampoline = roster.privacy_trampoline().map_err(failure)?;
220    // Resolve PATH with the same environment the Command will receive. For
221    // scripts retain the existing orphan-identity rule: the kernel chooses
222    // the interpreter, and its observed image is the one recorded. Do not
223    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
224    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225        let path = spec
226            .env
227            .iter()
228            .find(|(key, _)| key == "PATH")
229            .map(|(_, value)| std::ffi::OsString::from(value))
230            .or_else(|| std::env::var_os("PATH"))
231            .unwrap_or_else(|| "/usr/bin:/bin".into());
232        std::env::split_paths(&path)
233            .map(|dir| dir.join(&spec.program))
234            .find(|path| path.is_file())
235            .unwrap_or_else(|| spec.program.clone())
236    } else {
237        spec.program.clone()
238    };
239    let expected = subc_os::file_identity(&program);
240    let trampoline_image = subc_os::file_identity(&trampoline);
241    let script = {
242        use std::io::Read;
243        let mut prefix = [0u8; 2];
244        std::fs::File::open(&program)
245            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246    };
247    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248        return Err(failure(
249            "privacy identity module executable is missing or is the trampoline itself".to_string(),
250        ));
251    }
252    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253        .map_err(|error| failure(error.to_string()))?;
254    let reader =
255        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256    let mut command = Command::new(&trampoline);
257    command
258        .arg("__disclaim-exec")
259        .arg(ack.fd().to_string())
260        .arg(&program);
261    ack.install(command.as_std_mut());
262    Ok((
263        command,
264        Some(PrivacyExec {
265            reader,
266            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267            expected,
268            trampoline: trampoline_image,
269            script,
270            module_id: spec.module_id.clone(),
271        }),
272        ack,
273    ))
274}
275
276struct SupervisedChild {
277    child: Child,
278    #[cfg(target_os = "macos")]
279    privacy_exec: Option<PrivacyExec>,
280    /// Set once this launch's exec acknowledgement confirms the module image.
281    /// On macOS the pid first runs the `ck-subc` launch trampoline (see
282    /// `subc_os::privacy_identity`), which then replaces itself with the
283    /// module. The supervisor owns and can kill that pid from spawn, but
284    /// status readers report it only after this latch is set, so nothing
285    /// reports the trampoline's image as the module's.
286    #[cfg(target_os = "macos")]
287    report_ready: Arc<OnceLock<()>>,
288    /// Refusal before the module image was accepted, retained for terminal records.
289    spawn_failure: Option<String>,
290    /// The protocol this process was launched with. A reload can store a new
291    /// launch spec with a different protocol, but that takes effect only at the
292    /// next spawn, so this process keeps being handled by the protocol it
293    /// actually speaks.
294    protocol: ModuleProtocol,
295    /// This process's cgroup name: a bounded module/slot label followed by a
296    /// spawn suffix unique to this process (when cgroup placement is on). A
297    /// retired process in a slot may still be draining when a later one is
298    /// spawned into that slot, so the suffix keeps the later process out of
299    /// the retired one's cgroup, which is the domain a kill applies to.
300    #[cfg(target_os = "linux")]
301    module_id: String,
302    #[cfg(target_os = "linux")]
303    cgroup_placement: Option<subc_cgroup::Placement>,
304    /// The job that contains this child and every process it spawns (issue #109).
305    ///
306    /// Dropping this handle is what reaps a surviving tree when no supervisor
307    /// code runs — a daemon crash — because the job carries
308    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
309    ///
310    /// That limit is not crash-only, and the difference is worth knowing: a
311    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
312    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
313    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
314    /// module at once. Before this change they survived that, saw EOF on the
315    /// control socket, and ran their own teardown; Unix keeps that path
316    /// deliberately, so a module can seal a WAL or close a capture rather than
317    /// be killed mid-write. So this trades graceful teardown on every Windows
318    /// daemon stop for containment on a crash, which is the right way round
319    /// today: orphaned GPU workers are a reported, recurring problem, and the
320    /// modules that write most heavily do not run on Windows.
321    ///
322    /// The fix is a real Windows stop path — the daemon draining before it
323    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
324    /// reaches only what the drain left behind, which is what it should reach.
325    #[cfg(windows)]
326    job: Option<subc_jobobject::JobObject>,
327    stdout_pump: Option<JoinHandle<()>>,
328    stderr_pump: Option<StderrPump>,
329    stderr_ring: Arc<Mutex<StderrRing>>,
330    spawned_at_ms: u64,
331    spawned_from: PathBuf,
332    spawned_file_identity: Option<SpawnedFileIdentity>,
333    process_start_time: Option<u64>,
334    process_identity: Option<ProcessIdentity>,
335    pid: u32,
336    /// This process's entry in the daemon's child roster, released when the
337    /// process is reaped or this handle is dropped.
338    roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342    fn id(&self) -> Option<u32> {
343        Some(self.pid)
344    }
345
346    fn process_identity(&self) -> Option<ProcessIdentity> {
347        self.process_identity
348    }
349
350    async fn wait(&mut self) -> io::Result<ExitStatus> {
351        #[cfg(target_os = "macos")]
352        self.confirm_privacy_exec().await;
353        // The roster entry is NOT released here. A daemon shutdown waits for the
354        // roster to empty and then exits the process, so releasing at the reap
355        // let it exit before the exit handler wrote this child's terminal record
356        // (the stderr drain and snapshot update sit in between), and the
357        // shutdown's own `daemon_shutdown` record was intermittently lost. The
358        // caller releases it after recording the exit (`release_roster`), and
359        // dropping the handle releases it too.
360        let result = self.child.wait().await;
361        #[cfg(target_os = "linux")]
362        if result.is_ok() {
363            if let Some(placement) = self.cgroup_placement.as_ref() {
364                cleanup_reaped_cgroup(placement, &self.module_id).await;
365                // Keep ownership while awaiting kernel population changes: a
366                // drain timeout may cancel this wait and then escalate/reap.
367                self.cgroup_placement = None;
368            }
369        }
370        result
371    }
372
373    #[cfg(target_os = "macos")]
374    async fn confirm_privacy_exec(&mut self) {
375        let Some(pending) = &mut self.privacy_exec else {
376            return;
377        };
378        let result = tokio::time::timeout_at(pending.deadline, async {
379            let mut record = Vec::new();
380            loop {
381                let mut ready = pending.reader.readable().await?;
382                let read = ready.try_io(|reader| {
383                    use std::io::Read;
384                    let mut reader = reader.get_ref();
385                    let mut buffer = [0u8; 256];
386                    reader.read(&mut buffer).map(|count| (count, buffer))
387                });
388                match read {
389                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390                    Ok(Ok((count, buffer))) => {
391                        if record.len() + count > 1024 {
392                            return Err(io::Error::other(
393                                "privacy exec refusal record is too long",
394                            ));
395                        }
396                        record.extend_from_slice(&buffer[..count]);
397                    }
398                    Ok(Err(error)) => return Err(error),
399                    Err(_) => continue,
400                }
401            }
402        })
403        .await;
404        // Keep the reader in self across await: select cancellation must not
405        // discard the handshake or reset its original five-second deadline.
406        let pending = self.privacy_exec.as_ref().expect("pending exec");
407        let cause = match result {
408            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409            Ok(Err(error)) => Some(format!(
410                "privacy identity exec acknowledgement failed: {error}"
411            )),
412            Ok(Ok(record)) if !record.is_empty() => Some(
413                std::str::from_utf8(&record)
414                    .ok()
415                    .and_then(|record| {
416                        record
417                            .trim()
418                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419                    })
420                    .filter(|cause| !cause.is_empty())
421                    .unwrap_or("invalid privacy exec refusal record")
422                    .to_string(),
423            ),
424            Ok(Ok(_)) => match self.child.try_wait() {
425                // Empty EOF is the exec acknowledgement. A real module may exit
426                // immediately, including with a reserved trampoline status; no
427                // image is admitted, and its ordinary exit contract stays intact.
428                Ok(Some(_status)) => None,
429                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430                Ok(None) => {
431                    let image = observe_spawned_image(self.pid);
432                    if let Some(image) = image.filter(|image| {
433                        image.executable.is_some()
434                            && image.executable != pending.trampoline
435                            && (image.executable == pending.expected || pending.script)
436                    }) {
437                        if let Some(guard) = &self.roster_guard {
438                            guard.confirm_executable(image);
439                        }
440                        let _ = self.report_ready.set(());
441                        info!(module_id = %pending.module_id, pid = self.pid,
442                            "module spawned with own privacy identity (responsibility disclaimed)");
443                        None
444                    } else if image.is_none()
445                        || image.is_some_and(|image| image.executable.is_none())
446                    {
447                        // A process can exit between try_wait and the kernel
448                        // image read. Empty EOF already acknowledged exec, so
449                        // preserve that module's ordinary exit rather than
450                        // mislabel a disappearing image as trampoline refusal.
451                        // Keep pending in self across await so cancellation does
452                        // not discard validation or reset its original deadline.
453                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454                            Ok(Ok(_status)) => None,
455                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457                        }
458                    } else {
459                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460                    }
461                }
462            },
463        };
464        let pending = self.privacy_exec.take().expect("pending exec");
465        if let Some(cause) = cause {
466            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467            self.spawn_failure = Some(cause);
468            // No image is admitted on failure. Reach the entire fresh process
469            // group, including a module which spawned a helper before refusal.
470            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472            }
473            let _ = self.child.start_kill();
474        }
475    }
476
477    /// Releases this child's daemon-shutdown roster entry once its exit has
478    /// been recorded. The pid is already reaped and free for reuse, so the
479    /// entry must not outlive the record any longer than that.
480    fn release_roster(&mut self) {
481        self.roster_guard = None;
482    }
483
484    /// Kill the child and, where containment is available, its process tree.
485    ///
486    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
487    /// helper process leaked the helper — the Synapse embedding module's CUDA
488    /// worker holds the GPU allocation, so the leak cost VRAM until the next
489    /// restart of something else. Terminating the job reaches grandchildren that
490    /// a tree walk cannot, including one whose parent has already exited and
491    /// been reparented away.
492    ///
493    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
494    /// direct-child kill still decides the outcome, so containment can never
495    /// change whether a module is reported as stopped.
496    fn start_kill(&mut self) -> io::Result<()> {
497        #[cfg(windows)]
498        if let Some(job) = &self.job {
499            if let Err(error) = job.terminate() {
500                debug!(
501                    error = %error,
502                    "job termination failed; the direct-child kill still owns the outcome"
503                );
504            }
505        }
506        #[cfg(target_os = "linux")]
507        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508        self.child.start_kill()
509    }
510
511    async fn drain_stderr(&mut self, module_id: &str) {
512        if let Some(mut pump) = self.stdout_pump.take() {
513            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514                Ok(Ok(())) => {}
515                Ok(Err(error)) => {
516                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517                }
518                Err(_) => {
519                    pump.abort();
520                    warn!(
521                        module_id,
522                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523                        "stdout pump did not drain before restart; stopped it before the next process"
524                    );
525                }
526            }
527        }
528
529        let Some(pump) = self.stderr_pump.take() else {
530            return;
531        };
532        settle_stderr_pump(
533            module_id,
534            &self.stderr_ring,
535            pump,
536            STDERR_PUMP_DRAIN_TIMEOUT,
537        )
538        .await;
539    }
540}
541
542/// The reader task for one process's stderr, with the ring generation its
543/// lines are attributed to.
544struct StderrPump {
545    task: JoinHandle<()>,
546    generation: u64,
547}
548
549/// Retire an exited process's stderr reader and wait up to `bound` for it to
550/// reach EOF. A reader still running at the bound is detached, not stopped: it
551/// keeps filling the exited process's section of the ring until its pipe
552/// closes, and the tail reads `Incomplete` until then. See
553/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
554async fn settle_stderr_pump(
555    module_id: &str,
556    ring: &Arc<Mutex<StderrRing>>,
557    pump: StderrPump,
558    bound: Duration,
559) {
560    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561    let StderrPump {
562        mut task,
563        generation,
564    } = pump;
565    lock().retire_pump(generation);
566    match timeout(bound, &mut task).await {
567        Ok(Ok(())) => {}
568        Ok(Err(err)) => {
569            let mut ring = lock();
570            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571            ring.finish_pump(generation);
572            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573        }
574        Err(_) => {
575            // Dropping the handle detaches the task; it ends at EOF on its pipe.
576            drop(task);
577            lock().mark_pump_late(
578                generation,
579                format!(
580                    "stderr of the exited process had not reached EOF {bound:?} after it was \
581                     retired (a descendant may still hold the pipe open); lines it still \
582                     writes are kept in that process's section"
583                ),
584            );
585            warn!(
586                module_id,
587                waited = ?bound,
588                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589            );
590        }
591    }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596    EVENTS.get_or_init(|| {
597        let (sender, _receiver) = watch::channel(0);
598        sender
599    })
600}
601
602pub(crate) fn notify_registration_release() {
603    let events = registration_release_events();
604    let next_generation = (*events.borrow()).wrapping_add(1);
605    events.send_replace(next_generation);
606}
607
608/// How to launch one singleton module process.
609#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611    pub module_id: String,
612    pub program: PathBuf,
613    pub args: Vec<String>,
614    pub env: Vec<(String, String)>,
615    /// When true this is a reserved module: each spawn gets a fresh one-time launch
616    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
617    /// process can register this module_id (a security-boundary module like the
618    /// credential vault must not be impersonable while it is down/restarting).
619    pub reserved: bool,
620    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
621    /// Prefixes come from daemon config and must end in `:` before they reach the
622    /// supervisor; the owner module's current spawn nonce authorizes claims under
623    /// each prefix.
624    pub reserved_prefixes: Vec<String>,
625    /// The wire protocol this module speaks, as DECLARED in daemon config.
626    ///
627    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
628    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
629    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
630    /// and NO launch nonce, and a clean exit the daemon did not request is
631    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
632    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
633    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
634    /// because a process ignores an environment variable it does not read.
635    ///
636    /// The argument is the part that cannot be "harmless to a process that
637    /// ignores it": a stock binary exits on an unknown flag before it listens
638    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
639    /// first conformance run against this mode found it. The nonce is withheld
640    /// because a process that will never present it gains nothing from holding
641    /// it, and a secret in the environment of a process that does not need it is
642    /// a leak surface for no benefit.
643    pub protocol: ModuleProtocol,
644    /// Whether two processes of this module may run at once, which is what a
645    /// blue/green swap does for the length of its overlap. Declared in daemon
646    /// config because the daemon must be able to answer it while the module is
647    /// down, and so a module cannot talk itself into it after registering.
648    pub overlap: ModuleOverlap,
649}
650
651/// Whether a module tolerates a second process of itself running alongside.
652///
653/// Most modules are single-writer on their store (a WAL, a capture log, a
654/// resident index behind a writer barrier), and two processes on one store
655/// corrupt it. So a swap, which overlaps the old and new process by design,
656/// is refused unless the module's config opts in with `overlap: "safe"`.
657#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659    /// Never run two processes of this module at once. The default.
660    #[default]
661    Exclusive,
662    /// The module has said a second process of itself is harmless for the
663    /// length of a swap.
664    ///
665    /// Declare it only if a second instance can run for a few seconds without
666    /// touching ANY single-writer store: every database, WAL, index, projector
667    /// and scheduled job the module owns. A lease on part of that state is not
668    /// enough. broca's session lease guards WAL appends while its run index, its
669    /// store projector and its archive fold timer (which unlinks live WAL files)
670    /// stay single-writer, so broca is exclusive despite holding a lease. The
671    /// refusal only fires after this has been decided, so the decision is the
672    /// check.
673    Safe,
674}
675
676impl ModuleOverlap {
677    pub fn as_str(self) -> &'static str {
678        match self {
679            Self::Exclusive => "exclusive",
680            Self::Safe => "safe",
681        }
682    }
683}
684
685/// Environment variable telling a spawned module which case it was started
686/// for, before it sends HELLO. Only a swap candidate carries it, as
687/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
688///
689/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
690/// longer because nobody waits on it, while a plain restart must flip ready
691/// quickly because callers see `module_warming` until it does. Absence means
692/// plain restart, the safe reading. The daemon trusts nothing about it; the
693/// candidate is proven by its launch nonce at HELLO.
694pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
696pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697/// How long a swap waits for its candidate to register and declare itself
698/// ready when the operator does not say. A module warming as a swap candidate
699/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
700/// daemon allows that plus time to start the process and send HELLO.
701pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703/// Bounded restart policy for crash exits.
704///
705/// `max_restarts` is the number of replacement processes allowed after the
706/// initial spawn WITHIN `window`. After that many crash restarts inside one
707/// window the module enters [`ModuleState::Failed`] and the supervisor stops
708/// the crash loop.
709///
710/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
711/// and that only survived because crashes were rare: a module that crashed
712/// three times across a week was disabled forever by crashes that had nothing
713/// to do with each other. That stopped being survivable once modules began
714/// exiting non-zero whenever the daemon's connection to them drops, because
715/// then every daemon-side connection drop spends a unit of the same budget and
716/// one flappy hour permanently stops a healthy module. Restarts older than
717/// `window` release their slot, so a module that crashed twice yesterday has a
718/// full budget today, while a genuine crash loop -- which is fast by
719/// definition -- still reaches the cap and stops.
720#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722    pub max_restarts: u32,
723    /// Base delay before a crash replacement. The actual delay escalates with
724    /// the number of recent crash replacements and is capped by `max_backoff`.
725    pub backoff: Duration,
726    /// Maximum delay before a crash replacement.
727    pub max_backoff: Duration,
728    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
729    /// budget effectively infinite (nothing is ever in-window), which is why
730    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
731    pub window: Duration,
732}
733
734impl RestartPolicy {
735    /// A policy with the default crash window. Callers that care about the
736    /// window say so with [`Self::with_window`]; the ones that do not are
737    /// asking for the standard rate limit, not for no limit.
738    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739        Self {
740            max_restarts,
741            backoff,
742            max_backoff: DEFAULT_MAX_BACKOFF,
743            window: DEFAULT_RESTART_WINDOW,
744        }
745    }
746
747    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748        self.max_backoff = max_backoff;
749        self
750    }
751
752    pub fn with_window(mut self, window: Duration) -> Self {
753        self.window = window;
754        self
755    }
756
757    /// Calculate the capped exponential delay for the next crash replacement.
758    /// `restart_in_window` is zero for the first replacement after an operator
759    /// action (restart, reload, re-enable) cleared the crash ring, or after all
760    /// older crash replacements have aged out of the window.
761    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762        if self.backoff.is_zero() || self.max_backoff.is_zero() {
763            return Duration::ZERO;
764        }
765
766        let mut delay = self.backoff;
767        for _ in 0..restart_in_window {
768            if delay >= self.max_backoff {
769                return self.max_backoff;
770            }
771            delay = delay
772                .checked_mul(10)
773                .unwrap_or(self.max_backoff)
774                .min(self.max_backoff);
775        }
776        delay.min(self.max_backoff)
777    }
778
779    /// The one sentence that explains a budget-exhausted stop, used for both the
780    /// log line and the terminal record so the two cannot drift. It names the
781    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
782    /// exactly what this budget is not.
783    fn budget_exhausted_detail(&self) -> String {
784        format!(
785            "crash budget exhausted: max_restarts={} within window_secs={}",
786            self.max_restarts,
787            self.window.as_secs()
788        )
789    }
790}
791
792impl Default for RestartPolicy {
793    fn default() -> Self {
794        Self {
795            max_restarts: DEFAULT_MAX_RESTARTS,
796            backoff: DEFAULT_BACKOFF,
797            max_backoff: DEFAULT_MAX_BACKOFF,
798            window: DEFAULT_RESTART_WINDOW,
799        }
800    }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805    restart_in_window: u32,
806    delay: Duration,
807}
808
809/// Whether the daemon itself will bring this module back after the exit being
810/// handled: it is enabled AND its in-window crash restarts are below the cap.
811///
812/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
813/// the window are dropped here rather than by a timer, so the count is right
814/// the moment somebody asks and no bookkeeping runs for idle modules.
815fn daemon_will_restart(
816    state: &mut SupervisorSnapshot,
817    policy: &RestartPolicy,
818    now: Instant,
819) -> bool {
820    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830    Report,
831    Restart,
832    Alert,
833}
834
835impl fmt::Display for HealthAction {
836    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837        f.write_str(match self {
838            Self::Report => "report",
839            Self::Restart => "restart",
840            Self::Alert => "alert",
841        })
842    }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847    /// Optional loopback HTTP endpoint for a managed non-wire process.
848    /// Changing it applies live on rescan; the process protocol changes only
849    /// at its next spawn.
850    pub http: Option<String>,
851    pub cadence: Duration,
852    pub deadline: Duration,
853    pub failure_threshold: u32,
854    pub on_degraded: HealthAction,
855    pub on_failing: HealthAction,
856    pub critical: bool,
857}
858
859impl Default for HealthConfig {
860    fn default() -> Self {
861        Self {
862            http: None,
863            cadence: DEFAULT_HEALTH_CADENCE,
864            deadline: DEFAULT_HEALTH_DEADLINE,
865            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866            on_degraded: HealthAction::Report,
867            on_failing: HealthAction::Report,
868            critical: false,
869        }
870    }
871}
872
873/// The supervisor's view of one module's health, relayed to clients over
874/// channel-0 and rendered by `ck health`.
875///
876/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
877/// stated here rather than only at the wire type a consumer reads. A reader can
878/// look up what `None` means; only a writer can silently change it, and the
879/// writer has no reason to go looking at a downstream contract before editing.
880///
881/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
882/// back to `None` on re-registration precisely so a respawned module does not
883/// carry its predecessor's timestamp — so an old value and an absent one call for
884/// opposite readings, and anything that defaulted this to a number would make a
885/// never-probed module indistinguishable from one probed at the epoch.
886///
887/// `detail` and `metrics` are `None` when the module published none on this
888/// probe, which does not mean it reported nothing wrong — it is also the shape
889/// when the probe never reached it. `last_probe_ms` is what separates those.
890#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892    pub status: SupervisorHealthStatus,
893    pub last_probe_ms: Option<u64>,
894    pub detail: Option<String>,
895    pub metrics: Option<Value>,
896    pub consecutive_failures: u32,
897    /// Number of replies received after a recurring health probe's deadline.
898    /// Unlike a timeout, every increment proves the module was alive.
899    pub late_answer_count: u64,
900    /// End-to-end latency of the newest late reply, measured from probe start.
901    pub last_late_answer_latency_ms: Option<u64>,
902    pub last_action: Option<String>,
903    /// Set together with `last_action`; the pair moves as one, and both being
904    /// absent means no escalation has ever been taken rather than that the last
905    /// one succeeded.
906    pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910    fn default() -> Self {
911        Self {
912            status: SupervisorHealthStatus::Unknown,
913            last_probe_ms: None,
914            detail: None,
915            metrics: None,
916            consecutive_failures: 0,
917            late_answer_count: 0,
918            last_late_answer_latency_ms: None,
919            last_action: None,
920            last_action_ms: None,
921        }
922    }
923}
924
925/// Typed lifecycle state for a supervised module.
926#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928    Starting,
929    Running,
930    Unresponsive,
931    Restarting,
932    Draining,
933    Stopped,
934    Failed,
935    Disabled,
936}
937
938impl fmt::Display for ModuleState {
939    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940        f.write_str(match self {
941            Self::Starting => "starting",
942            Self::Running => "running",
943            Self::Unresponsive => "unresponsive",
944            Self::Restarting => "restarting",
945            Self::Draining => "draining",
946            Self::Stopped => "stopped",
947            Self::Failed => "failed",
948            Self::Disabled => "disabled",
949        })
950    }
951}
952
953/// Supervisor classification of a child-process exit.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956    Clean,
957    Crash,
958    DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962    fn from(kind: ExitKind) -> Self {
963        match kind {
964            ExitKind::Clean => Self::Clean,
965            ExitKind::Crash => Self::Crash,
966            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967        }
968    }
969}
970
971/// Exact process identity retained when a supervised module registers its
972/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
973#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975    pub(crate) pid: u32,
976    pub(crate) start_time: u64,
977}
978
979/// Last observed child exit, if any.
980#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982    pub kind: ExitKind,
983    pub code: Option<i32>,
984    pub signal: Option<i32>,
985    pub at_ms: u64,
986}
987
988/// Point-in-time module status answerable by subc without forwarding to the
989/// module process.
990#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992    pub module_id: String,
993    pub state: ModuleState,
994    pub enabled: bool,
995    pub process_alive: bool,
996    pub registration_active: bool,
997    /// The module's declared wire protocol, carried beside `live` because it is
998    /// what makes `live` readable: the two fields answer one question together.
999    /// While a process is alive this is its launch declaration, not a later
1000    /// pending-reload edit. When down it is the configured next launch protocol.
1001    pub protocol: ModuleProtocol,
1002    /// Whether the module is serving, under the strongest definition the daemon
1003    /// can assert for its protocol.
1004    ///
1005    /// A subc module must also be REGISTERED: its process being alive says
1006    /// nothing about whether it can take a request. A `protocol: "none"` module
1007    /// never registers, so that term is dropped and this falls back to "enabled,
1008    /// running, and the process the daemon launched is alive" -- which is all
1009    /// the daemon observes about a process that speaks no subc wire. It stays a
1010    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
1011    /// rather than printing it bare.
1012    pub live: bool,
1013    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
1014    /// restarts have already released their slot, so this count can go down
1015    /// without anybody touching the module.
1016    pub restart_count: u32,
1017    /// Replacement processes spawned over this module's entire supervisor lifetime;
1018    /// unlike `restart_count`, this value is never reset by an operator action
1019    /// and never falls out of a window.
1020    pub lifetime_restarts: u32,
1021    pub spawn_generation: u64,
1022    /// The budget `restart_count` is spent against. Carried alongside the count
1023    /// because the count alone does not say how close the module is to being
1024    /// disabled, and reporting one without the other is what makes an
1025    /// about-to-be-retired module look ordinary.
1026    pub max_restarts: u32,
1027    /// The span `restart_count` is counted over. Carried with the pair above for
1028    /// the same reason they are carried together: "2 of 3" means one thing for a
1029    /// ten-minute window and something else entirely for a lifetime.
1030    pub restart_window: Duration,
1031    /// Effective drain and restart timing policy used by this running module.
1032    /// These values are carried together with the restart budget so status
1033    /// readers can compare configured intent with what the supervisor applied.
1034    pub drain_timeout: Duration,
1035    pub restart_backoff: Duration,
1036    pub restart_max_backoff: Duration,
1037    /// The module's process. On macOS this stays absent while the `ck-subc`
1038    /// launch trampoline is still running in that pid, and appears once the
1039    /// exec acknowledgement confirms the module image has replaced it. Launch
1040    /// time and the supervisor's own hold on the process are unaffected.
1041    pub pid: Option<u32>,
1042    pub spawned_at_ms: Option<u64>,
1043    pub spawned_from: Option<PathBuf>,
1044    pub process_start_time: Option<u64>,
1045    pub last_exit: Option<ExitReport>,
1046    pub health: ModuleHealthStatus,
1047}
1048
1049#[cfg(test)]
1050#[derive(Debug, Clone, Copy, PartialEq)]
1051struct ActorSelectCheckpoint {
1052    generation: u64,
1053    turn: u64,
1054    registered_connection: Option<ConnectionId>,
1055    next_probe_at: Option<Instant>,
1056    wake_after: Option<Duration>,
1057}
1058
1059#[derive(Debug, Clone, PartialEq)]
1060struct SupervisorSnapshot {
1061    /// Counts running-child loop turns, not executor polls of a parked wait.
1062    #[cfg(test)]
1063    actor_turns: u64,
1064    /// Running select after exec confirmation and with no unconsumed registry
1065    /// event. Recording a spawn or starting a loop turn is not this barrier.
1066    #[cfg(test)]
1067    actor_select: Option<ActorSelectCheckpoint>,
1068    state: ModuleState,
1069    enabled: bool,
1070    process_alive: bool,
1071    spawned_protocol: Option<ModuleProtocol>,
1072    spawn_failure: Option<String>,
1073    /// When each crash restart was spent, oldest first. This IS the crash
1074    /// budget: its in-window length is the count an operator sees and the count
1075    /// the restart decision is made against, so there is no second counter that
1076    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1077    /// operator actions that used to zero the old lifetime counter.
1078    crash_restarts: VecDeque<Instant>,
1079    lifetime_restarts: u32,
1080    /// Successful child spawns in this daemon incarnation.
1081    ///
1082    /// `lifetime_restarts` was considered and rejected: it starts at zero
1083    /// (line 640), successful initial/operator spawns in `set_running` do not
1084    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1085    /// increments before a successful replacement exists (lines 604, 3846,
1086    /// and 3921), so a failed spawn can consume it. This counter moves only
1087    /// when a live PID is accepted below.
1088    spawn_generation: u64,
1089    pid: Option<u32>,
1090    #[cfg(target_os = "macos")]
1091    report_ready: Option<Arc<OnceLock<()>>>,
1092    /// Last reaped child, retained after current process facts are cleared.
1093    reaped_pid: Option<u32>,
1094    /// Whether the command-serving supervision loop has a scheduled respawn.
1095    respawn_pending: bool,
1096    /// A second restart is waiting for the replacement already scheduled.
1097    coalesced_restart_pending: bool,
1098    spawned_at_ms: Option<u64>,
1099    spawned_from: Option<PathBuf>,
1100    spawned_file_identity: Option<SpawnedFileIdentity>,
1101    process_start_time: Option<u64>,
1102    deliberate_severance: Option<ProcessIdentity>,
1103    last_exit: Option<ExitReport>,
1104    /// Diagnostic attached to the next drain's terminal record, if any.
1105    drain_disposition_detail: Option<String>,
1106    health: ModuleHealthStatus,
1107    /// Whether the current process was started as a swap candidate and so
1108    /// lives in the module's alternate cgroup. The next swap's candidate takes
1109    /// the other one, so the two processes of a swap never share a cgroup. A
1110    /// plain spawn always uses the primary cgroup.
1111    in_alternate_slot: bool,
1112    /// Whether the current `Draining` state ends in a replacement process
1113    /// (restart, reload, health restart) rather than a stop. Only meaningful
1114    /// while `state` is `Draining`; every entry into that state rewrites it.
1115    /// It is what lets route.open answer the retryable `module_reloading` to a
1116    /// consumer that reaches a still-registered process mid-restart, instead of
1117    /// the `supervisor_not_live` a stop or disable deserves.
1118    draining_to_replace: bool,
1119    /// Whether a configuration update has been applied since the current
1120    /// process was spawned, so that process runs an older spec than the one
1121    /// the supervisor now holds. A queued restart is only coalesced into a
1122    /// fresher process when this is false: a restart requested to pick up a
1123    /// new configuration must not be satisfied by a process that predates it.
1124    configuration_updated_since_spawn: bool,
1125}
1126
1127impl SupervisorSnapshot {
1128    /// The pid that status, provenance and resource readings may report. While
1129    /// the launch trampoline still runs in the pid, reading its executable or
1130    /// resource use would describe `ck-subc`, not the module, so none is
1131    /// reported until the exec acknowledgement confirms the module image.
1132    fn reported_pid(&self) -> Option<u32> {
1133        #[cfg(target_os = "macos")]
1134        if self
1135            .report_ready
1136            .as_ref()
1137            .is_some_and(|ready| ready.get().is_none())
1138        {
1139            return None;
1140        }
1141        self.pid
1142    }
1143
1144    fn starting() -> Self {
1145        Self::new(ModuleState::Starting, true)
1146    }
1147
1148    fn disabled() -> Self {
1149        Self::new(ModuleState::Disabled, false)
1150    }
1151
1152    fn failed() -> Self {
1153        Self::new(ModuleState::Failed, true)
1154    }
1155
1156    /// Crash restarts still inside `window`, having dropped the ones that are
1157    /// not. Pruning on read is what makes the budget a rate: an instant older
1158    /// than the window stops holding a slot the moment anybody counts.
1159    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1160        while let Some(oldest) = self.crash_restarts.front() {
1161            if now.duration_since(*oldest) > window {
1162                self.crash_restarts.pop_front();
1163            } else {
1164                break;
1165            }
1166        }
1167        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1168    }
1169
1170    /// Spend one unit of the crash budget and record the restart in the ledger.
1171    ///
1172    /// The ring is bounded by the cap because more than `max_restarts` in-window
1173    /// instants can never be reached (the caller refuses the restart first), so
1174    /// anything beyond that is an unbounded queue waiting to happen.
1175    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1176        self.crash_restarts.push_back(now);
1177        while self.crash_restarts.len() > policy.max_restarts as usize {
1178            self.crash_restarts.pop_front();
1179        }
1180        self.lifetime_restarts += 1;
1181    }
1182
1183    /// Reserve one crash-restart slot and calculate the delay before respawning.
1184    /// The count is captured before recording this restart, so the first retry
1185    /// uses the base delay and each later in-window retry escalates once.
1186    fn next_crash_restart(
1187        &mut self,
1188        policy: &RestartPolicy,
1189        now: Instant,
1190    ) -> Option<CrashRestartSchedule> {
1191        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1192        if restart_in_window >= policy.max_restarts {
1193            return None;
1194        }
1195        self.record_crash_restart(policy, now);
1196        Some(CrashRestartSchedule {
1197            restart_in_window,
1198            delay: policy.delay_for_restart(restart_in_window),
1199        })
1200    }
1201
1202    /// Give the module its full budget back, as an operator restart, reload, or
1203    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1204    /// ledger of what actually happened, and an operator action does not unmake
1205    /// the crashes.
1206    fn clear_crash_restarts(&mut self) {
1207        self.crash_restarts.clear();
1208    }
1209
1210    fn new(state: ModuleState, enabled: bool) -> Self {
1211        Self {
1212            #[cfg(test)]
1213            actor_turns: 0,
1214            #[cfg(test)]
1215            actor_select: None,
1216            state,
1217            enabled,
1218            process_alive: false,
1219            spawned_protocol: None,
1220            spawn_failure: None,
1221            crash_restarts: VecDeque::new(),
1222            lifetime_restarts: 0,
1223            spawn_generation: 0,
1224            pid: None,
1225            #[cfg(target_os = "macos")]
1226            report_ready: None,
1227            reaped_pid: None,
1228            respawn_pending: false,
1229            coalesced_restart_pending: false,
1230            spawned_at_ms: None,
1231            spawned_from: None,
1232            spawned_file_identity: None,
1233            process_start_time: None,
1234            deliberate_severance: None,
1235            last_exit: None,
1236            drain_disposition_detail: None,
1237            health: ModuleHealthStatus::default(),
1238            in_alternate_slot: false,
1239            draining_to_replace: false,
1240            configuration_updated_since_spawn: false,
1241        }
1242    }
1243}
1244
1245type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1246
1247type SpawnSubscriberKey = (ConnectionId, u64);
1248
1249#[derive(Debug)]
1250struct SpawnSubscriber {
1251    version: u8,
1252    frames: mpsc::Sender<Frame>,
1253    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1254    /// from which event. The full frame channel cannot carry that news, so it
1255    /// travels beside it; see `SpawnEventFeed::subscribe`.
1256    lagged: Option<oneshot::Sender<SpawnCursor>>,
1257}
1258
1259#[derive(Debug)]
1260struct SpawnEventState {
1261    daemon_incarnation: String,
1262    seq: u64,
1263    capacity: usize,
1264    live: HashMap<String, LiveSpawn>,
1265    generations: HashMap<String, u64>,
1266    events: VecDeque<SpawnEvent>,
1267    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1268}
1269
1270impl Default for SpawnEventState {
1271    fn default() -> Self {
1272        Self {
1273            daemon_incarnation: "unconfigured".to_string(),
1274            seq: 0,
1275            capacity: SPAWN_EVENT_RING_CAPACITY,
1276            live: HashMap::new(),
1277            generations: HashMap::new(),
1278            events: VecDeque::new(),
1279            subscribers: HashMap::new(),
1280        }
1281    }
1282}
1283
1284#[derive(Debug, Clone, Default)]
1285struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1286
1287#[derive(Debug, Clone, PartialEq, Eq)]
1288pub(crate) enum SpawnSubscribeRefusal {
1289    ForeignIncarnation { current: String },
1290    TooOld { oldest: SpawnCursor },
1291    Frame(String),
1292}
1293
1294impl SpawnEventFeed {
1295    fn configure_incarnation(&self, daemon_incarnation: String) {
1296        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1297        state.daemon_incarnation = daemon_incarnation;
1298        state.seq = 0;
1299        state.live.clear();
1300        state.generations.clear();
1301        state.events.clear();
1302        state.subscribers.clear();
1303    }
1304
1305    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1306        SpawnCursor {
1307            daemon_incarnation: state.daemon_incarnation.clone(),
1308            seq: state.seq,
1309        }
1310    }
1311
1312    fn snapshot(&self) -> SpawnSnapshot {
1313        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1314        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1315        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1316        SpawnSnapshot {
1317            cursor: Self::cursor(&state),
1318            ring_bound: state.capacity as u64,
1319            live,
1320        }
1321    }
1322
1323    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1324        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1325        let generation = state
1326            .generations
1327            .get(module_id)
1328            .copied()
1329            .unwrap_or(0)
1330            .checked_add(1)
1331            .expect("spawn generation exhausted");
1332        state.generations.insert(module_id.to_string(), generation);
1333        let live = LiveSpawn {
1334            module_id: module_id.to_string(),
1335            spawn_generation: generation,
1336            pid,
1337            spawned_at_ms,
1338        };
1339        state.live.insert(module_id.to_string(), live);
1340        Self::emit_locked(
1341            &mut state,
1342            SpawnEventKind::Spawned,
1343            module_id.to_string(),
1344            generation,
1345            pid,
1346            None,
1347            None,
1348        );
1349        generation
1350    }
1351
1352    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1353        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1354        let Some(live) = state.live.remove(module_id) else {
1355            warn!(
1356                module_id,
1357                "terminal record had no live spawn event identity"
1358            );
1359            return;
1360        };
1361        Self::emit_locked(
1362            &mut state,
1363            SpawnEventKind::Exited,
1364            module_id.to_string(),
1365            live.spawn_generation,
1366            live.pid,
1367            exit_code,
1368            exit_signal,
1369        );
1370    }
1371
1372    /// Report the exit of a process that a swap has already replaced.
1373    ///
1374    /// `emit_exited` removes the module's live entry, which after a swap's
1375    /// cutover describes the promoted candidate, not the old process now
1376    /// exiting. This emits the old generation's exit and leaves the live entry
1377    /// alone unless it still names that generation.
1378    fn emit_superseded_exited(
1379        &self,
1380        module_id: &str,
1381        spawn_generation: u64,
1382        pid: u32,
1383        exit_code: Option<i32>,
1384        exit_signal: Option<i32>,
1385    ) {
1386        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1387        if state
1388            .live
1389            .get(module_id)
1390            .is_some_and(|live| live.spawn_generation == spawn_generation)
1391        {
1392            state.live.remove(module_id);
1393        }
1394        Self::emit_locked(
1395            &mut state,
1396            SpawnEventKind::Exited,
1397            module_id.to_string(),
1398            spawn_generation,
1399            pid,
1400            exit_code,
1401            exit_signal,
1402        );
1403    }
1404
1405    #[allow(clippy::too_many_arguments)]
1406    fn emit_locked(
1407        state: &mut SpawnEventState,
1408        kind: SpawnEventKind,
1409        module_id: String,
1410        spawn_generation: u64,
1411        pid: u32,
1412        exit_code: Option<i32>,
1413        exit_signal: Option<i32>,
1414    ) {
1415        state.seq = state
1416            .seq
1417            .checked_add(1)
1418            .expect("spawn event sequence exhausted");
1419        let event = SpawnEvent {
1420            cursor: Self::cursor(state),
1421            kind,
1422            module_id,
1423            spawn_generation,
1424            pid,
1425            exit_code,
1426            exit_signal,
1427        };
1428        state.events.push_back(event.clone());
1429        while state.events.len() > state.capacity {
1430            state.events.pop_front();
1431        }
1432        let body = match serde_json::to_vec(&event) {
1433            Ok(body) => body,
1434            Err(error) => {
1435                error!(%error, "failed to serialize supervisor spawn event");
1436                return;
1437            }
1438        };
1439        state.subscribers.retain(|(connection_id, corr), subscriber| {
1440            let frame = Frame::build_with_version(
1441                subscriber.version,
1442                FrameType::StreamData,
1443                control_flags(),
1444                0,
1445                0,
1446                *corr,
1447                body.clone(),
1448            );
1449            match frame {
1450                Ok(frame) => {
1451                    if subscriber.frames.try_send(frame).is_ok() {
1452                        true
1453                    } else {
1454                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1455                        if let Some(lagged) = subscriber.lagged.take() {
1456                            let _ = lagged.send(event.cursor.clone());
1457                        }
1458                        false
1459                    }
1460                }
1461                Err(error) => {
1462                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1463                    false
1464                }
1465            }
1466        });
1467    }
1468
1469    fn subscribe(
1470        &self,
1471        connection_id: ConnectionId,
1472        corr: u64,
1473        version: u8,
1474        since: Option<SpawnCursor>,
1475        sink: FrameSink,
1476    ) -> Result<(), SpawnSubscribeRefusal> {
1477        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1478        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1479        {
1480            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1481            let replay = if let Some(since) = since {
1482                if since.daemon_incarnation != state.daemon_incarnation {
1483                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1484                        current: state.daemon_incarnation.clone(),
1485                    });
1486                }
1487                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1488                    if since.seq < oldest.seq.saturating_sub(1) {
1489                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1490                    }
1491                }
1492                state
1493                    .events
1494                    .iter()
1495                    .filter(|event| event.cursor.seq > since.seq)
1496                    .cloned()
1497                    .collect::<Vec<_>>()
1498            } else {
1499                Vec::new()
1500            };
1501            for event in replay {
1502                let body = serde_json::to_vec(&event)
1503                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1504                let frame = Frame::build_with_version(
1505                    version,
1506                    FrameType::StreamData,
1507                    control_flags(),
1508                    0,
1509                    0,
1510                    corr,
1511                    body,
1512                )
1513                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1514                frames
1515                    .try_send(frame)
1516                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1517            }
1518            state.subscribers.insert(
1519                (connection_id, corr),
1520                SpawnSubscriber {
1521                    version,
1522                    frames,
1523                    lagged: Some(lagged),
1524                },
1525            );
1526        }
1527        // The lagged terminal is sent here, by the forwarder, rather than by
1528        // the emitter: at the moment of the drop the subscriber's own channel
1529        // is full, and writing to the connection sink directly from the emitter
1530        // would put the Error AHEAD of the events still queued in that channel
1531        // (and the emitter holds the feed lock, so it cannot await the sink).
1532        // Dropping the subscriber drops the only sender, so `recv` drains every
1533        // queued event and then returns `None`; only then is the Error sent, so
1534        // the client sees each event it can keep, then the reason it was cut.
1535        // Cancel and connection removal drop the oneshot unsent, so they end
1536        // the stream with no Error.
1537        tokio::spawn(async move {
1538            while let Some(frame) = receiver.recv().await {
1539                if sink.send(frame).await.is_err() {
1540                    return;
1541                }
1542            }
1543            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1544                return;
1545            };
1546            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1547                Ok(frame) => {
1548                    let _ = sink.send(frame).await;
1549                }
1550                Err(error) => {
1551                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1552                }
1553            }
1554        });
1555        Ok(())
1556    }
1557
1558    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1559        let Some(subscriber) = self
1560            .0
1561            .lock()
1562            .unwrap_or_else(|p| p.into_inner())
1563            .subscribers
1564            .remove(&(connection_id, corr))
1565        else {
1566            return false;
1567        };
1568        if let Ok(frame) = Frame::build_with_version(
1569            subscriber.version,
1570            FrameType::StreamEnd,
1571            control_flags(),
1572            0,
1573            0,
1574            corr,
1575            Vec::new(),
1576        ) {
1577            tokio::spawn(async move {
1578                let _ = subscriber.frames.send(frame).await;
1579            });
1580        }
1581        true
1582    }
1583
1584    fn remove_connection(&self, connection_id: ConnectionId) {
1585        self.0
1586            .lock()
1587            .unwrap_or_else(|p| p.into_inner())
1588            .subscribers
1589            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1590    }
1591
1592    #[cfg(any(test, feature = "test-support"))]
1593    fn set_capacity(&self, capacity: usize) {
1594        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1595    }
1596
1597    #[cfg(any(test, feature = "test-support"))]
1598    fn subscriber_count(&self) -> usize {
1599        self.0
1600            .lock()
1601            .unwrap_or_else(|p| p.into_inner())
1602            .subscribers
1603            .len()
1604    }
1605}
1606
1607/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1608/// The terminal Error a lagged spawn subscriber receives after its queued events.
1609fn spawn_subscriber_lagged_frame(
1610    version: u8,
1611    corr: u64,
1612    first_undelivered: SpawnCursor,
1613) -> Result<Frame, String> {
1614    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1615        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1616        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1617            .to_string(),
1618        detail: Some(serde_json::json!({
1619            "first_undelivered_cursor": first_undelivered
1620        })),
1621    })
1622    .map_err(|error| error.to_string())?;
1623    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1624        .map_err(|error| error.to_string())
1625}
1626
1627pub trait ModuleProcessLiveness: Send + Sync {
1628    fn process_live(&self, module_id: &str) -> Option<bool>;
1629
1630    /// Whether the supervisor is replacing this module's process right now: an
1631    /// operator restart or reload, a health restart, or a crash respawn whose
1632    /// backoff is running. A module in that state is not live, but a consumer
1633    /// refused now should retry shortly rather than treat the target as gone.
1634    /// Stopped, failed, and disabled modules are not replacing.
1635    fn process_replacing(&self, _module_id: &str) -> bool {
1636        false
1637    }
1638}
1639
1640/// Shared process-liveness registry keyed by supervised `module_id`.
1641#[derive(Debug, Clone, Default)]
1642pub struct SupervisorProcessLiveness {
1643    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1644}
1645
1646impl SupervisorProcessLiveness {
1647    pub fn new() -> Self {
1648        Self::default()
1649    }
1650
1651    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1652        let mut snapshots = self
1653            .snapshots
1654            .lock()
1655            .unwrap_or_else(|poisoned| poisoned.into_inner());
1656        snapshots.insert(module_id, snapshot);
1657    }
1658
1659    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1660        let mut snapshots = self
1661            .snapshots
1662            .lock()
1663            .unwrap_or_else(|poisoned| poisoned.into_inner());
1664        let is_current = snapshots
1665            .get(module_id)
1666            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1667            .unwrap_or(false);
1668        if is_current {
1669            snapshots.remove(module_id);
1670        }
1671    }
1672}
1673
1674impl ModuleProcessLiveness for SupervisorProcessLiveness {
1675    fn process_live(&self, module_id: &str) -> Option<bool> {
1676        let snapshot = {
1677            let snapshots = self
1678                .snapshots
1679                .lock()
1680                .unwrap_or_else(|poisoned| poisoned.into_inner());
1681            snapshots.get(module_id).cloned()
1682        }?;
1683        let snapshot = snapshot
1684            .lock()
1685            .unwrap_or_else(|poisoned| poisoned.into_inner());
1686        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1687    }
1688
1689    fn process_replacing(&self, module_id: &str) -> bool {
1690        let Some(snapshot) = self
1691            .snapshots
1692            .lock()
1693            .unwrap_or_else(|poisoned| poisoned.into_inner())
1694            .get(module_id)
1695            .cloned()
1696        else {
1697            return false;
1698        };
1699        let snapshot = snapshot
1700            .lock()
1701            .unwrap_or_else(|poisoned| poisoned.into_inner());
1702        snapshot.enabled
1703            && match snapshot.state {
1704                ModuleState::Restarting => true,
1705                ModuleState::Draining => snapshot.draining_to_replace,
1706                ModuleState::Starting
1707                | ModuleState::Running
1708                | ModuleState::Unresponsive
1709                | ModuleState::Stopped
1710                | ModuleState::Failed
1711                | ModuleState::Disabled => false,
1712            }
1713    }
1714}
1715
1716#[cfg(test)]
1717#[derive(Debug, Default)]
1718struct ReloadExitRecordGate {
1719    reached: tokio::sync::Notify,
1720    resume: tokio::sync::Notify,
1721}
1722
1723#[derive(Debug, Clone, Copy)]
1724enum RespawnKind {
1725    Spawn,
1726    Reload,
1727}
1728
1729#[derive(Debug, Clone, Copy)]
1730struct PendingRespawn {
1731    deadline: Instant,
1732    kind: RespawnKind,
1733}
1734
1735type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1736
1737#[derive(Debug, Clone)]
1738struct SupervisorRuntimeConfig {
1739    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1740    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1741    /// A reload acknowledges completion only after its replacement registers.
1742    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1743    restart_policy: RestartPolicy,
1744    /// This module's RESOLVED drain budget: per-module config when present,
1745    /// else `default_drain_timeout`.
1746    drain_timeout: Duration,
1747    /// Shared with the status handle so the attested value changes atomically
1748    /// when a rescan updates the running drain policy.
1749    effective_drain_timeout: Arc<Mutex<Duration>>,
1750    /// The supervisor-wide fallback, kept so a configuration update that
1751    /// REMOVES the per-module override can re-resolve to it.
1752    default_drain_timeout: Duration,
1753    health: HealthConfig,
1754    connection_file_path: Option<PathBuf>,
1755    capture_logs_dir: Option<PathBuf>,
1756    forwarding: Option<Arc<ForwardingTable>>,
1757    /// The shared handle, so every spawn path (initial, restart, reload) records the
1758    /// reserved-module launch nonce the HELLO verifier checks against.
1759    supervisor_handle: Option<SupervisorHandle>,
1760    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1761    /// status queries.
1762    ///
1763    /// One ring per module, held across every respawn. The lines explaining an exit
1764    /// are written BEFORE that exit, so a ring recreated per process would be empty
1765    /// exactly when it is asked for.
1766    stderr_ring: Arc<Mutex<StderrRing>>,
1767    terminal_ring: Arc<Mutex<TerminalRing>>,
1768    spawn_events: SpawnEventFeed,
1769    child_roster: ChildRoster,
1770    #[cfg(target_os = "linux")]
1771    cgroup_placement: Option<subc_cgroup::Placement>,
1772    #[cfg(test)]
1773    test_seed_stale_facts_before_enable_spawn: bool,
1774    #[cfg(test)]
1775    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1776}
1777
1778#[derive(Debug, Clone, PartialEq, Eq)]
1779struct SupervisedConfiguration {
1780    spec: ModuleSpec,
1781    health: HealthConfig,
1782}
1783
1784/// Shared daemon lookup table for supervised module handles.
1785///
1786/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1787/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1788/// launch nonces recorded at spawn are checked by the same daemon instance.
1789#[derive(Debug, Clone, Default)]
1790pub struct SupervisorHandle {
1791    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1792    /// Module ids the supervisor has taken on. An id is added BEFORE the
1793    /// module's first process is spawned and removed only when the module
1794    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1795    /// the keys of `modules`.
1796    ///
1797    /// `modules` cannot answer "is this module configured?" on its own: a
1798    /// [`SupervisedModule`] only exists once its process has been spawned, and
1799    /// a fast child can connect, register, sync its scopes and ask about them
1800    /// before the supervisor has inserted it. Answering "not configured" in that
1801    /// gap makes scope admission refuse with the terminal "will never sync"
1802    /// instead of the retryable "has not synced yet".
1803    configured_ids: Arc<Mutex<HashSet<String>>>,
1804    spawn_events: SpawnEventFeed,
1805    /// The current expected launch nonce for each reserved module_id. Set when the
1806    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1807    /// non-reserved module never has an entry here and is never nonce-checked.
1808    /// Reserved module ids and the nonce that authorizes their next HELLO.
1809    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1810    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1811    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1812    /// had NO entry and admitted anyone: the reservation protected the nonce
1813    /// holder, not the NAME (found live by CKCRED's canary probe registering
1814    /// against a reserved scratch id).
1815    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1816    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1817    ///
1818    /// This is deliberately in-memory only: subc is state-free across daemon
1819    /// restarts, and the tombstone only explains the hours-after-removal window
1820    /// while this executing daemon is still alive. Do not persist it in a store.
1821    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1822    /// The current launch nonce for every supervised spawn. This is separate from
1823    /// reserved_nonces because consumer route.open attestation applies to all spawned
1824    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1825    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1826    /// Reserved namespace prefixes mapped to the supervised owner module whose
1827    /// current spawn nonce authorizes HELLO claims below the prefix.
1828    ///
1829    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1830    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1831    /// accidental collisions and lower-trust processes from squatting protected
1832    /// namespaces.
1833    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1834    /// Blue/green swaps in progress, by module id. An entry exists from just
1835    /// before the candidate process is spawned until the swap has failed, or
1836    /// has cut over and the old process is gone. While it exists, HELLO for the
1837    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1838    /// consumer attestation accepts both processes' nonces.
1839    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1840    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1841    promotion_observer: PromotionObserverSlot,
1842    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1843    /// this daemon-wide ordering, a rescan could retire or update a module while a
1844    /// concurrent reload still held its old handle and launch specification.
1845    operation_lock: Arc<AsyncMutex<()>>,
1846}
1847
1848/// Told when a swap has promoted its candidate to be the module's active
1849/// registration.
1850///
1851/// An ordinary HELLO runs the control plane's registration side effects (the
1852/// capability cache, the deny census, the requirement recompute) as it
1853/// registers. A swap candidate's HELLO does not, because it is not routable;
1854/// promotion is when those must run instead, and promotion happens in the
1855/// supervisor, which has no other way into the control handler.
1856pub(crate) trait SwapPromotionObserver: Send + Sync {
1857    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1858}
1859
1860/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1861/// control handler) owns this handle, so a strong reference back would be a
1862/// cycle that keeps both alive.
1863#[derive(Clone, Default)]
1864struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1865
1866impl fmt::Debug for PromotionObserverSlot {
1867    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1868        f.write_str("PromotionObserverSlot")
1869    }
1870}
1871
1872/// The nonces of one open swap.
1873#[derive(Debug, Clone)]
1874struct OpenSwap {
1875    /// The launch nonce minted for the candidate process. It is the swap
1876    /// token: the only thing that admits a HELLO into the candidate slot.
1877    candidate_nonce: String,
1878    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1879    /// here because cutover moves the module's recorded spawn nonce to the
1880    /// candidate while the incumbent is still draining and its consumers are
1881    /// still attesting with this one.
1882    incumbent_nonce: Option<String>,
1883    /// Set once a HELLO has been admitted with the swap token, so the token
1884    /// admits one registration and cannot be replayed after cutover empties
1885    /// the candidate slot.
1886    candidate_admitted: bool,
1887}
1888
1889/// What the swap gate says about a HELLO. See
1890/// [`SupervisorHandle::swap_hello_admission`].
1891#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1892pub(crate) enum SwapHelloAdmission {
1893    /// No swap is open for the id (or the HELLO carries the incumbent's own
1894    /// nonce); the ordinary gates decide.
1895    NotSwapping,
1896    /// The HELLO carries the swap token: register it into the candidate slot.
1897    Candidate,
1898    /// A swap is open and the HELLO carries a nonce the supervisor did not
1899    /// mint for this id, no nonce, or a token already used.
1900    Refused,
1901}
1902
1903#[derive(Debug, Clone, PartialEq, Eq)]
1904pub(crate) enum ReservedHelloRejection {
1905    Exact {
1906        module_id: String,
1907    },
1908    Prefix {
1909        prefix: String,
1910        owner_module_id: String,
1911    },
1912}
1913
1914impl SupervisorHandle {
1915    pub fn new() -> Self {
1916        Self::default()
1917    }
1918
1919    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1920        self.spawn_events.snapshot()
1921    }
1922
1923    pub(crate) fn subscribe_spawns(
1924        &self,
1925        connection_id: ConnectionId,
1926        corr: u64,
1927        version: u8,
1928        since: Option<SpawnCursor>,
1929        sink: FrameSink,
1930    ) -> Result<(), SpawnSubscribeRefusal> {
1931        self.spawn_events
1932            .subscribe(connection_id, corr, version, since, sink)
1933    }
1934
1935    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1936        self.spawn_events.cancel(connection_id, corr)
1937    }
1938
1939    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1940        self.spawn_events.remove_connection(connection_id);
1941    }
1942
1943    #[cfg(any(test, feature = "test-support"))]
1944    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1945        assert!(capacity > 0, "spawn event capacity must be non-zero");
1946        self.spawn_events.set_capacity(capacity);
1947    }
1948
1949    #[cfg(any(test, feature = "test-support"))]
1950    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1951        self.spawn_events.subscriber_count()
1952    }
1953
1954    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1955    /// a respawn invalidates stale consumer identities.
1956    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1957        self.spawn_nonces
1958            .lock()
1959            .unwrap_or_else(|poisoned| poisoned.into_inner())
1960            .insert(module_id.to_string(), nonce);
1961    }
1962
1963    /// Record the launch nonce expected from the next HELLO for a reserved module,
1964    /// replacing any prior nonce (a respawn invalidates the previous one).
1965    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1966        self.reserved_nonces
1967            .lock()
1968            .unwrap_or_else(|poisoned| poisoned.into_inner())
1969            .insert(module_id.to_string(), Some(nonce));
1970    }
1971
1972    /// Record namespace prefixes owned by a supervised module.
1973    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1974        let mut owners = self
1975            .reserved_prefix_owners
1976            .lock()
1977            .unwrap_or_else(|poisoned| poisoned.into_inner());
1978        owners.retain(|_, owner| owner != owner_module_id);
1979        for prefix in prefixes {
1980            owners.insert(prefix.clone(), owner_module_id.to_string());
1981        }
1982    }
1983
1984    /// The launch nonce most recently minted for a module's spawn, if any.
1985    #[cfg(test)]
1986    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1987        self.spawn_nonces
1988            .lock()
1989            .unwrap_or_else(|poisoned| poisoned.into_inner())
1990            .get(module_id)
1991            .cloned()
1992    }
1993
1994    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1995        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1996        let spawn_nonce = self
1997            .spawn_nonces
1998            .lock()
1999            .unwrap_or_else(|poisoned| poisoned.into_inner())
2000            .get(&spec.module_id)
2001            .cloned();
2002        let mut reserved_nonces = self
2003            .reserved_nonces
2004            .lock()
2005            .unwrap_or_else(|poisoned| poisoned.into_inner());
2006        if spec.reserved {
2007            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
2008            // reserved name whose module has never spawned has no legitimate
2009            // holder, and the entry's absence is what used to leave the name
2010            // open to the first claimant.
2011            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
2012        }
2013        drop(reserved_nonces);
2014        // A later unreserved declaration must not silently unreserve an id that
2015        // was retained after its reserved configuration was removed. The explicit
2016        // release ceremony is the only operation that retires that gate.
2017        self.removal_tombstones
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner())
2020            .remove(&spec.module_id);
2021    }
2022
2023    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
2024    /// authorized only by its expected nonce; otherwise a matching reserved prefix
2025    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
2026    /// with no matching prefix are always authorized.
2027    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2028        self.reserved_hello_rejection(module_id, presented)
2029            .is_none()
2030    }
2031
2032    pub(crate) fn reserved_hello_rejection(
2033        &self,
2034        module_id: &str,
2035        presented: Option<&str>,
2036    ) -> Option<ReservedHelloRejection> {
2037        let nonces = self
2038            .reserved_nonces
2039            .lock()
2040            .unwrap_or_else(|poisoned| poisoned.into_inner());
2041        if let Some(expected) = nonces.get(module_id) {
2042            // `None` = reserved with no legitimate holder: refuse every
2043            // presentation, because no process can hold a nonce that was never
2044            // minted. Only a real minted nonce admits, in constant time.
2045            let authorized = match expected {
2046                Some(expected) => {
2047                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2048                }
2049                None => false,
2050            };
2051            if authorized {
2052                return None;
2053            }
2054            return Some(ReservedHelloRejection::Exact {
2055                module_id: module_id.to_string(),
2056            });
2057        }
2058        drop(nonces);
2059
2060        let matched_prefix = self
2061            .reserved_prefix_owners
2062            .lock()
2063            .unwrap_or_else(|poisoned| poisoned.into_inner())
2064            .iter()
2065            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2066            .max_by_key(|(prefix, _)| prefix.len())
2067            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2068        let (prefix, owner_module_id) = matched_prefix?;
2069
2070        let authorized = presented.is_some_and(|presented| {
2071            self.spawn_nonces
2072                .lock()
2073                .unwrap_or_else(|poisoned| poisoned.into_inner())
2074                .get(&owner_module_id)
2075                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2076                // While the owner is being swapped, children started by
2077                // either of its two processes hold that process's nonce.
2078                || self.swap_nonce_matches(&owner_module_id, presented)
2079        });
2080        if authorized {
2081            None
2082        } else {
2083            Some(ReservedHelloRejection::Prefix {
2084                prefix,
2085                owner_module_id,
2086            })
2087        }
2088    }
2089
2090    /// Whether a consumer connection proved it came from a daemon-spawned module.
2091    ///
2092    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2093    /// accepted only for module ids the supervisor has spawned.
2094    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2095        if presented.is_empty() {
2096            return false;
2097        }
2098        let nonces = self
2099            .spawn_nonces
2100            .lock()
2101            .unwrap_or_else(|poisoned| poisoned.into_inner());
2102        let current = nonces
2103            .get(module_id)
2104            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2105        drop(nonces);
2106        // During a swap two processes of the module are alive, and a consumer
2107        // started by either one presents that process's nonce. Accepting only
2108        // the recorded one would fail the incumbent's consumers for the whole
2109        // overlap once cutover moves the record to the candidate.
2110        current || self.swap_nonce_matches(module_id, presented)
2111    }
2112
2113    /// Whether `presented` is either nonce of an open swap for `module_id`.
2114    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2115        let swaps = self
2116            .swaps
2117            .lock()
2118            .unwrap_or_else(|poisoned| poisoned.into_inner());
2119        swaps.get(module_id).is_some_and(|swap| {
2120            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2121                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2122                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2123                })
2124        })
2125    }
2126
2127    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2128    /// Called before the candidate process exists.
2129    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2130        let incumbent_nonce = self
2131            .spawn_nonces
2132            .lock()
2133            .unwrap_or_else(|poisoned| poisoned.into_inner())
2134            .get(module_id)
2135            .cloned();
2136        self.swaps
2137            .lock()
2138            .unwrap_or_else(|poisoned| poisoned.into_inner())
2139            .insert(
2140                module_id.to_string(),
2141                OpenSwap {
2142                    candidate_nonce,
2143                    incumbent_nonce,
2144                    candidate_admitted: false,
2145                },
2146            );
2147    }
2148
2149    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2150    /// the module's recorded one.
2151    pub(crate) fn close_swap(&self, module_id: &str) {
2152        self.swaps
2153            .lock()
2154            .unwrap_or_else(|poisoned| poisoned.into_inner())
2155            .remove(module_id);
2156    }
2157
2158    /// Install the observer told about swap promotions, replacing any earlier
2159    /// one.
2160    pub(crate) fn set_swap_promotion_observer(
2161        &self,
2162        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2163    ) {
2164        *self
2165            .promotion_observer
2166            .0
2167            .lock()
2168            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2169    }
2170
2171    /// Tell the installed observer, if it is still alive, that a swap promoted
2172    /// `registration`.
2173    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2174        let observer = self
2175            .promotion_observer
2176            .0
2177            .lock()
2178            .unwrap_or_else(|poisoned| poisoned.into_inner())
2179            .as_ref()
2180            .and_then(std::sync::Weak::upgrade);
2181        if let Some(observer) = observer {
2182            observer.swap_promoted(registration);
2183        }
2184    }
2185
2186    /// Whether a swap is open for `module_id`.
2187    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2188        self.swaps
2189            .lock()
2190            .unwrap_or_else(|poisoned| poisoned.into_inner())
2191            .contains_key(module_id)
2192    }
2193
2194    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2195    /// respawn would, once cutover has made the candidate the module's process.
2196    /// The swap stays open so the incumbent's nonce keeps attesting until the
2197    /// incumbent has drained and exited.
2198    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2199        let candidate_nonce = self
2200            .swaps
2201            .lock()
2202            .unwrap_or_else(|poisoned| poisoned.into_inner())
2203            .get(module_id)
2204            .map(|swap| swap.candidate_nonce.clone());
2205        let Some(nonce) = candidate_nonce else {
2206            return;
2207        };
2208        self.set_spawn_nonce(module_id, nonce.clone());
2209        if reserved {
2210            self.set_reserved_nonce(module_id, nonce);
2211        }
2212    }
2213
2214    /// The swap gate for a HELLO claiming `module_id`.
2215    ///
2216    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2217    /// presents the candidate nonce, which the reserved gate (holding the
2218    /// incumbent's nonce) would refuse as `reserved_module` before swap
2219    /// admission was ever reached. And it applies to unreserved ids too: for an
2220    /// unreserved id the only thing that ever stopped a second process claiming
2221    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2222    /// refusal a swap lifts for its candidate.
2223    ///
2224    /// The incumbent's own nonce falls through to the ordinary gates, which
2225    /// treat it as they always have (a live incumbent is refused as a
2226    /// duplicate). Anything else while a swap is open is refused, including an
2227    /// absent nonce.
2228    pub(crate) fn swap_hello_admission(
2229        &self,
2230        module_id: &str,
2231        presented: Option<&str>,
2232    ) -> SwapHelloAdmission {
2233        let swaps = self
2234            .swaps
2235            .lock()
2236            .unwrap_or_else(|poisoned| poisoned.into_inner());
2237        let Some(swap) = swaps.get(module_id) else {
2238            return SwapHelloAdmission::NotSwapping;
2239        };
2240        let Some(presented) = presented else {
2241            return SwapHelloAdmission::Refused;
2242        };
2243        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2244            return if swap.candidate_admitted {
2245                SwapHelloAdmission::Refused
2246            } else {
2247                SwapHelloAdmission::Candidate
2248            };
2249        }
2250        if swap
2251            .incumbent_nonce
2252            .as_deref()
2253            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2254        {
2255            return SwapHelloAdmission::NotSwapping;
2256        }
2257        SwapHelloAdmission::Refused
2258    }
2259
2260    /// Record that the swap token has registered a candidate, so it admits no
2261    /// second HELLO.
2262    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2263        if let Some(swap) = self
2264            .swaps
2265            .lock()
2266            .unwrap_or_else(|poisoned| poisoned.into_inner())
2267            .get_mut(module_id)
2268        {
2269            swap.candidate_admitted = true;
2270        }
2271    }
2272
2273    /// Test/support lookup for the current launch nonce of a supervised spawn.
2274    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2275        self.spawn_nonces
2276            .lock()
2277            .unwrap_or_else(|poisoned| poisoned.into_inner())
2278            .get(module_id)
2279            .cloned()
2280    }
2281
2282    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2283    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2284        self.reserved_nonces
2285            .lock()
2286            .unwrap_or_else(|poisoned| poisoned.into_inner())
2287            .get(module_id)
2288            .cloned()
2289            .flatten()
2290    }
2291
2292    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2293        // Normally already marked before the process was spawned; marking here
2294        // too keeps `configured_ids` a superset of the roster for any caller
2295        // that inserts a module directly.
2296        self.mark_configured(module.module_id());
2297        let mut modules = self
2298            .modules
2299            .lock()
2300            .unwrap_or_else(|poisoned| poisoned.into_inner());
2301        modules.insert(module.module_id().to_string(), module)
2302    }
2303
2304    /// Record that the supervisor has taken on `module_id`. Called before the
2305    /// module's first process is spawned, so that by the time that process can
2306    /// register, [`Self::is_configured`] already answers true.
2307    fn mark_configured(&self, module_id: &str) {
2308        self.configured_ids
2309            .lock()
2310            .unwrap_or_else(|poisoned| poisoned.into_inner())
2311            .insert(module_id.to_string());
2312    }
2313
2314    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2315    /// before it was ever put on the roster. A module already on the roster
2316    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2317    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2318        let modules = self
2319            .modules
2320            .lock()
2321            .unwrap_or_else(|poisoned| poisoned.into_inner());
2322        if !modules.contains_key(module_id) {
2323            self.configured_ids
2324                .lock()
2325                .unwrap_or_else(|poisoned| poisoned.into_inner())
2326                .remove(module_id);
2327        }
2328    }
2329
2330    /// Whether `module_id` is a module this daemon supervises: on the roster,
2331    /// or about to be (its process is being spawned right now).
2332    ///
2333    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2334    /// for scopes: a supervised module's process can register and sync before
2335    /// [`Self::get`] can return it, and in that window it is still a module
2336    /// that will sync, not one that never will.
2337    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2338        self.configured_ids
2339            .lock()
2340            .unwrap_or_else(|poisoned| poisoned.into_inner())
2341            .contains(module_id)
2342    }
2343
2344    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2345        let modules = self
2346            .modules
2347            .lock()
2348            .unwrap_or_else(|poisoned| poisoned.into_inner());
2349        modules.get(module_id).cloned()
2350    }
2351
2352    pub(crate) fn record_late_health_answer(
2353        &self,
2354        module_id: &str,
2355        latency_ms: u64,
2356    ) -> Result<bool, SuperviseError> {
2357        let Some(module) = self.get(module_id) else {
2358            return Ok(false);
2359        };
2360        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2361            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2362            state.health.last_late_answer_latency_ms = Some(latency_ms);
2363            // A late answer is an answer: the module served the probe, just past
2364            // the deadline. Leaving the miss streak in place while logging
2365            // "proves the module is alive" is how a CPU-starved module that
2366            // answers every probe a few seconds late still marches to the
2367            // threshold and gets killed — the exact kill class `NoAnswer` is
2368            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2369            // is degradation, and degradation reports; it does not restart.
2370            state.health.consecutive_failures = 0;
2371        })?;
2372        Ok(true)
2373    }
2374
2375    /// Arm the one-shot marker for the module process that this caller
2376    /// deliberately initiated severance against. Generic connection teardown
2377    /// must not call this:
2378    /// a surviving process would otherwise retain an exemption for a later
2379    /// genuine crash.
2380    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2381        let Some(module) = self.get(module_id) else {
2382            return Ok(false);
2383        };
2384        let snapshot = lock_snapshot(&module.inner.snapshot)?;
2385        let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2386            return Ok(false);
2387        };
2388        drop(snapshot);
2389        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2390    }
2391
2392    pub fn list(&self) -> Vec<SupervisedModule> {
2393        let modules = self
2394            .modules
2395            .lock()
2396            .unwrap_or_else(|poisoned| poisoned.into_inner());
2397        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2398        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2399        modules
2400    }
2401
2402    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2403        self.spawn_nonces
2404            .lock()
2405            .unwrap_or_else(|poisoned| poisoned.into_inner())
2406            .remove(module_id);
2407        self.close_swap(module_id);
2408        let mut reserved_nonces = self
2409            .reserved_nonces
2410            .lock()
2411            .unwrap_or_else(|poisoned| poisoned.into_inner());
2412        if reserved_nonces.contains_key(module_id) {
2413            // The old nonce must die with the removed process, but the exact-id
2414            // gate remains until an operator explicitly releases it.
2415            reserved_nonces.insert(module_id.to_string(), None);
2416        }
2417        drop(reserved_nonces);
2418        self.reserved_prefix_owners
2419            .lock()
2420            .unwrap_or_else(|poisoned| poisoned.into_inner())
2421            .retain(|_, owner| owner != module_id);
2422        let removed = self
2423            .modules
2424            .lock()
2425            .unwrap_or_else(|poisoned| poisoned.into_inner())
2426            .remove(module_id);
2427        self.configured_ids
2428            .lock()
2429            .unwrap_or_else(|poisoned| poisoned.into_inner())
2430            .remove(module_id);
2431        removed
2432    }
2433
2434    /// Remember a module removed by a non-preview rescan so route.open can
2435    /// distinguish that intentional removal from an unknown id.
2436    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2437        self.removal_tombstones
2438            .lock()
2439            .unwrap_or_else(|poisoned| poisoned.into_inner())
2440            .insert(module_id.to_string(), unix_ms_now());
2441    }
2442
2443    /// Return how long ago a rescan removed this module in milliseconds.
2444    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2445        self.removal_tombstones
2446            .lock()
2447            .unwrap_or_else(|poisoned| poisoned.into_inner())
2448            .get(module_id)
2449            .copied()
2450            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2451    }
2452
2453    /// Retire a reserved-id gate only after its module has left supervision.
2454    ///
2455    /// A retained gate has no live nonce (`None`), so releasing any other entry
2456    /// would weaken a currently configured or otherwise active reservation.
2457    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2458        if self.get(module_id).is_some() {
2459            return false;
2460        }
2461        let mut reserved_nonces = self
2462            .reserved_nonces
2463            .lock()
2464            .unwrap_or_else(|poisoned| poisoned.into_inner());
2465        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2466            return false;
2467        }
2468        reserved_nonces.remove(module_id);
2469        true
2470    }
2471
2472    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2473        Arc::clone(&self.operation_lock)
2474    }
2475}
2476
2477/// Process supervisor for subc-owned singleton modules.
2478#[derive(Debug, Clone)]
2479pub struct Supervisor {
2480    registry: Arc<Registry>,
2481    restart_policy: RestartPolicy,
2482    drain_timeout: Duration,
2483    connection_file_path: Option<PathBuf>,
2484    capture_logs_dir: Option<PathBuf>,
2485    forwarding: Option<Arc<ForwardingTable>>,
2486    process_liveness: Arc<SupervisorProcessLiveness>,
2487    supervisor_handle: Option<SupervisorHandle>,
2488    health: HealthConfig,
2489    daemon_start_clock: crate::clock::StartClock,
2490    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2491    spawn_events: SpawnEventFeed,
2492    provenance_probe: ExecutableIdentityProbe,
2493    /// Every process spawned through this supervisor (and its clones) and not
2494    /// yet reaped, so daemon shutdown can end them.
2495    child_roster: ChildRoster,
2496    #[cfg(target_os = "linux")]
2497    cgroup_placement: Option<subc_cgroup::Placement>,
2498    #[cfg(test)]
2499    test_after_first_spawn: AfterFirstSpawnHook,
2500}
2501
2502/// Test-only hook run on the path that takes on a new module, right after its
2503/// first `spawn_child` returns (the process exists and could already be
2504/// registering) and before that process is handed to the module's supervise
2505/// loop and put on the roster. Lets a test observe what a fast child would see
2506/// in that window without racing a real one.
2507#[cfg(test)]
2508#[derive(Clone, Default)]
2509struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2510
2511#[cfg(test)]
2512type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2513
2514#[cfg(test)]
2515impl fmt::Debug for AfterFirstSpawnHook {
2516    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2517        f.write_str("AfterFirstSpawnHook")
2518    }
2519}
2520
2521#[cfg(test)]
2522impl AfterFirstSpawnHook {
2523    fn run(&self, module_id: &str) {
2524        if let Some(hook) = &self.0 {
2525            hook(module_id);
2526        }
2527    }
2528}
2529
2530impl Supervisor {
2531    #[cfg(test)]
2532    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2533        let supervisor = Self::new(registry, policy);
2534        #[cfg(target_os = "macos")]
2535        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2536        supervisor
2537    }
2538    /// Verify the trampoline once when configured. Missing private OS support
2539    /// refuses every macOS launch by name but does not stop the daemon's control
2540    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2541    /// entry point; the library must not exec an arbitrary hosting program.
2542    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2543        let path = path.into();
2544        #[cfg(target_os = "macos")]
2545        {
2546            let result = probe_privacy_trampoline(&path).map(|()| path);
2547            if let Err(cause) = &result {
2548                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2549            }
2550            self.child_roster.set_privacy_trampoline(result);
2551        }
2552        #[cfg(not(target_os = "macos"))]
2553        let _ = path;
2554        self
2555    }
2556    /// The first step of an announced daemon shutdown, before the notice and
2557    /// before any connection is closed.
2558    ///
2559    /// Sets the daemon-shutdown flag first: from here on no module is
2560    /// respawned (crash restart, operator restart, or swap), and every child
2561    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2562    /// the module exits on the EOF this shutdown gives it or is signalled by a
2563    /// service manager that kills the whole cgroup. Then writes the journal's
2564    /// shutdown marker, which records the instant and closes this daemon
2565    /// incarnation's stretch of the journal.
2566    #[cfg(unix)]
2567    pub(crate) fn begin_daemon_shutdown(&self) {
2568        self.child_roster.close();
2569        if let Some(journal) = &self.terminal_journal {
2570            journal.stamp_shutdown();
2571        }
2572    }
2573
2574    /// Announce a cut while established connections can still carry replies.
2575    /// These budgets promise notice and a bounded wait, not child completion;
2576    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2577    #[cfg(unix)]
2578    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2579        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2580        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2581        let Some(forwarding) = &self.forwarding else {
2582            return Ok(());
2583        };
2584        let module_ids = forwarding
2585            .begin_daemon_drain()
2586            .map_err(SuperviseError::Forwarding)?;
2587        let deadline_ms =
2588            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2589        let mut notices = tokio::task::JoinSet::new();
2590        let mut drains = Vec::new();
2591        for module_id in module_ids {
2592            let Some(target) = forwarding
2593                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2594                .map_err(SuperviseError::Forwarding)?
2595            else {
2596                continue;
2597            };
2598            let routes = forwarding
2599                .endpoint_routes(target.endpoint)
2600                .map_err(SuperviseError::Forwarding)?;
2601            // Restart allows deployed consumers to reopen after the new daemon
2602            // appears. The wire reason stays `restart`; what tells a daemon cut
2603            // apart from a module restart afterwards is the terminal record
2604            // itself, whose disposition is `daemon_shutdown` for every exit
2605            // observed once `begin_daemon_shutdown` has run.
2606            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2607                reason: RouteCloseReason::Restart,
2608                deadline_ms,
2609            })
2610            .expect("module draining serializes");
2611            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2612            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2613            for route in routes {
2614                let client = route.goodbye_target;
2615                if let Some((_, channels)) = clients
2616                    .iter_mut()
2617                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2618                {
2619                    channels.push(client.channel);
2620                } else {
2621                    let channel = client.channel;
2622                    clients.push((client, vec![channel]));
2623                }
2624            }
2625            for (client, mut channels) in clients {
2626                channels.sort_unstable();
2627                channels.dedup();
2628                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2629                    module_id: module_id.clone(),
2630                    channels,
2631                    reason: RouteCloseReason::Restart,
2632                })
2633                .expect("route closing serializes");
2634                recipients.push((client.sink, client.negotiated_ver, closing));
2635            }
2636            for (sink, version, body) in recipients {
2637                notices.spawn(async move {
2638                    let frame = Frame::build_with_version(
2639                        version,
2640                        FrameType::Push,
2641                        control_flags(),
2642                        0,
2643                        0,
2644                        0,
2645                        body,
2646                    )
2647                    .expect("bounded lifecycle notice frame builds");
2648                    sink.send_flushed(frame).await
2649                });
2650            }
2651            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2652            drains.push((module_id, target.endpoint, gauges));
2653        }
2654        // A quiet forwarding table is not proof that queued notices reached the
2655        // socket. Wait for writer flush acknowledgements before testing quiescence.
2656        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2657        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2658            if !matches!(result, Ok(Ok(()))) {
2659                warn!(?result, "daemon shutdown notice delivery failed");
2660            }
2661        }
2662        notices.abort_all();
2663        let deadline = Instant::now() + DRAIN_BUDGET;
2664        let mut waits = tokio::task::JoinSet::new();
2665        for (module_id, endpoint, gauges) in drains {
2666            let forwarding = Arc::clone(forwarding);
2667            let mut runtime = self.runtime_config();
2668            runtime.health.cadence = Duration::from_millis(100);
2669            waits.spawn(async move {
2670                wait_for_forwarding_quiescence(
2671                    &forwarding,
2672                    &module_id,
2673                    &runtime,
2674                    endpoint,
2675                    deadline,
2676                    &gauges,
2677                    DrainScope::Active,
2678                )
2679                .await
2680            });
2681        }
2682        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2683            if !matches!(result, Ok(Ok(true))) {
2684                warn!(?result, "daemon shutdown drain did not reach quiescence");
2685            }
2686        }
2687        Ok(())
2688    }
2689
2690    /// The last step of an announced daemon shutdown, after the notice and the
2691    /// drain: send every registered module a module GOODBYE, the same planned
2692    /// stop signal `ck module stop` gives, then close every connection so each
2693    /// subc module sees EOF and starts its own teardown, then end every
2694    /// supervised child that has not exited
2695    /// by its own deadline (its drain budget, capped). Modules lead their own
2696    /// process groups, so a
2697    /// service manager's group kill no longer reaches them; without this a
2698    /// child that does not stop on EOF (every `protocol: "none"` child, which
2699    /// has no connection) would outlive the daemon. Every wait is bounded (see
2700    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2701    #[cfg(unix)]
2702    pub(crate) async fn end_children_for_daemon_shutdown(
2703        &self,
2704        already_escalated: bool,
2705        escalate: impl std::future::Future<Output = ()>,
2706    ) {
2707        tokio::pin!(escalate);
2708        let mut escalated = already_escalated;
2709        if let Some(forwarding) = &self.forwarding {
2710            let reason = CloseReason::new(
2711                "daemon_shutdown",
2712                "the daemon is exiting after its shutdown notice and drain",
2713            );
2714            if escalated {
2715                // The operator asked to stop waiting: queue the GOODBYEs but
2716                // do not wait for them to be written.
2717                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2718            } else {
2719                tokio::select! {
2720                    biased;
2721                    _ = escalate.as_mut() => {
2722                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2723                        escalated = true;
2724                    }
2725                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2726                }
2727            }
2728            let closed = forwarding.close_all_connections(&reason);
2729            debug!(closed, "closed established connections for daemon shutdown");
2730        }
2731        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2732        // already completed and must not be polled again; the child shutdown
2733        // wait is told it is escalated and gets a future that never fires.
2734        let escalated_here = escalated && !already_escalated;
2735        let remaining_escalate = async move {
2736            if escalated_here {
2737                std::future::pending::<()>().await;
2738            } else {
2739                escalate.await;
2740            }
2741        };
2742        crate::child_roster::end_children_for_daemon_shutdown(
2743            &self.child_roster,
2744            escalated,
2745            remaining_escalate,
2746        )
2747        .await;
2748    }
2749
2750    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2751        Self {
2752            registry,
2753            restart_policy,
2754            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2755            connection_file_path: None,
2756            capture_logs_dir: None,
2757            forwarding: None,
2758            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2759            supervisor_handle: None,
2760            health: HealthConfig::default(),
2761            daemon_start_clock: crate::clock::StartClock::capture(),
2762            terminal_journal: None,
2763            spawn_events: SpawnEventFeed::default(),
2764            provenance_probe: ExecutableIdentityProbe::default(),
2765            child_roster: ChildRoster::default(),
2766            #[cfg(target_os = "linux")]
2767            cgroup_placement: None,
2768            #[cfg(test)]
2769            test_after_first_spawn: AfterFirstSpawnHook::default(),
2770        }
2771    }
2772
2773    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2774        self.drain_timeout = drain_timeout;
2775        self
2776    }
2777
2778    pub fn with_process_liveness(
2779        mut self,
2780        process_liveness: Arc<SupervisorProcessLiveness>,
2781    ) -> Self {
2782        self.process_liveness = process_liveness;
2783        self
2784    }
2785
2786    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2787        self.connection_file_path = Some(connection_file_path.into());
2788        self
2789    }
2790
2791    /// Enables daemon-owned capture files for supervised stdout and stderr.
2792    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2793        self.capture_logs_dir = Some(logs_dir.into());
2794        self
2795    }
2796
2797    /// Names this daemon lifetime in spawn events, independently of whether a
2798    /// terminal journal is configured.
2799    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2800        // A millisecond start stamp can repeat after clock rollback or a rapid
2801        // restart. Use the connection file's random daemon_id instead: it already
2802        // identifies this daemon lifetime independently of the wall clock.
2803        self.spawn_events.configure_incarnation(daemon_incarnation);
2804        self
2805    }
2806
2807    /// Enables best-effort history shared by every supervised module. Without
2808    /// it, terminal history is kept only in each module's in-memory ring.
2809    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2810        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2811        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2812            path,
2813            daemon_incarnation,
2814        )));
2815        this
2816    }
2817
2818    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2819        self.forwarding = Some(forwarding);
2820        self
2821    }
2822
2823    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2824        self.spawn_events = supervisor_handle.spawn_events.clone();
2825        self.supervisor_handle = Some(supervisor_handle);
2826        self
2827    }
2828
2829    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2830        self.health = health;
2831        self
2832    }
2833
2834    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2835    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2836    /// record is kept.
2837    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2838        self.child_roster.record_to(path.into());
2839        self
2840    }
2841
2842    #[cfg(target_os = "linux")]
2843    pub fn with_cgroup_placement(
2844        mut self,
2845        cgroup_placement: Option<subc_cgroup::Placement>,
2846    ) -> Self {
2847        self.cgroup_placement = cgroup_placement;
2848        self
2849    }
2850
2851    /// Spawn `spec.program` and start monitoring it.
2852    ///
2853    /// The child is expected to parse `--subc <connection-file-path>`, read the
2854    /// TCP+key connection file, authenticate to the already-running listener, and
2855    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2856    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2857        validate_spec(&spec)?;
2858        self.establish_identity(&spec);
2859
2860        let runtime = self.runtime_config();
2861        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2862        let spawned = spawn_child(
2863            &spec,
2864            runtime.connection_file_path.as_deref(),
2865            self.supervisor_handle.as_ref(),
2866            &runtime.stderr_ring,
2867            runtime.capture_logs_dir.as_deref(),
2868            &runtime.child_roster,
2869            #[cfg(target_os = "linux")]
2870            runtime.cgroup_placement.as_ref(),
2871        );
2872        #[cfg(test)]
2873        self.test_after_first_spawn.run(&spec.module_id);
2874        let child = match spawned {
2875            Ok(child) => child,
2876            Err(err) => {
2877                // Unlike the configured paths, a failed `spawn` leaves nothing
2878                // on the roster, so the module must not stay marked configured.
2879                self.abandon_unrostered(&spec.module_id);
2880                return Err(err);
2881            }
2882        };
2883        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2884
2885        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2886    }
2887
2888    /// Make `spec`'s module count as configured, with its identity gates
2889    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2890    /// exists.
2891    ///
2892    /// Every path that takes on a new module calls this before `spawn_child`.
2893    /// The order is the point: the child can connect, register, sync its
2894    /// scopes and ask about them as soon as it is spawned, and the module is
2895    /// only put on the roster after `spawn_child` returns. Were the mark set
2896    /// with the roster entry, a fast child would see its own owner reported
2897    /// as not configured, and a scoped `route.open` in that window would be
2898    /// refused as terminal `scope_not_live` ("will never sync") instead of
2899    /// retryable `scope_not_synced`.
2900    fn establish_identity(&self, spec: &ModuleSpec) {
2901        if let Some(supervisor_handle) = &self.supervisor_handle {
2902            supervisor_handle.apply_identity_configuration(spec);
2903            supervisor_handle.mark_configured(&spec.module_id);
2904        }
2905    }
2906
2907    /// Take back [`Self::establish_identity`]'s configured mark when the
2908    /// module will not be put on the roster after all.
2909    fn abandon_unrostered(&self, module_id: &str) {
2910        if let Some(supervisor_handle) = &self.supervisor_handle {
2911            supervisor_handle.unmark_configured_unless_rostered(module_id);
2912        }
2913    }
2914
2915    /// Record a freshly spawned first process as running. On failure the
2916    /// module never reaches the roster, so its configured mark is taken back.
2917    fn mark_first_process_running(
2918        &self,
2919        spec: &ModuleSpec,
2920        runtime: &SupervisorRuntimeConfig,
2921        snapshot: &SharedSnapshot,
2922        child: &SupervisedChild,
2923    ) -> Result<(), SuperviseError> {
2924        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2925            self.abandon_unrostered(&spec.module_id);
2926            return Err(err);
2927        }
2928        self.process_liveness
2929            .track(spec.module_id.clone(), Arc::clone(snapshot));
2930        Ok(())
2931    }
2932
2933    /// Start supervising a module declared in daemon configuration.
2934    ///
2935    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2936    /// failures in the supervisor handle so operator-facing `supervisor.list`
2937    /// reflects every configured module while daemon startup continues.
2938    pub fn supervise_configured(
2939        &self,
2940        spec: ModuleSpec,
2941        enabled: bool,
2942    ) -> Result<SupervisedModule, SuperviseError> {
2943        validate_spec(&spec)?;
2944        self.establish_identity(&spec);
2945
2946        let runtime = self.runtime_config();
2947        if !enabled {
2948            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2949            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2950        }
2951
2952        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2953        let spawned = spawn_child(
2954            &spec,
2955            runtime.connection_file_path.as_deref(),
2956            self.supervisor_handle.as_ref(),
2957            &runtime.stderr_ring,
2958            runtime.capture_logs_dir.as_deref(),
2959            &runtime.child_roster,
2960            #[cfg(target_os = "linux")]
2961            runtime.cgroup_placement.as_ref(),
2962        );
2963        #[cfg(test)]
2964        self.test_after_first_spawn.run(&spec.module_id);
2965        match spawned {
2966            Ok(child) => {
2967                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2968                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2969            }
2970            Err(err) => {
2971                error!(
2972                    module_id = %spec.module_id,
2973                    program = %spec.program.display(),
2974                    error = %err,
2975                    "configured module failed to spawn; marking failed and continuing"
2976                );
2977                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2978                Ok(self.supervised_module(spec, runtime, snapshot, None))
2979            }
2980        }
2981    }
2982
2983    /// Supervise a configured module with its own health, drain, and crash
2984    /// budget. The restart policy is per-module because the config file is:
2985    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2986    /// module that is expensive to restart should not be forced onto the same
2987    /// budget as one that is cheap.
2988    pub fn supervise_configured_with_health(
2989        &self,
2990        spec: ModuleSpec,
2991        enabled: bool,
2992        health: HealthConfig,
2993        drain_timeout_ms: Option<u64>,
2994        restart_policy: RestartPolicy,
2995    ) -> Result<SupervisedModule, SuperviseError> {
2996        validate_spec(&spec)?;
2997        self.establish_identity(&spec);
2998
2999        let mut runtime = self.runtime_config();
3000        runtime.health = health.clone();
3001        runtime.restart_policy = restart_policy;
3002        if let Some(ms) = drain_timeout_ms {
3003            runtime.drain_timeout = Duration::from_millis(ms);
3004            *runtime
3005                .effective_drain_timeout
3006                .lock()
3007                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
3008        }
3009        if !enabled {
3010            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
3011            return Ok(self.supervised_module(spec, runtime, snapshot, None));
3012        }
3013
3014        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
3015        let spawned = spawn_child(
3016            &spec,
3017            runtime.connection_file_path.as_deref(),
3018            self.supervisor_handle.as_ref(),
3019            &runtime.stderr_ring,
3020            runtime.capture_logs_dir.as_deref(),
3021            &runtime.child_roster,
3022            #[cfg(target_os = "linux")]
3023            runtime.cgroup_placement.as_ref(),
3024        );
3025        #[cfg(test)]
3026        self.test_after_first_spawn.run(&spec.module_id);
3027        match spawned {
3028            Ok(child) => {
3029                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3030                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3031            }
3032            Err(err) => {
3033                if health.critical {
3034                    error!(
3035                        module_id = %spec.module_id,
3036                        program = %spec.program.display(),
3037                        error = %err,
3038                        "critical configured module failed to spawn; marking failed and alerting"
3039                    );
3040                } else {
3041                    error!(
3042                        module_id = %spec.module_id,
3043                        program = %spec.program.display(),
3044                        error = %err,
3045                        "configured module failed to spawn; marking failed and continuing"
3046                    );
3047                }
3048                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3049                Ok(self.supervised_module(spec, runtime, snapshot, None))
3050            }
3051        }
3052    }
3053
3054    fn runtime_config(&self) -> SupervisorRuntimeConfig {
3055        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3056        SupervisorRuntimeConfig {
3057            scheduled_respawn: Arc::default(),
3058            deferred_reload_reply: Arc::default(),
3059            restart_policy: self.restart_policy,
3060            drain_timeout: self.drain_timeout,
3061            // Shared with this module's roster copy: daemon shutdown waits on
3062            // each child for the module's own drain budget, as resolved now.
3063            child_roster: self
3064                .child_roster
3065                .for_module(Arc::clone(&effective_drain_timeout)),
3066            effective_drain_timeout,
3067            default_drain_timeout: self.drain_timeout,
3068            health: self.health.clone(),
3069            connection_file_path: self.connection_file_path.clone(),
3070            capture_logs_dir: self.capture_logs_dir.clone(),
3071            forwarding: self.forwarding.clone(),
3072            supervisor_handle: self.supervisor_handle.clone(),
3073            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3074            terminal_ring: Arc::new(Mutex::new(
3075                TerminalRing::new(
3076                    TerminalRingConfig::default(),
3077                    self.daemon_start_clock.started_at_ms(),
3078                )
3079                .with_start_clock(self.daemon_start_clock)
3080                .with_journal(self.terminal_journal.clone())
3081                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3082            )),
3083            spawn_events: self.spawn_events.clone(),
3084            #[cfg(target_os = "linux")]
3085            cgroup_placement: self.cgroup_placement.clone(),
3086            #[cfg(test)]
3087            test_seed_stale_facts_before_enable_spawn: false,
3088            #[cfg(test)]
3089            test_reload_exit_record_gate: None,
3090        }
3091    }
3092
3093    fn supervised_module(
3094        &self,
3095        spec: ModuleSpec,
3096        runtime: SupervisorRuntimeConfig,
3097        snapshot: SharedSnapshot,
3098        child: Option<SupervisedChild>,
3099    ) -> SupervisedModule {
3100        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3101            spec: spec.clone(),
3102            health: runtime.health.clone(),
3103        }));
3104        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3105        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3106        // The module's OWN policy, which may be its per-module config rather than
3107        // the supervisor-wide one; status must report the budget the supervise
3108        // loop actually enforces.
3109        let restart_policy = runtime.restart_policy;
3110        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3111        let (tx, rx) = mpsc::channel(4);
3112        let monitor = tokio::spawn(supervise_loop(
3113            spec.clone(),
3114            runtime,
3115            Arc::clone(&self.registry),
3116            Arc::clone(&self.process_liveness),
3117            Arc::clone(&snapshot),
3118            child,
3119            rx,
3120        ));
3121
3122        let module_id = spec.module_id.clone();
3123        let module = SupervisedModule {
3124            inner: Arc::new(SupervisedModuleInner {
3125                module_id: module_id.clone(),
3126                registry: Arc::clone(&self.registry),
3127                snapshot,
3128                configuration,
3129                stderr_ring,
3130                terminal_ring,
3131                commands: tx,
3132                monitor: Mutex::new(Some(monitor)),
3133                restart_policy,
3134                effective_drain_timeout,
3135                provenance_probe: self.provenance_probe.clone(),
3136            }),
3137        };
3138        // The identity gates and the configured mark were set by
3139        // `establish_identity` before any process was spawned; only the roster
3140        // entry waits for the module handle, which needs the spawned child.
3141        if let Some(supervisor_handle) = &self.supervisor_handle {
3142            supervisor_handle.insert(module.clone());
3143        }
3144        module
3145    }
3146}
3147
3148impl Default for Supervisor {
3149    fn default() -> Self {
3150        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3151    }
3152}
3153
3154/// Handle to one supervised child process.
3155#[derive(Clone)]
3156pub struct SupervisedModule {
3157    inner: Arc<SupervisedModuleInner>,
3158}
3159
3160struct SupervisedModuleInner {
3161    module_id: String,
3162    registry: Arc<Registry>,
3163    snapshot: SharedSnapshot,
3164    configuration: Arc<Mutex<SupervisedConfiguration>>,
3165    stderr_ring: Arc<Mutex<StderrRing>>,
3166    terminal_ring: Arc<Mutex<TerminalRing>>,
3167    commands: mpsc::Sender<SupervisorCommand>,
3168    monitor: Mutex<Option<JoinHandle<()>>>,
3169    /// Copied from the supervisor's runtime config at spawn so `status()` can
3170    /// report the restart budget without reaching back into the supervisor. The
3171    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3172    restart_policy: RestartPolicy,
3173    effective_drain_timeout: Arc<Mutex<Duration>>,
3174    provenance_probe: ExecutableIdentityProbe,
3175}
3176
3177impl fmt::Debug for SupervisedModule {
3178    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3179        f.debug_struct("SupervisedModule")
3180            .field("module_id", &self.inner.module_id)
3181            .field("status", &self.status())
3182            .finish_non_exhaustive()
3183    }
3184}
3185
3186impl SupervisedModule {
3187    pub fn module_id(&self) -> &str {
3188        &self.inner.module_id
3189    }
3190
3191    /// Test-only: put one probe miss on the streak, the way
3192    /// `handle_health_probe_failure` does, so tests can assert what a later
3193    /// event does to the streak without driving the whole probe loop.
3194    #[cfg(test)]
3195    pub(crate) fn record_health_probe_failure_for_test(
3196        &self,
3197        detail: &str,
3198    ) -> Result<(), SuperviseError> {
3199        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3200            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3201            state.health.detail = Some(detail.to_string());
3202        })
3203    }
3204
3205    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3206        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3207    }
3208
3209    /// The module's retained stderr, newest lines last.
3210    ///
3211    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3212    /// module, `supervisor.list` renders every module, and putting it in the
3213    /// shared snapshot would make each status read carry a payload almost nobody
3214    /// asked for. Callers that want the text ask for it.
3215    pub fn stderr_tail(
3216        &self,
3217        max_lines: Option<usize>,
3218        max_bytes: Option<usize>,
3219    ) -> StderrTailSnapshot {
3220        self.inner
3221            .stderr_ring
3222            .lock()
3223            .unwrap_or_else(|poisoned| poisoned.into_inner())
3224            .snapshot(max_lines, max_bytes)
3225    }
3226
3227    /// The module's bounded terminal history, oldest retained exit first.
3228    ///
3229    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3230    /// daemon whose in-memory history was necessarily reset.
3231    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3232        self.inner
3233            .terminal_ring
3234            .lock()
3235            .unwrap_or_else(|poisoned| poisoned.into_inner())
3236            .snapshot()
3237    }
3238
3239    /// Retained observations from the current ring and all journal generations.
3240    ///
3241    /// Blocking: this reads the journal files. Async callers use
3242    /// [`Self::read_durable_terminal_history`].
3243    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3244        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3245    }
3246
3247    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3248    /// read (up to every retained generation) never occupies a runtime worker.
3249    /// Fails only if the blocking task could not finish (runtime shutdown or a
3250    /// panic in the read).
3251    pub(crate) async fn read_durable_terminal_history(
3252        &self,
3253    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3254        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3255        let module_id = self.inner.module_id.clone();
3256        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3257            .await
3258    }
3259
3260    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3261        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3262            .map(|(status, _)| status)
3263    }
3264
3265    pub(crate) fn record_deliberate_severance(
3266        &self,
3267        identity: ProcessIdentity,
3268    ) -> Result<bool, SuperviseError> {
3269        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3270        if snapshot.pid != Some(identity.pid)
3271            || snapshot.process_start_time != Some(identity.start_time)
3272        {
3273            return Ok(false);
3274        }
3275        snapshot.deliberate_severance = Some(identity);
3276        Ok(true)
3277    }
3278
3279    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3280    ///
3281    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3282    /// does not produce reader-observability logs.
3283    pub(crate) fn status_for_control(
3284        &self,
3285        caller: &'static str,
3286    ) -> Result<ModuleStatus, SuperviseError> {
3287        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3288            .map(|(status, _)| status)
3289    }
3290
3291    fn status_with_snapshot_lock(
3292        &self,
3293        snapshot: &SharedSnapshot,
3294        caller: Option<&'static str>,
3295    ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3296        let mut guard = match caller {
3297            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3298            None => lock_snapshot(snapshot)?,
3299        };
3300        // Read the budget through the pruning path so a reader sees the same
3301        // in-window count the restart decision would use, not a stale total.
3302        let restart_count =
3303            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3304        let snapshot = guard.clone();
3305        drop(guard);
3306        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3307            SuperviseError::StatePoisoned {
3308                module_id: Some(self.inner.module_id.clone()),
3309            }
3310        })?;
3311        let registration_active = self
3312            .inner
3313            .registry
3314            .get_module(&self.inner.module_id)
3315            .map_err(SuperviseError::Registry)?
3316            .is_some();
3317        let protocol = snapshot
3318            .spawned_protocol
3319            .unwrap_or(self.declared_protocol()?);
3320        let running_process =
3321            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3322        // Registration is the difference between the two protocols and the only
3323        // one: a subc module that has not registered cannot serve a request even
3324        // though its process is up, and a `none` module never registers at all,
3325        // so requiring it there would pin `live` to false for the whole life of
3326        // a perfectly healthy process.
3327        let live = match protocol {
3328            ModuleProtocol::Subc => running_process && registration_active,
3329            ModuleProtocol::None => running_process,
3330        };
3331
3332        Ok((
3333            ModuleStatus {
3334                module_id: self.inner.module_id.clone(),
3335                state: snapshot.state,
3336                enabled: snapshot.enabled,
3337                process_alive: snapshot.process_alive,
3338                registration_active,
3339                protocol,
3340                live,
3341                restart_count,
3342                lifetime_restarts: snapshot.lifetime_restarts,
3343                spawn_generation: snapshot.spawn_generation,
3344                max_restarts: self.inner.restart_policy.max_restarts,
3345                restart_window: self.inner.restart_policy.window,
3346                drain_timeout,
3347                restart_backoff: self.inner.restart_policy.backoff,
3348                restart_max_backoff: self.inner.restart_policy.max_backoff,
3349                pid: snapshot.reported_pid(),
3350                spawned_at_ms: snapshot.spawned_at_ms,
3351                spawned_from: snapshot.spawned_from,
3352                process_start_time: snapshot.process_start_time,
3353                last_exit: snapshot.last_exit,
3354                health: snapshot.health,
3355            },
3356            snapshot.spawned_file_identity,
3357        ))
3358    }
3359
3360    #[cfg(test)]
3361    pub(crate) fn hold_snapshot_for_test(
3362        &self,
3363        acquired: std::sync::mpsc::Sender<()>,
3364        hold: Duration,
3365    ) -> std::thread::JoinHandle<()> {
3366        let snapshot = Arc::clone(&self.inner.snapshot);
3367        std::thread::spawn(move || {
3368            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3369            acquired
3370                .send(())
3371                .expect("test receiver waits for snapshot lock");
3372            std::thread::sleep(hold);
3373        })
3374    }
3375
3376    /// The status and the running-image check for `supervisor.provenance`,
3377    /// taken from one status read. The exec acknowledgement can land between
3378    /// two separate reads, and the reply would then pair "no pid yet" with an
3379    /// image observed after the module started, which describes no single
3380    /// moment.
3381    pub(crate) async fn status_and_running_image_agreement(
3382        &self,
3383    ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3384        let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3385        let image = self
3386            .inner
3387            .provenance_probe
3388            .observe(
3389                status.pid,
3390                status.spawned_from.as_deref(),
3391                identity,
3392                status.process_start_time,
3393            )
3394            .await;
3395        Ok((status, image))
3396    }
3397
3398    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3399        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3400            Ok(snapshot) => snapshot.clone(),
3401            Err(_) => {
3402                return subc_control::RunningImageAgreement::Unavailable {
3403                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3404                };
3405            }
3406        };
3407        self.inner
3408            .provenance_probe
3409            .observe(
3410                snapshot.reported_pid(),
3411                snapshot.spawned_from.as_deref(),
3412                snapshot.spawned_file_identity,
3413                snapshot.process_start_time,
3414            )
3415            .await
3416    }
3417
3418    /// Memory and CPU time of the module's current process, read now. Only the
3419    /// process the supervisor spawned is read, not processes it has started.
3420    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3421        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3422            Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3423            Err(_) => {
3424                return subc_control::ChildResourceUsage::Unavailable {
3425                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3426                }
3427            }
3428        };
3429        crate::child_resources::read(pid, start_time)
3430    }
3431
3432    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3433        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3434        Ok(match snapshot.state {
3435            ModuleState::Restarting => true,
3436            ModuleState::Failed | ModuleState::Disabled => false,
3437            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3438        })
3439    }
3440
3441    #[cfg(test)]
3442    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3443        self.is_warming_with_snapshot_lock(None)
3444    }
3445
3446    pub(crate) fn is_warming_for_control(
3447        &self,
3448        caller: &'static str,
3449    ) -> Result<bool, SuperviseError> {
3450        self.is_warming_with_snapshot_lock(Some(caller))
3451    }
3452
3453    fn is_warming_with_snapshot_lock(
3454        &self,
3455        caller: Option<&'static str>,
3456    ) -> Result<bool, SuperviseError> {
3457        let snapshot = match caller {
3458            Some(caller) => {
3459                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3460            }
3461            None => lock_snapshot(&self.inner.snapshot)?,
3462        }
3463        .clone();
3464        Ok(matches!(
3465            snapshot.state,
3466            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3467        ))
3468    }
3469
3470    /// Drain the module and stop monitoring it.
3471    pub async fn drain(&self) -> Result<(), SuperviseError> {
3472        self.stop().await
3473    }
3474
3475    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3476        match self.state()? {
3477            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3478            ModuleState::Starting
3479            | ModuleState::Running
3480            | ModuleState::Unresponsive
3481            | ModuleState::Restarting
3482            | ModuleState::Draining
3483            | ModuleState::Disabled => {}
3484        }
3485
3486        let (reply_tx, reply_rx) = oneshot::channel();
3487        self.inner
3488            .commands
3489            .send(SupervisorCommand::Retire { reply: reply_tx })
3490            .await
3491            .map_err(|_| SuperviseError::CommandClosed {
3492                module_id: self.inner.module_id.clone(),
3493            })?;
3494        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3495            module_id: self.inner.module_id.clone(),
3496        })?
3497    }
3498
3499    pub async fn stop(&self) -> Result<(), SuperviseError> {
3500        match self.state()? {
3501            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3502            ModuleState::Starting
3503            | ModuleState::Running
3504            | ModuleState::Unresponsive
3505            | ModuleState::Restarting
3506            | ModuleState::Draining
3507            | ModuleState::Disabled => {}
3508        }
3509
3510        let (reply_tx, reply_rx) = oneshot::channel();
3511        self.inner
3512            .commands
3513            .send(SupervisorCommand::Drain { reply: reply_tx })
3514            .await
3515            .map_err(|_| SuperviseError::CommandClosed {
3516                module_id: self.inner.module_id.clone(),
3517            })?;
3518        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3519            module_id: self.inner.module_id.clone(),
3520        })?
3521    }
3522
3523    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3524        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3525        let (reply_tx, reply_rx) = oneshot::channel();
3526        self.inner
3527            .commands
3528            .send(SupervisorCommand::Restart {
3529                drain_timeout_ms,
3530                received_at_generation,
3531                queued_at: Instant::now(),
3532                reply: reply_tx,
3533            })
3534            .await
3535            .map_err(|_| SuperviseError::CommandClosed {
3536                module_id: self.inner.module_id.clone(),
3537            })?;
3538        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3539            module_id: self.inner.module_id.clone(),
3540        })?
3541    }
3542
3543    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3544    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3545    /// process then drains in the background of the supervise loop) or has
3546    /// failed, leaving the old process serving.
3547    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3548        let (reply_tx, reply_rx) = oneshot::channel();
3549        self.inner
3550            .commands
3551            .send(SupervisorCommand::Swap {
3552                ready_timeout,
3553                reply: reply_tx,
3554            })
3555            .await
3556            .map_err(|_| SuperviseError::CommandClosed {
3557                module_id: self.inner.module_id.clone(),
3558            })?;
3559        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3560            module_id: self.inner.module_id.clone(),
3561        })?
3562    }
3563
3564    pub async fn reload(&self) -> Result<(), SuperviseError> {
3565        let (reply_tx, reply_rx) = oneshot::channel();
3566        self.inner
3567            .commands
3568            .send(SupervisorCommand::Reload { reply: reply_tx })
3569            .await
3570            .map_err(|_| SuperviseError::CommandClosed {
3571                module_id: self.inner.module_id.clone(),
3572            })?;
3573        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3574            module_id: self.inner.module_id.clone(),
3575        })?
3576    }
3577
3578    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3579        let (reply_tx, reply_rx) = oneshot::channel();
3580        self.inner
3581            .commands
3582            .send(SupervisorCommand::SetEnabled {
3583                enabled,
3584                reply: reply_tx,
3585            })
3586            .await
3587            .map_err(|_| SuperviseError::CommandClosed {
3588                module_id: self.inner.module_id.clone(),
3589            })?;
3590        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3591            module_id: self.inner.module_id.clone(),
3592        })?
3593    }
3594
3595    /// The current process's protocol, or the configured protocol when down.
3596    /// A rescan stores the next launch spec without changing how an existing
3597    /// process registers, serves routes, is probed, or exits.
3598    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3599        let configured = self
3600            .inner
3601            .configuration
3602            .lock()
3603            .map_err(|_| SuperviseError::StatePoisoned {
3604                module_id: Some(self.inner.module_id.clone()),
3605            })?
3606            .spec
3607            .protocol;
3608        let state = lock_snapshot(&self.inner.snapshot)?;
3609        Ok(state.spawned_protocol.unwrap_or(configured))
3610    }
3611
3612    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3613        let configuration =
3614            self.inner
3615                .configuration
3616                .lock()
3617                .map_err(|_| SuperviseError::StatePoisoned {
3618                    module_id: Some(self.inner.module_id.clone()),
3619                })?;
3620        Ok((configuration.spec.clone(), configuration.health.clone()))
3621    }
3622
3623    /// Replace this module's launch spec, keeping its health and drain policy,
3624    /// the way a rescan does for a changed config entry. The running process is
3625    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3626    #[cfg(any(test, feature = "test-support"))]
3627    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3628        let (_, health) = self.configuration()?;
3629        let drain_timeout_ms = u64::try_from(
3630            self.inner
3631                .effective_drain_timeout
3632                .lock()
3633                .unwrap_or_else(|poisoned| poisoned.into_inner())
3634                .as_millis(),
3635        )
3636        .ok();
3637        self.update_configuration(spec, health, drain_timeout_ms)
3638            .await
3639    }
3640
3641    pub(crate) async fn update_configuration(
3642        &self,
3643        spec: ModuleSpec,
3644        health: HealthConfig,
3645        drain_timeout_ms: Option<u64>,
3646    ) -> Result<(), SuperviseError> {
3647        if spec.module_id != self.inner.module_id {
3648            return Err(SuperviseError::InvalidSpec {
3649                reason: "a supervised module's module_id cannot be changed".to_string(),
3650            });
3651        }
3652        validate_spec(&spec)?;
3653        let (reply_tx, reply_rx) = oneshot::channel();
3654        self.inner
3655            .commands
3656            .send(SupervisorCommand::UpdateConfiguration {
3657                spec: spec.clone(),
3658                health: health.clone(),
3659                drain_timeout_ms,
3660                reply: reply_tx,
3661            })
3662            .await
3663            .map_err(|_| SuperviseError::CommandClosed {
3664                module_id: self.inner.module_id.clone(),
3665            })?;
3666        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3667            module_id: self.inner.module_id.clone(),
3668        })?;
3669        let mut configuration =
3670            self.inner
3671                .configuration
3672                .lock()
3673                .map_err(|_| SuperviseError::StatePoisoned {
3674                    module_id: Some(self.inner.module_id.clone()),
3675                })?;
3676        configuration.spec = spec;
3677        configuration.health = health;
3678        Ok(())
3679    }
3680}
3681
3682impl Drop for SupervisedModuleInner {
3683    fn drop(&mut self) {
3684        let Ok(mut monitor) = self.monitor.lock() else {
3685            return;
3686        };
3687        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3688            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3689                state.state = ModuleState::Stopped;
3690                clear_current_process_facts(state);
3691            });
3692            monitor.abort();
3693        }
3694        let _ = monitor.take();
3695    }
3696}
3697
3698#[derive(Debug)]
3699enum SupervisorCommand {
3700    Drain {
3701        reply: oneshot::Sender<Result<(), SuperviseError>>,
3702    },
3703    Retire {
3704        reply: oneshot::Sender<Result<(), SuperviseError>>,
3705    },
3706    Restart {
3707        /// Operator override for this one restart's drain budget, in ms. `None`
3708        /// uses the module's configured/default budget; `Some(0)` cuts
3709        /// immediately (wedge bounce: a stuck request never settles, so
3710        /// waiting only delays recovery).
3711        drain_timeout_ms: Option<u64>,
3712        /// The module's `spawn_generation` when the request was received, before
3713        /// it waited in the command queue. A queued restart whose module has
3714        /// since spawned a newer process is already satisfied (see the handler).
3715        received_at_generation: u64,
3716        /// When the request entered the command queue, so the handler can log
3717        /// how long it waited behind the loop's other work.
3718        queued_at: Instant,
3719        reply: oneshot::Sender<Result<(), SuperviseError>>,
3720    },
3721    Reload {
3722        reply: oneshot::Sender<Result<(), SuperviseError>>,
3723    },
3724    SetEnabled {
3725        enabled: bool,
3726        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3727    },
3728    UpdateConfiguration {
3729        spec: ModuleSpec,
3730        health: HealthConfig,
3731        /// Per-module drain override from the new config; `None` re-resolves to
3732        /// the supervisor-wide default.
3733        drain_timeout_ms: Option<u64>,
3734        reply: oneshot::Sender<()>,
3735    },
3736    Swap {
3737        /// How long the candidate may take to register and declare itself
3738        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3739        ready_timeout: Option<Duration>,
3740        /// Answered at cutover or failure; the incumbent's drain follows.
3741        reply: oneshot::Sender<Result<(), SuperviseError>>,
3742    },
3743}
3744
3745#[derive(Debug)]
3746pub enum SuperviseError {
3747    InvalidSpec {
3748        reason: String,
3749    },
3750    Spawn {
3751        program: PathBuf,
3752        source: io::Error,
3753        cgroup_path: Option<PathBuf>,
3754    },
3755    Cgroup {
3756        module_id: String,
3757        source: io::Error,
3758    },
3759    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3760    /// than spawn a reserved module without its identity binding.
3761    LaunchNonce {
3762        reason: String,
3763    },
3764    Wait {
3765        module_id: String,
3766        source: io::Error,
3767    },
3768    Kill {
3769        module_id: String,
3770        source: io::Error,
3771    },
3772    Forwarding(ForwardingError),
3773    Registry(RegistryError),
3774    ReloadUnavailable {
3775        module_id: String,
3776        reason: String,
3777    },
3778    /// An operator restart/reload was requested for a module that is currently
3779    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3780    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3781    /// by a restart, so these commands are rejected instead of re-enabling it.
3782    Disabled {
3783        module_id: String,
3784    },
3785    ReloadFailed {
3786        module_id: String,
3787        reason: String,
3788    },
3789    RegistrationStillActive {
3790        module_id: String,
3791        waited: Duration,
3792    },
3793    StatePoisoned {
3794        module_id: Option<String>,
3795    },
3796    CommandClosed {
3797        module_id: String,
3798    },
3799    /// A restart or reload arrived while a swap's candidate was warming. The
3800    /// swap owns the module until it cuts over or fails; a stop or disable
3801    /// would have aborted it instead.
3802    SwapInProgress {
3803        module_id: String,
3804    },
3805    /// A swap was refused before anything was spawned.
3806    SwapRefused {
3807        module_id: String,
3808        reason: SwapRefusal,
3809    },
3810    /// A swap spawned a candidate and gave up on it. The candidate has been
3811    /// killed and its slot freed; the incumbent was left serving and was never
3812    /// drained, except in the one `CutoverLost` case described on that arm.
3813    SwapFailed {
3814        module_id: String,
3815        arm: SwapFailureArm,
3816        detail: String,
3817        /// How the candidate exited, when it exited on its own before the
3818        /// supervisor gave up on it.
3819        candidate_exit: Option<ExitReport>,
3820    },
3821}
3822
3823/// Why a swap was refused before a candidate was spawned.
3824#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3825pub enum SwapRefusal {
3826    /// The module's config does not declare `overlap: "safe"`.
3827    OverlapExclusive,
3828    /// The module is not registered, so there is no incumbent to keep serving
3829    /// and nothing a swap would improve on; a plain restart is the tool.
3830    NotRegistered,
3831    /// The module does not speak the subc wire, so a candidate could never
3832    /// register or declare itself ready.
3833    ProtocolNone,
3834    /// The supervisor lacks the forwarding table (to cut routes over) or the
3835    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3836    NotConfigured,
3837    /// A swap is already open for this module.
3838    AlreadySwapping,
3839}
3840
3841impl SwapRefusal {
3842    pub fn as_str(self) -> &'static str {
3843        match self {
3844            Self::OverlapExclusive => "overlap_exclusive",
3845            Self::NotRegistered => "not_registered",
3846            Self::ProtocolNone => "protocol_none",
3847            Self::NotConfigured => "not_configured",
3848            Self::AlreadySwapping => "already_swapping",
3849        }
3850    }
3851}
3852
3853/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3854/// serving and undrained; see `CutoverLost`.
3855#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3856pub enum SwapFailureArm {
3857    /// The candidate process could not be started.
3858    SpawnFailed,
3859    /// The candidate did not register within the readiness budget.
3860    NeverRegistered,
3861    /// The candidate registered but did not declare itself ready in time.
3862    NeverReady,
3863    /// The candidate exited before cutover.
3864    CandidateExited,
3865    /// The candidate declared itself ready but failed its health probe.
3866    CandidateUnhealthy,
3867    /// An operator stop, disable or retire arrived while the candidate warmed.
3868    /// The candidate was killed and the operator's command then carried out on
3869    /// the incumbent.
3870    Interrupted,
3871    /// The candidate's connection closed at the moment of cutover. If it
3872    /// closed before forwarding moved, the incumbent is untouched. If it closed
3873    /// between the forwarding and registry halves of cutover, forwarding can no
3874    /// longer route to the incumbent, so the module is restarted plainly.
3875    CutoverLost,
3876}
3877
3878impl SwapFailureArm {
3879    pub fn as_str(self) -> &'static str {
3880        match self {
3881            Self::SpawnFailed => "spawn_failed",
3882            Self::NeverRegistered => "never_registered",
3883            Self::NeverReady => "never_ready",
3884            Self::CandidateExited => "candidate_exited",
3885            Self::CandidateUnhealthy => "candidate_unhealthy",
3886            Self::Interrupted => "interrupted",
3887            Self::CutoverLost => "cutover_lost",
3888        }
3889    }
3890}
3891
3892impl fmt::Display for SuperviseError {
3893    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3894        match self {
3895            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3896            Self::Spawn {
3897                program,
3898                source,
3899                cgroup_path: Some(cgroup_path),
3900            } => write!(
3901                f,
3902                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3903                cgroup_path.display(),
3904                program.display()
3905            ),
3906            Self::Spawn {
3907                program,
3908                source,
3909                cgroup_path: None,
3910            } => write!(
3911                f,
3912                "failed to spawn module '{}': {source}",
3913                program.display()
3914            ),
3915            Self::Cgroup { module_id, source } => {
3916                write!(
3917                    f,
3918                    "failed to prepare cgroup for module '{module_id}': {source}"
3919                )
3920            }
3921            Self::LaunchNonce { reason } => {
3922                write!(
3923                    f,
3924                    "failed to generate reserved-module launch nonce: {reason}"
3925                )
3926            }
3927            Self::Wait { module_id, source } => {
3928                write!(f, "failed to wait for module '{module_id}': {source}")
3929            }
3930            Self::Kill { module_id, source } => {
3931                write!(f, "failed to kill module '{module_id}': {source}")
3932            }
3933            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3934            Self::Registry(err) => write!(f, "registry error: {err}"),
3935            Self::ReloadUnavailable { module_id, reason } => {
3936                write!(f, "reload unavailable for module '{module_id}': {reason}")
3937            }
3938            Self::Disabled { module_id } => {
3939                write!(
3940                    f,
3941                    "module '{module_id}' is disabled; enable it before restart or reload"
3942                )
3943            }
3944            Self::ReloadFailed { module_id, reason } => {
3945                write!(f, "reload failed for module '{module_id}': {reason}")
3946            }
3947            Self::RegistrationStillActive { module_id, waited } => write!(
3948                f,
3949                "module '{module_id}' registration remained active after waiting {waited:?}"
3950            ),
3951            Self::StatePoisoned { module_id } => match module_id {
3952                Some(module_id) => {
3953                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3954                }
3955                None => write!(f, "supervisor state was poisoned"),
3956            },
3957            Self::CommandClosed { module_id } => {
3958                write!(
3959                    f,
3960                    "supervisor command channel for module '{module_id}' is closed"
3961                )
3962            }
3963            Self::SwapInProgress { module_id } => write!(
3964                f,
3965                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3966            ),
3967            Self::SwapRefused { module_id, reason } => match reason {
3968                SwapRefusal::OverlapExclusive => write!(
3969                    f,
3970                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3971                ),
3972                SwapRefusal::NotRegistered => write!(
3973                    f,
3974                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3975                ),
3976                SwapRefusal::ProtocolNone => write!(
3977                    f,
3978                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3979                ),
3980                SwapRefusal::NotConfigured => write!(
3981                    f,
3982                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3983                ),
3984                SwapRefusal::AlreadySwapping => {
3985                    write!(f, "module '{module_id}' is already being swapped")
3986                }
3987            },
3988            Self::SwapFailed {
3989                module_id,
3990                arm,
3991                detail,
3992                ..
3993            } => write!(
3994                f,
3995                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3996                arm.as_str()
3997            ),
3998        }
3999    }
4000}
4001
4002impl Error for SuperviseError {
4003    fn source(&self) -> Option<&(dyn Error + 'static)> {
4004        match self {
4005            Self::Spawn { source, .. }
4006            | Self::Cgroup { source, .. }
4007            | Self::Wait { source, .. }
4008            | Self::Kill { source, .. } => Some(source),
4009            Self::Forwarding(err) => Some(err),
4010            Self::Registry(err) => Some(err),
4011            Self::LaunchNonce { .. }
4012            | Self::InvalidSpec { .. }
4013            | Self::ReloadUnavailable { .. }
4014            | Self::Disabled { .. }
4015            | Self::ReloadFailed { .. }
4016            | Self::RegistrationStillActive { .. }
4017            | Self::StatePoisoned { .. }
4018            | Self::CommandClosed { .. }
4019            | Self::SwapInProgress { .. }
4020            | Self::SwapRefused { .. }
4021            | Self::SwapFailed { .. } => None,
4022        }
4023    }
4024}
4025
4026pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4027    if spec.module_id.trim().is_empty() {
4028        return Err(SuperviseError::InvalidSpec {
4029            reason: "module_id must not be empty".to_string(),
4030        });
4031    }
4032
4033    Ok(())
4034}
4035
4036#[derive(Debug, Default)]
4037struct HealthProbeRuntime {
4038    configured_health: Option<HealthConfig>,
4039    registered_connection: Option<crate::ConnectionId>,
4040    advertised: bool,
4041    next_probe_at: Option<Instant>,
4042    probe_index: u64,
4043}
4044
4045fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4046    lock_snapshot(snapshot)
4047        .ok()
4048        .and_then(|state| state.spawned_protocol)
4049        .unwrap_or(spec.protocol)
4050}
4051
4052impl HealthProbeRuntime {
4053    fn refresh_registration(
4054        &mut self,
4055        spec: &ModuleSpec,
4056        runtime: &SupervisorRuntimeConfig,
4057        registry: &Registry,
4058        snapshot: &SharedSnapshot,
4059    ) {
4060        if self.configured_health.as_ref() != Some(&runtime.health) {
4061            self.configured_health = Some(runtime.health.clone());
4062            self.next_probe_at = None;
4063            self.registered_connection = None;
4064            self.probe_index = 0;
4065        }
4066        // A non-wire process never registers. Only an explicitly configured
4067        // HTTP endpoint can arm its health probe; an absent HELLO is not a
4068        // health failure for that kind of process.
4069        if running_protocol(spec, snapshot) == ModuleProtocol::None {
4070            self.registered_connection = None;
4071            self.advertised = runtime.health.http.is_some();
4072            if !self.advertised {
4073                self.next_probe_at = None;
4074                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4075                    let unknown = ModuleHealthStatus::default();
4076                    if state.health != unknown {
4077                        state.health = unknown;
4078                    }
4079                });
4080            } else if self.next_probe_at.is_none() {
4081                self.next_probe_at = Some(
4082                    Instant::now()
4083                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4084                );
4085            }
4086            return;
4087        }
4088
4089        let registration = match registry.get_module(&spec.module_id) {
4090            Ok(registration) => registration,
4091            Err(err) => {
4092                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4093                self.advertised = false;
4094                self.next_probe_at = None;
4095                return;
4096            }
4097        };
4098
4099        let Some(registration) = registration else {
4100            self.registered_connection = None;
4101            self.advertised = false;
4102            self.next_probe_at = None;
4103            return;
4104        };
4105
4106        let advertised = registration
4107            .control_ops
4108            .iter()
4109            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4110        if !advertised {
4111            self.registered_connection = Some(registration.connection_id);
4112            self.advertised = false;
4113            self.next_probe_at = None;
4114            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4115                if state.health.status != SupervisorHealthStatus::Unknown
4116                    || state.health.consecutive_failures != 0
4117                    || state.health.last_probe_ms.is_some()
4118                    || state.health.detail.is_some()
4119                    || state.health.metrics.is_some()
4120                {
4121                    state.health.status = SupervisorHealthStatus::Unknown;
4122                    state.health.consecutive_failures = 0;
4123                    state.health.last_probe_ms = None;
4124                    state.health.detail = None;
4125                    state.health.metrics = None;
4126                }
4127            });
4128            return;
4129        }
4130
4131        let reregistered = self.registered_connection != Some(registration.connection_id);
4132        self.registered_connection = Some(registration.connection_id);
4133        self.advertised = true;
4134        if reregistered || self.next_probe_at.is_none() {
4135            self.probe_index = 0;
4136            self.next_probe_at = Some(
4137                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4138            );
4139            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4140                state.health.status = SupervisorHealthStatus::Unknown;
4141                state.health.consecutive_failures = 0;
4142                state.health.detail = None;
4143                state.health.metrics = None;
4144            });
4145        }
4146    }
4147
4148    fn wake_after(&self) -> Option<Duration> {
4149        if !self.advertised {
4150            return None;
4151        }
4152        self.next_probe_at
4153            .map(|next| next.saturating_duration_since(Instant::now()))
4154    }
4155
4156    fn due(&self) -> bool {
4157        self.advertised
4158            && self
4159                .next_probe_at
4160                .is_some_and(|next| Instant::now() >= next)
4161    }
4162
4163    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4164        self.probe_index = self.probe_index.wrapping_add(1);
4165        self.next_probe_at = Some(
4166            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4167        );
4168    }
4169}
4170
4171/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4172///
4173/// This was a struct with a single `message: String`, and every one of the
4174/// fifteen construction sites collapsed into it. Each site knows exactly what it
4175/// saw -- the lane is gone, the module did not answer in time, the module
4176/// answered with the wrong thing -- and `handle_health_probe_failure` then
4177/// treated all of them identically: increment a counter, compare to a threshold,
4178/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4179/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4180///
4181/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4182///
4183/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4184///   answer on it again.
4185/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4186///   AND with a perfectly healthy one that lost a CPU race -- which is what
4187///   happens under machine load, and is how this supervisor killed a healthy
4188///   module three times in one day.
4189/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4190///   Restarting on it is defensible, but it is not the silence case and should
4191///   never be counted as one.
4192/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4193///   anything, so it cannot be evidence about the module at all.
4194///
4195/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4196/// one that fires most often, and while every variant collapsed into one string
4197/// it carried the same weight as the strongest.
4198///
4199/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4200/// DESIGN and a reader stopping at it gets the build backwards: the restart
4201/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4202/// probes still increment the failure streak and drive escalation at the
4203/// threshold (see `is_proof_of_death` below for why that is deliberate and
4204/// what gates the change). Absence of evidence restarts modules today.
4205#[derive(Debug)]
4206enum HealthProbeEvidence {
4207    /// The module's control lane is gone. Proof of death.
4208    LaneDead,
4209    /// No reply within the deadline. Proves nothing about the module's state.
4210    NoAnswer,
4211    /// The module replied, but not with a usable health report. Proves it is alive.
4212    BadAnswer,
4213    /// The daemon could not ask. Says nothing about the module.
4214    Misconfigured,
4215}
4216
4217#[derive(Debug)]
4218struct HealthProbeError {
4219    evidence: HealthProbeEvidence,
4220    message: String,
4221}
4222
4223impl HealthProbeError {
4224    fn lane_dead(message: impl Into<String>) -> Self {
4225        Self::with(HealthProbeEvidence::LaneDead, message)
4226    }
4227
4228    fn no_answer(message: impl Into<String>) -> Self {
4229        Self::with(HealthProbeEvidence::NoAnswer, message)
4230    }
4231
4232    fn bad_answer(message: impl Into<String>) -> Self {
4233        Self::with(HealthProbeEvidence::BadAnswer, message)
4234    }
4235
4236    fn misconfigured(message: impl Into<String>) -> Self {
4237        Self::with(HealthProbeEvidence::Misconfigured, message)
4238    }
4239
4240    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4241        Self {
4242            evidence,
4243            message: message.into(),
4244        }
4245    }
4246
4247    /// Whether this observation is proof the module cannot serve.
4248    ///
4249    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4250    /// variant that fires under CPU starvation, and treating it as proof is the
4251    /// defect this enum exists to make impossible to reintroduce silently.
4252    ///
4253    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4254    /// to restart also needs a bound for the case it excludes -- a genuinely
4255    /// wedged module, alive but never answering -- and that bound must come from
4256    /// the distribution of real late-answer latencies, which nothing measures
4257    /// yet. Landing the classification first makes the later change a one-line
4258    /// decision against evidence that already exists, rather than two unproven
4259    /// changes at once.
4260    #[allow(dead_code)]
4261    fn is_proof_of_death(&self) -> bool {
4262        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4263    }
4264
4265    /// Short stable label for logs and the health snapshot.
4266    ///
4267    /// An operator reading `ck health` currently cannot tell "the module is gone"
4268    /// from "the module did not answer in five seconds", because both render as
4269    /// prose in the same field. These labels are what make the two
4270    /// distinguishable at a glance, and they are what a later restart-policy
4271    /// change will be argued from.
4272    fn label(&self) -> &'static str {
4273        match self.evidence {
4274            HealthProbeEvidence::LaneDead => "lane-dead",
4275            HealthProbeEvidence::NoAnswer => "no-answer",
4276            HealthProbeEvidence::BadAnswer => "bad-answer",
4277            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4278        }
4279    }
4280}
4281
4282impl fmt::Display for HealthProbeError {
4283    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4284        f.write_str(&self.message)
4285    }
4286}
4287
4288async fn run_health_probe_cycle(
4289    spec: &ModuleSpec,
4290    runtime: &SupervisorRuntimeConfig,
4291    registry: &Registry,
4292    process_liveness: &SupervisorProcessLiveness,
4293    snapshot: &SharedSnapshot,
4294    child: &mut Option<SupervisedChild>,
4295) {
4296    let now_ms = unix_ms_now();
4297    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4298        .then_some(runtime.health.http.as_deref())
4299        .flatten();
4300    let result = match http {
4301        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4302        None => probe_module_health(&spec.module_id, runtime, None).await,
4303    };
4304    match result {
4305        Ok(report) => {
4306            handle_health_report(
4307                spec,
4308                runtime,
4309                registry,
4310                process_liveness,
4311                snapshot,
4312                child,
4313                report,
4314                now_ms,
4315            )
4316            .await;
4317        }
4318        Err(err) => {
4319            if http.is_some() {
4320                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4321                    state.health.status = SupervisorHealthStatus::Failing;
4322                });
4323            }
4324            handle_health_probe_failure(
4325                spec,
4326                runtime,
4327                registry,
4328                process_liveness,
4329                snapshot,
4330                child,
4331                err,
4332                now_ms,
4333            )
4334            .await;
4335        }
4336    }
4337}
4338
4339pub(crate) struct HttpProbeTarget<'a> {
4340    address: std::net::SocketAddr,
4341    localhost: bool,
4342    authority: &'a str,
4343    path: String,
4344}
4345
4346/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4347/// or TLS. A URL cannot turn a local health check into an outbound connection.
4348pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4349    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4350        return Err("must not contain whitespace, controls, or a fragment".into());
4351    }
4352    let rest = url
4353        .strip_prefix("http://")
4354        .ok_or("must use plain http://")?;
4355    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4356    let (authority, suffix) = rest.split_at(split);
4357    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4358        ("::1", rest)
4359    } else {
4360        let split = authority.find(':').unwrap_or(authority.len());
4361        authority.split_at(split)
4362    };
4363    let ip: std::net::IpAddr = match host {
4364        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4365        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4366        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4367    };
4368    let port = if port.is_empty() {
4369        80
4370    } else {
4371        port.strip_prefix(':')
4372            .and_then(|p| p.parse::<u16>().ok())
4373            .filter(|p| *p > 0)
4374            .ok_or("must have a valid nonzero TCP port")?
4375    };
4376    let path = if suffix.is_empty() {
4377        "/".into()
4378    } else if suffix.starts_with('?') {
4379        format!("/{suffix}")
4380    } else {
4381        suffix.into()
4382    };
4383    Ok(HttpProbeTarget {
4384        address: std::net::SocketAddr::new(ip, port),
4385        localhost: host == "localhost",
4386        authority,
4387        path,
4388    })
4389}
4390
4391async fn probe_http_health(
4392    url: &str,
4393    deadline: Duration,
4394) -> Result<HealthReport, HealthProbeError> {
4395    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4396    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4397    // Keep partial diagnostics outside the timed future so cancellation does
4398    // not discard a status line or body bytes already received.
4399    let mut response_status = String::new();
4400    let mut body = Vec::new();
4401    let probe = async {
4402        // Resolve localhost ourselves so a hosts-file override cannot turn
4403        // this into an outbound request, while IPv6-only local servers work.
4404        let connection = match tokio::net::TcpStream::connect(target.address).await {
4405            Err(_) if target.localhost => {
4406                tokio::net::TcpStream::connect((
4407                    std::net::Ipv6Addr::LOCALHOST,
4408                    target.address.port(),
4409                ))
4410                .await
4411            }
4412            result => result,
4413        };
4414        let mut stream = connection.map_err(|error| {
4415            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4416        })?;
4417        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4418            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4419        let mut reader = BufReader::new(stream);
4420        let mut budget = 16 * 1024;
4421        let status = http_line(&mut reader, &mut budget).await?;
4422        let mut words = status.split_ascii_whitespace();
4423        let version = words.next();
4424        let code = words
4425            .next()
4426            .filter(|word| word.len() == 3)
4427            .and_then(|word| word.parse::<u16>().ok());
4428        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4429            || !code.is_some_and(|code| (100..600).contains(&code))
4430        {
4431            return Err(HealthProbeError::bad_answer(format!(
4432                "invalid HTTP status: {status}"
4433            )));
4434        }
4435        let code = code.expect("validated status code");
4436        response_status = status.clone();
4437        let mut length = None;
4438        let mut chunked = false;
4439        loop {
4440            let line = http_line(&mut reader, &mut budget).await?;
4441            if line.is_empty() {
4442                break;
4443            }
4444            if let Some((name, value)) = line.split_once(':') {
4445                if name.eq_ignore_ascii_case("content-length") {
4446                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4447                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4448                    })?);
4449                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4450                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4451                }
4452            }
4453        }
4454        if chunked {
4455            while body.len() < 200 {
4456                let line = http_line(&mut reader, &mut budget).await?;
4457                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4458                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4459                if size == 0 {
4460                    break;
4461                }
4462                let count = size.min((200 - body.len()) as u64) as usize;
4463                let start = body.len();
4464                (&mut reader)
4465                    .take(count as u64)
4466                    .read_to_end(&mut body)
4467                    .await
4468                    .map_err(|error| {
4469                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4470                    })?;
4471                if body.len() - start != count {
4472                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4473                }
4474                if size > count as u64 || body.len() == 200 {
4475                    break;
4476                }
4477                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4478                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4479                }
4480            }
4481        } else {
4482            reader
4483                .take(length.unwrap_or(200).min(200))
4484                .read_to_end(&mut body)
4485                .await
4486                .map_err(|error| {
4487                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4488                })?;
4489        }
4490        if (200..300).contains(&code) {
4491            Ok(HealthReport::ok())
4492        } else {
4493            Err(HealthProbeError::bad_answer(
4494                "HTTP health endpoint returned non-2xx",
4495            ))
4496        }
4497    };
4498    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4499        Err(HealthProbeError::no_answer(format!(
4500            "HTTP probe timed out after {deadline:?}"
4501        )))
4502    });
4503    if let Err(error) = &mut result {
4504        if !response_status.is_empty() {
4505            error.message = format!(
4506                "{}; {response_status}: {}",
4507                error.message,
4508                String::from_utf8_lossy(&body)
4509            );
4510        }
4511    }
4512    result
4513}
4514
4515async fn http_line(
4516    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4517    remaining: &mut usize,
4518) -> Result<String, HealthProbeError> {
4519    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4520    let mut line = Vec::new();
4521    (&mut *reader)
4522        .take(*remaining as u64)
4523        .read_until(b'\n', &mut line)
4524        .await
4525        .map_err(|error| {
4526            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4527        })?;
4528    *remaining -= line.len();
4529    if !line.ends_with(b"\r\n") {
4530        return Err(HealthProbeError::bad_answer(
4531            "HTTP headers are incomplete or exceed 16 KiB",
4532        ));
4533    }
4534    line.truncate(line.len() - 2);
4535    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4536}
4537
4538async fn probe_module_health(
4539    module_id: &str,
4540    runtime: &SupervisorRuntimeConfig,
4541    drain_deadline: Option<Instant>,
4542) -> Result<HealthReport, HealthProbeError> {
4543    let Some(forwarding) = runtime.forwarding.as_ref() else {
4544        return Err(HealthProbeError::misconfigured(
4545            "supervisor was not configured with a forwarding table",
4546        ));
4547    };
4548    let probe_started_at = Instant::now();
4549    let mut deadline = probe_started_at + runtime.health.deadline;
4550    if let Some(drain_deadline) = drain_deadline {
4551        deadline = deadline.min(drain_deadline);
4552    }
4553    let pending = if drain_deadline.is_some() {
4554        forwarding.begin_drain_health_probe_rpc_for(
4555            module_id,
4556            MODULE_CONTROL_OP_HEALTH_CHECK,
4557            probe_started_at,
4558            deadline,
4559        )
4560    } else {
4561        forwarding.begin_health_probe_rpc_for(
4562            module_id,
4563            MODULE_CONTROL_OP_HEALTH_CHECK,
4564            probe_started_at,
4565            deadline,
4566        )
4567    }
4568    .map_err(|err| {
4569        // The endpoint is not registered, so there is no live control lane to
4570        // ask. That is the module being absent, not slow.
4571        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4572    })?;
4573    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4574}
4575
4576/// [`probe_module_health`] for one endpoint rather than the id's active one.
4577///
4578/// A swap probes two processes that no by-id lookup reaches: its candidate
4579/// before cutover, and its superseded incumbent (for busy gauges) while the
4580/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4581/// bounds the by-id drain probe.
4582async fn probe_endpoint_health(
4583    endpoint: crate::ModuleEndpointId,
4584    runtime: &SupervisorRuntimeConfig,
4585    deadline_cap: Option<Instant>,
4586) -> Result<HealthReport, HealthProbeError> {
4587    let Some(forwarding) = runtime.forwarding.as_ref() else {
4588        return Err(HealthProbeError::misconfigured(
4589            "supervisor was not configured with a forwarding table",
4590        ));
4591    };
4592    let probe_started_at = Instant::now();
4593    let mut deadline = probe_started_at + runtime.health.deadline;
4594    if let Some(cap) = deadline_cap {
4595        deadline = deadline.min(cap);
4596    }
4597    let pending = forwarding
4598        .begin_endpoint_health_probe_rpc_for(
4599            endpoint,
4600            MODULE_CONTROL_OP_HEALTH_CHECK,
4601            probe_started_at,
4602            deadline,
4603        )
4604        .map_err(|err| {
4605            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4606        })?;
4607    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4608}
4609
4610/// Send a begun health probe and classify its answer.
4611async fn await_health_probe(
4612    forwarding: &ForwardingTable,
4613    pending: PendingModuleControlRpc,
4614    deadline: Instant,
4615    probe_budget: Duration,
4616) -> Result<HealthReport, HealthProbeError> {
4617    let PendingModuleControlRpc {
4618        endpoint,
4619        module_sink,
4620        negotiated_ver,
4621        corr,
4622        receiver,
4623    } = pending;
4624    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4625        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4626    })?;
4627    let frame = Frame::build_with_version(
4628        negotiated_ver,
4629        FrameType::Request,
4630        control_flags(),
4631        0,
4632        0,
4633        corr,
4634        body,
4635    )
4636    .map_err(|err| {
4637        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4638    })?;
4639
4640    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4641    // blocks waiting for capacity when the module's egress queue is full, and an
4642    // unbounded await here freezes the whole supervision actor (it stops polling
4643    // Child::wait and supervisor commands), making the module unrecoverable
4644    // in-band. On timeout the probe fails like any transport failure.
4645    match timeout_at(deadline, module_sink.send(frame)).await {
4646        Ok(Ok(())) => {}
4647        Ok(Err(err)) => {
4648            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4649            // A closed sink means the module's egress channel is gone -- the
4650            // receiving half is dropped when its connection tears down. Proof.
4651            return Err(HealthProbeError::lane_dead(format!(
4652                "failed to send health.check: {err}"
4653            )));
4654        }
4655        Err(_elapsed) => {
4656            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4657            // A full egress queue means the module is not draining its socket, which
4658            // is consistent with a wedged module AND with one whose reader is merely
4659            // starved. Silence, not proof.
4660            return Err(HealthProbeError::no_answer(
4661                "health.check send timed out before enqueue (module egress full)",
4662            ));
4663        }
4664    }
4665
4666    match timeout_at(deadline, receiver).await {
4667        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4668        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4669        // and those prove it is alive even though the probe failed.
4670        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4671            response.health_report().ok_or_else(|| {
4672                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4673            })
4674        }
4675        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4676            format!("health.check rejected: {}", body.message),
4677        )),
4678        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4679            Err(HealthProbeError::lane_dead(message))
4680        }
4681        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4682            Err(HealthProbeError::bad_answer(message))
4683        }
4684        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4685            Err(HealthProbeError::bad_answer(format!(
4686                "expected module-control op '{expected}', got '{actual}'"
4687            )))
4688        }
4689        // A reply that crosses the deadline before this waiter observes it is
4690        // still proof of life. The forwarding path records its end-to-end latency
4691        // before delivering this classification.
4692        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4693            "module answered health.check after its daemon deadline",
4694        )),
4695        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4696            "health.check waiter was canceled before the module responded",
4697        )),
4698        Err(_) => {
4699            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4700            Err(HealthProbeError::no_answer(format!(
4701                "module did not answer health.check within {probe_budget:?}"
4702            )))
4703        }
4704    }
4705}
4706
4707#[allow(clippy::too_many_arguments)]
4708async fn handle_health_report(
4709    spec: &ModuleSpec,
4710    runtime: &SupervisorRuntimeConfig,
4711    registry: &Registry,
4712    process_liveness: &SupervisorProcessLiveness,
4713    snapshot: &SharedSnapshot,
4714    child: &mut Option<SupervisedChild>,
4715    report: HealthReport,
4716    now_ms: u64,
4717) {
4718    let status = supervisor_health_status(report.status);
4719    let detail = report.detail.clone();
4720    let metrics = truncate_health_metrics(report.metrics);
4721    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4722        state.health.status = status;
4723        state.health.last_probe_ms = Some(now_ms);
4724        state.health.detail = detail.clone();
4725        state.health.metrics = metrics.clone();
4726        state.health.consecutive_failures = 0;
4727    });
4728
4729    let action = match report.status {
4730        HealthStatus::Ok => return,
4731        HealthStatus::Degraded => runtime.health.on_degraded,
4732        HealthStatus::Failing => runtime.health.on_failing,
4733    };
4734    apply_l3_health_action(
4735        spec,
4736        runtime,
4737        registry,
4738        process_liveness,
4739        snapshot,
4740        child,
4741        status,
4742        detail.as_deref(),
4743        action,
4744        now_ms,
4745    )
4746    .await;
4747}
4748
4749#[allow(clippy::too_many_arguments)]
4750async fn handle_health_probe_failure(
4751    spec: &ModuleSpec,
4752    runtime: &SupervisorRuntimeConfig,
4753    registry: &Registry,
4754    process_liveness: &SupervisorProcessLiveness,
4755    snapshot: &SharedSnapshot,
4756    child: &mut Option<SupervisedChild>,
4757    err: HealthProbeError,
4758    now_ms: u64,
4759) {
4760    let threshold = runtime.health.failure_threshold.max(1);
4761    let mut failures = 0;
4762    // Carry the evidence class into the operator-visible detail. Without it,
4763    // "module did not answer within 5s" and "the control lane is gone" are two
4764    // prose strings in the same field, and the reader has to know the codebase to
4765    // tell which one is proof of anything.
4766    let detail = format!("[{}] {err}", err.label());
4767    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4768        // A failed wire probe invalidates the last report, even before the
4769        // restart threshold. HTTP probes already mark failures as Failing.
4770        if state.spawned_protocol.unwrap_or(spec.protocol) == ModuleProtocol::Subc {
4771            state.health.status = SupervisorHealthStatus::Unknown;
4772        }
4773        state.health.last_probe_ms = Some(now_ms);
4774        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4775        state.health.detail = Some(detail.clone());
4776        state.health.metrics = None;
4777        failures = state.health.consecutive_failures;
4778    });
4779
4780    if failures < threshold {
4781        warn!(
4782            module_id = %spec.module_id,
4783            consecutive_failures = failures,
4784            threshold,
4785            evidence = err.label(),
4786            detail = %detail,
4787            "health.check probe failed"
4788        );
4789        return;
4790    }
4791
4792    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4793        state.state = ModuleState::Unresponsive;
4794        state.health.status = SupervisorHealthStatus::Unresponsive;
4795    });
4796    // The evidence class is logged at the kill site because this is the line an
4797    // operator reads after an unexplained restart. A streak of `no-answer` under
4798    // machine load is the known false-positive shape; a `lane-dead` is not.
4799    if runtime.health.critical {
4800        error!(
4801            module_id = %spec.module_id,
4802            status = "unresponsive",
4803            evidence = err.label(),
4804            detail = %detail,
4805            "critical module health alert"
4806        );
4807    } else {
4808        warn!(
4809            module_id = %spec.module_id,
4810            status = "unresponsive",
4811            evidence = err.label(),
4812            detail = %detail,
4813            "module health threshold breached"
4814        );
4815    }
4816    if let Err(err) = health_restart_child(
4817        spec,
4818        runtime,
4819        registry,
4820        process_liveness,
4821        snapshot,
4822        child,
4823        SupervisorHealthStatus::Unresponsive,
4824        Some(&detail),
4825        now_ms,
4826    )
4827    .await
4828    {
4829        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4830    }
4831}
4832
4833#[allow(clippy::too_many_arguments)]
4834async fn apply_l3_health_action(
4835    spec: &ModuleSpec,
4836    runtime: &SupervisorRuntimeConfig,
4837    registry: &Registry,
4838    process_liveness: &SupervisorProcessLiveness,
4839    snapshot: &SharedSnapshot,
4840    child: &mut Option<SupervisedChild>,
4841    status: SupervisorHealthStatus,
4842    detail: Option<&str>,
4843    action: HealthAction,
4844    now_ms: u64,
4845) {
4846    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4847    match action {
4848        HealthAction::Report => {
4849            info!(
4850                module_id = %spec.module_id,
4851                status = ?status,
4852                detail,
4853                "module reported non-ok health"
4854            );
4855        }
4856        HealthAction::Alert => {
4857            error!(
4858                module_id = %spec.module_id,
4859                status = ?status,
4860                detail,
4861                "module health alert"
4862            );
4863        }
4864        HealthAction::Restart => {
4865            if let Err(err) = health_restart_child(
4866                spec,
4867                runtime,
4868                registry,
4869                process_liveness,
4870                snapshot,
4871                child,
4872                status,
4873                detail,
4874                now_ms,
4875            )
4876            .await
4877            {
4878                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4879            }
4880        }
4881    }
4882}
4883
4884#[allow(clippy::too_many_arguments)]
4885async fn health_restart_child(
4886    spec: &ModuleSpec,
4887    runtime: &SupervisorRuntimeConfig,
4888    registry: &Registry,
4889    process_liveness: &SupervisorProcessLiveness,
4890    snapshot: &SharedSnapshot,
4891    child: &mut Option<SupervisedChild>,
4892    status: SupervisorHealthStatus,
4893    detail: Option<&str>,
4894    now_ms: u64,
4895) -> Result<(), SuperviseError> {
4896    let (enabled, schedule) = {
4897        let mut state = lock_snapshot(snapshot)?;
4898        let enabled = state.enabled;
4899        let schedule = if enabled {
4900            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4901        } else {
4902            None
4903        };
4904        (enabled, schedule)
4905    };
4906
4907    if !enabled {
4908        return Err(SuperviseError::Disabled {
4909            module_id: spec.module_id.clone(),
4910        });
4911    }
4912
4913    if schedule.is_none() {
4914        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4915        error!(
4916            module_id = %spec.module_id,
4917            status = ?status,
4918            detail,
4919            max_restarts = runtime.restart_policy.max_restarts,
4920            window_secs = runtime.restart_policy.window.as_secs(),
4921            reason = %runtime.restart_policy.budget_exhausted_detail(),
4922            "health restart budget exhausted; marking module failed"
4923        );
4924        let stop_notice = begin_forwarding_drain_if_configured(
4925            spec,
4926            runtime,
4927            registry,
4928            snapshot,
4929            Some(true),
4930            RouteCloseReason::Disable,
4931        )
4932        .await?;
4933        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4934            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4935        })?;
4936        drain_optional_child(
4937            &spec.module_id,
4938            spec.protocol,
4939            stop_notice,
4940            registry,
4941            runtime.forwarding.as_deref(),
4942            snapshot,
4943            &runtime.terminal_ring,
4944            &runtime.spawn_events,
4945            child,
4946            runtime.drain_timeout,
4947            ModuleState::Failed,
4948            Some(true),
4949        )
4950        .await?;
4951        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4952        return Ok(());
4953    }
4954
4955    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4956    let mut restart_count = 0;
4957    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4958        restart_count = state.crash_restarts.len();
4959        state.state = ModuleState::Unresponsive;
4960        state.health.status = status;
4961        state.health.last_action = Some(HealthAction::Restart.to_string());
4962        state.health.last_action_ms = Some(now_ms);
4963    })?;
4964    warn!(
4965        module_id = %spec.module_id,
4966        status = ?status,
4967        detail,
4968        restart_count,
4969        restart_in_window = schedule.restart_in_window,
4970        delay_ms = schedule.delay.as_millis() as u64,
4971        "health-triggered module restart"
4972    );
4973
4974    let stop_notice = begin_forwarding_drain_if_configured(
4975        spec,
4976        runtime,
4977        registry,
4978        snapshot,
4979        Some(true),
4980        RouteCloseReason::Restart,
4981    )
4982    .await?;
4983    drain_optional_child(
4984        &spec.module_id,
4985        spec.protocol,
4986        stop_notice,
4987        registry,
4988        runtime.forwarding.as_deref(),
4989        snapshot,
4990        &runtime.terminal_ring,
4991        &runtime.spawn_events,
4992        child,
4993        runtime.drain_timeout,
4994        ModuleState::Restarting,
4995        Some(true),
4996    )
4997    .await?;
4998    schedule_respawn(
4999        runtime,
5000        snapshot,
5001        &spec.module_id,
5002        schedule.delay,
5003        RespawnKind::Spawn,
5004    )
5005}
5006
5007fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
5008    if let Some(reply) = runtime
5009        .deferred_reload_reply
5010        .lock()
5011        .unwrap_or_else(|p| p.into_inner())
5012        .take()
5013    {
5014        let _ = reply.send(Err(SuperviseError::ReloadFailed {
5015            module_id: module_id.to_string(),
5016            reason: reason.to_string(),
5017        }));
5018    }
5019}
5020
5021fn schedule_respawn(
5022    runtime: &SupervisorRuntimeConfig,
5023    snapshot: &SharedSnapshot,
5024    module_id: &str,
5025    delay: Duration,
5026    kind: RespawnKind,
5027) -> Result<(), SuperviseError> {
5028    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
5029    update_snapshot(snapshot, Some(module_id), |state| {
5030        state.respawn_pending = true
5031    })?;
5032    *runtime
5033        .scheduled_respawn
5034        .lock()
5035        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5036        deadline: Instant::now() + delay,
5037        kind,
5038    });
5039    Ok(())
5040}
5041
5042fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5043    let _ = update_snapshot(snapshot, Some(module_id), |state| {
5044        state.health.last_action = Some(action);
5045        state.health.last_action_ms = Some(now_ms);
5046    });
5047}
5048
5049fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5050    match status {
5051        HealthStatus::Ok => SupervisorHealthStatus::Ok,
5052        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5053        HealthStatus::Failing => SupervisorHealthStatus::Failing,
5054    }
5055}
5056
5057/// Caps the metrics blob stored in the cached supervisor snapshot, which is
5058/// returned to every `supervisor.list` and `supervisor.health` caller.
5059///
5060/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
5061/// path: that request exists to return a module's complete metrics object, and
5062/// `ck health <module-id>` documents it as the way to see what the cached view
5063/// truncates. The asymmetry is the feature.
5064///
5065/// So a new caller must decide which side it is on rather than assume the cap is
5066/// universal. Reaching for it on a fresh-probe path would silently reintroduce
5067/// the truncation that path exists to avoid.
5068fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5069    let metrics = metrics?;
5070    match serde_json::to_vec(&metrics) {
5071        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5072            "truncated": true,
5073            "original_bytes": encoded.len(),
5074        })),
5075        Ok(_) | Err(_) => Some(metrics),
5076    }
5077}
5078
5079/// Spread health probes so a fleet-wide restart does not converge them.
5080///
5081/// The delay is derived from the module id and probe index rather than a random
5082/// source, so it is deterministic per module: a module keeps its own offset
5083/// across daemon restarts instead of re-rolling into a collision.
5084fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5085    if cadence.is_zero() {
5086        return Duration::ZERO;
5087    }
5088    let cadence_ms = cadence.as_millis() as u64;
5089    // This early return is REDUNDANT, deliberately, and a mutation run will show
5090    // it surviving removal. Recording why here so the next person to notice does
5091    // not have to re-derive it:
5092    //
5093    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
5094    //   a zero cadence and builds the Duration from whole milliseconds, so a
5095    //   sub-millisecond cadence cannot come from config.
5096    // - Even if reached it changes no answer. The `.max(1)` below makes the span
5097    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
5098    //   -- exactly what this returns.
5099    //
5100    // Kept as a guard against a future widening of the config parser (accepting
5101    // microseconds, say), which would make the sub-millisecond case reachable.
5102    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
5103    // divides by zero. Remove this and nothing changes.
5104    if cadence_ms == 0 {
5105        return cadence;
5106    }
5107    // Note that this never returns less than one cadence, including for the FIRST
5108    // probe. So a freshly registered module reports health `unknown` for a full
5109    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5110    // ready to answer.
5111    //
5112    // That is a property of the supervisor's schedule, not of any module: an
5113    // operator watching a restart sees `unknown` and cannot tell it from a module
5114    // that is slow to warm. Measured on two unrelated modules, both flipping to
5115    // `ok` between 22s and 32s after restart.
5116    //
5117    // Left as-is because spreading the first probe is what keeps a fleet-wide
5118    // restart from firing fourteen simultaneous probes into a cold machine. The
5119    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5120    // that thundering herd for a faster first reading.
5121    let jitter_span = (cadence_ms / 10).max(1);
5122    let hash = module_id.as_bytes().iter().fold(
5123        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5124        |acc, byte| {
5125            acc.wrapping_mul(1099511628211)
5126                .wrapping_add(u64::from(*byte))
5127        },
5128    );
5129    cadence + Duration::from_millis(hash % jitter_span)
5130}
5131
5132#[cfg(test)]
5133mod tests {
5134    use super::*;
5135
5136    #[test]
5137    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5138        let handle = SupervisorHandle::new();
5139        let module_id = "readded-tombstone";
5140        handle.record_rescan_removal(module_id);
5141        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5142
5143        handle.apply_identity_configuration(&ModuleSpec {
5144            module_id: module_id.to_string(),
5145            program: PathBuf::from("/test/module"),
5146            args: Vec::new(),
5147            env: Vec::new(),
5148            reserved: false,
5149            reserved_prefixes: Vec::new(),
5150            protocol: ModuleProtocol::Subc,
5151            overlap: Default::default(),
5152        });
5153
5154        assert!(
5155            handle.removal_tombstone_age_ms(module_id).is_none(),
5156            "a re-added module must not retain a stale removal tombstone"
5157        );
5158    }
5159
5160    /// What one module's owner looked like from the control plane at the
5161    /// instant after its first process was spawned.
5162    #[derive(Debug, PartialEq, Eq)]
5163    struct OwnerInSpawnWindow {
5164        module_id: String,
5165        configured: bool,
5166        on_roster: bool,
5167        admission_refusal: Option<&'static str>,
5168    }
5169
5170    /// A supervised module's process can connect, register, sync its scopes
5171    /// and describe them as soon as it is spawned, which is BEFORE the
5172    /// supervisor puts the module on the roster. In that window the owner must
5173    /// already read as configured, so a scoped `route.open` against it is
5174    /// refused as retryable `scope_not_synced` and not as terminal
5175    /// `scope_not_live` ("will never sync").
5176    ///
5177    /// The hook runs in exactly that window on every path that takes on a new
5178    /// module, so no race with a real child is needed: `on_roster: false`
5179    /// proves each observation was taken before the roster insert.
5180    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5181    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5182        use crate::scopes::ScopeTable;
5183        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5184
5185        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5186        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5187            module_id: module_id.to_string(),
5188            program,
5189            args: Vec::new(),
5190            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5191                .into_iter()
5192                .map(|key| (key.to_string(), dir.path().display().to_string()))
5193                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5194                .collect(),
5195            reserved: false,
5196            reserved_prefixes: Vec::new(),
5197            protocol: ModuleProtocol::Subc,
5198            overlap: Default::default(),
5199        };
5200        let live = super::terminal_history_tests::fake_aft_stub_path();
5201        let missing = dir.path().join("definitely-missing-module");
5202
5203        let handle = SupervisorHandle::new();
5204        let mut supervisor =
5205            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5206                .with_handle(handle.clone());
5207        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5208        let hook_handle = handle.clone();
5209        let hook_observed = Arc::clone(&observed);
5210        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5211            // Exactly what the control plane computes for a scoped route.open
5212            // naming this module as the owner of a scope it has not synced.
5213            let configured = hook_handle.is_configured(module_id);
5214            let selector = ScopeSelector {
5215                owner: Principal::Reserved {
5216                    module_id: module_id.to_string(),
5217                },
5218                scope_ref: "s".to_string(),
5219                scope_epoch: Some(1),
5220            };
5221            let carrier = Principal::Reserved {
5222                module_id: "carrier".to_string(),
5223            };
5224            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5225                .admit(&carrier, module_id, &selector, configured)
5226            {
5227                Ok(_) => None,
5228                Err(refusal) => Some(refusal.code),
5229            };
5230            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5231                module_id: module_id.to_string(),
5232                configured,
5233                on_roster: hook_handle.get(module_id).is_some(),
5234                admission_refusal,
5235            });
5236        })));
5237
5238        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5239        let configured = supervisor
5240            .supervise_configured(stub("configured", live.clone()), true)
5241            .unwrap();
5242        let with_health = supervisor
5243            .supervise_configured_with_health(
5244                stub("with-health", live.clone()),
5245                true,
5246                HealthConfig::default(),
5247                None,
5248                RestartPolicy::default(),
5249            )
5250            .unwrap();
5251        // The failed-spawn path still puts the module on the roster (as
5252        // failed), so it is configured throughout.
5253        let failed = supervisor
5254            .supervise_configured_with_health(
5255                stub("failed-spawn", missing.clone()),
5256                true,
5257                HealthConfig::default(),
5258                None,
5259                RestartPolicy::default(),
5260            )
5261            .unwrap();
5262        // A failed plain `spawn` puts nothing on the roster, so its mark is
5263        // taken back once the spawn has failed.
5264        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5265
5266        let expected = [
5267            "plain",
5268            "configured",
5269            "with-health",
5270            "failed-spawn",
5271            "spawn-error",
5272        ]
5273        .into_iter()
5274        .map(|module_id| OwnerInSpawnWindow {
5275            module_id: module_id.to_string(),
5276            configured: true,
5277            on_roster: false,
5278            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5279        })
5280        .collect::<Vec<_>>();
5281        assert_eq!(*observed.lock().unwrap(), expected);
5282
5283        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5284            assert!(
5285                handle.get(module_id).is_some(),
5286                "{module_id} is on the roster"
5287            );
5288            assert!(
5289                handle.is_configured(module_id),
5290                "{module_id} stays configured"
5291            );
5292        }
5293        assert!(handle.get("spawn-error").is_none());
5294        assert!(
5295            !handle.is_configured("spawn-error"),
5296            "a plain spawn that failed must not leave its module marked configured"
5297        );
5298
5299        // Leaving the roster clears the mark with it.
5300        handle.retire("failed-spawn");
5301        assert!(!handle.is_configured("failed-spawn"));
5302
5303        for module in [plain, configured, with_health] {
5304            module.stop().await.unwrap();
5305        }
5306        drop(failed);
5307    }
5308
5309    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5310        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5311        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5312            snapshot.process_alive = true;
5313            snapshot.pid = Some(41);
5314            snapshot.spawned_at_ms = Some(42);
5315            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5316            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5317                device: 43,
5318                inode: 44,
5319            });
5320        })
5321        .unwrap();
5322        snapshot
5323    }
5324
5325    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5326        let snapshot = lock_snapshot(snapshot).unwrap();
5327        assert!(!snapshot.process_alive);
5328        assert_eq!(snapshot.pid, None);
5329        assert_eq!(snapshot.spawned_at_ms, None);
5330        assert_eq!(snapshot.spawned_from, None);
5331        assert_eq!(snapshot.spawned_file_identity, None);
5332    }
5333
5334    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5335    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5336        let supervisor =
5337            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5338        let mut runtime = supervisor.runtime_config();
5339        runtime.test_seed_stale_facts_before_enable_spawn = true;
5340        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5341        let mut child = None;
5342        let spec = ModuleSpec {
5343            module_id: "failed-enable-clears-facts".to_string(),
5344            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5345            args: Vec::new(),
5346            env: Vec::new(),
5347            reserved: false,
5348            reserved_prefixes: Vec::new(),
5349            protocol: ModuleProtocol::Subc,
5350            overlap: Default::default(),
5351        };
5352
5353        let result = set_child_enabled(
5354            &spec,
5355            &runtime,
5356            &supervisor.registry,
5357            &supervisor.process_liveness,
5358            &snapshot,
5359            &mut child,
5360            true,
5361        )
5362        .await;
5363
5364        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5365        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5366        assert_snapshot_process_facts_cleared(&snapshot);
5367    }
5368
5369    #[tokio::test]
5370    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5371        let supervisor =
5372            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5373        let runtime = supervisor.runtime_config();
5374        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5375            ModuleState::Restarting,
5376            true,
5377        )));
5378        let spec = ModuleSpec {
5379            module_id: "start-stranded-restarting".to_string(),
5380            program: super::terminal_history_tests::fake_aft_stub_path(),
5381            args: Vec::new(),
5382            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5383            reserved: false,
5384            reserved_prefixes: Vec::new(),
5385            protocol: ModuleProtocol::None,
5386            overlap: Default::default(),
5387        };
5388        let mut child = None;
5389        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5390        assert!(!super::set_child_enabled(
5391            &spec,
5392            &runtime,
5393            &Registry::default(),
5394            &supervisor.process_liveness,
5395            &snapshot,
5396            &mut child,
5397            true
5398        )
5399        .await
5400        .unwrap());
5401        assert!(child.is_none());
5402        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5403        assert!(super::set_child_enabled(
5404            &spec,
5405            &runtime,
5406            &Registry::default(),
5407            &supervisor.process_liveness,
5408            &snapshot,
5409            &mut child,
5410            true
5411        )
5412        .await
5413        .unwrap());
5414        assert_eq!(
5415            lock_snapshot(&snapshot).unwrap().state,
5416            ModuleState::Running
5417        );
5418        let mut child = child.unwrap();
5419        child.start_kill().unwrap();
5420        child.wait().await.unwrap();
5421    }
5422
5423    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5424    async fn failed_reload_spawn_clears_current_process_facts() {
5425        let supervisor =
5426            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5427        let mut runtime = supervisor.runtime_config();
5428        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5429        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5430        let mut child = None;
5431        let spec = ModuleSpec {
5432            module_id: "failed-reload-clears-facts".to_string(),
5433            program: PathBuf::from("/unused/failed-reload-module"),
5434            args: Vec::new(),
5435            env: Vec::new(),
5436            reserved: false,
5437            reserved_prefixes: Vec::new(),
5438            protocol: ModuleProtocol::Subc,
5439            overlap: Default::default(),
5440        };
5441
5442        let result = handle_reload_spawn_failure(
5443            &spec,
5444            &runtime,
5445            &supervisor.process_liveness,
5446            &snapshot,
5447            &mut child,
5448            "forced reload spawn failure".to_string(),
5449        )
5450        .await;
5451
5452        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5453        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5454        assert_snapshot_process_facts_cleared(&snapshot);
5455    }
5456
5457    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5458    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5459        let supervisor =
5460            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5461        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5462        let module = supervisor.supervised_module(
5463            ModuleSpec {
5464                module_id: "drop-clears-facts".to_string(),
5465                program: PathBuf::from("/unused/drop-module"),
5466                args: Vec::new(),
5467                env: Vec::new(),
5468                reserved: false,
5469                reserved_prefixes: Vec::new(),
5470                protocol: ModuleProtocol::Subc,
5471                overlap: Default::default(),
5472            },
5473            supervisor.runtime_config(),
5474            Arc::clone(&snapshot),
5475            None,
5476        );
5477        assert!(!module
5478            .inner
5479            .monitor
5480            .lock()
5481            .unwrap()
5482            .as_ref()
5483            .unwrap()
5484            .is_finished());
5485
5486        drop(module);
5487
5488        assert_eq!(
5489            lock_snapshot(&snapshot).unwrap().state,
5490            ModuleState::Stopped
5491        );
5492        assert_snapshot_process_facts_cleared(&snapshot);
5493    }
5494
5495    #[cfg(unix)]
5496    #[tokio::test]
5497    async fn rescan_preserves_running_protocol_until_respawn() {
5498        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5499        let initial = ModuleSpec {
5500            module_id: "rescan-protocol".into(),
5501            program: PathBuf::from("/bin/sleep"),
5502            args: vec!["60".into()],
5503            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5504                .into_iter()
5505                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5506                .collect(),
5507            reserved: false,
5508            reserved_prefixes: vec![],
5509            protocol: ModuleProtocol::None,
5510            overlap: Default::default(),
5511        };
5512        let supervisor =
5513            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5514        let module = supervisor.spawn(initial.clone()).unwrap();
5515        assert!(module.status().unwrap().live);
5516        let mut next = initial;
5517        next.protocol = ModuleProtocol::Subc;
5518        module
5519            .update_configuration(next.clone(), HealthConfig::default(), None)
5520            .await
5521            .unwrap();
5522        assert!(
5523            module.status().unwrap().live,
5524            "rescan must not require HELLO from the old non-wire process"
5525        );
5526        let runtime = supervisor.runtime_config();
5527        let action = on_child_exit(
5528            &next,
5529            RestartPolicy::default(),
5530            &supervisor.registry,
5531            &module.inner.snapshot,
5532            &runtime.terminal_ring,
5533            &runtime.spawn_events,
5534            &runtime.child_roster,
5535            ExitReport {
5536                kind: ExitKind::Clean,
5537                code: Some(0),
5538                signal: None,
5539                at_ms: unix_ms_now(),
5540            },
5541        )
5542        .await;
5543        assert!(
5544            matches!(action, NextAction::Restart { .. }),
5545            "the old non-wire process's clean exit must restart"
5546        );
5547        module.drain().await.unwrap();
5548    }
5549
5550    #[cfg(unix)]
5551    #[tokio::test]
5552    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5553        use std::os::unix::fs::PermissionsExt;
5554        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5555        let script = dir.join("module.sh");
5556        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5557        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5558        let record_path = dir.join("live-children.json");
5559        let supervisor =
5560            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5561                .with_live_children_record(&record_path);
5562        for (program, args) in [
5563            (PathBuf::from("sleep"), vec!["60".into()]),
5564            (script, vec![]),
5565        ] {
5566            let spec = ModuleSpec {
5567                module_id: "image-identity".into(),
5568                program: program.clone(),
5569                args,
5570                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5571                    .into_iter()
5572                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5573                    .collect(),
5574                reserved: false,
5575                reserved_prefixes: vec![],
5576                protocol: ModuleProtocol::None,
5577                overlap: Default::default(),
5578            };
5579            let module = supervisor.spawn(spec).unwrap();
5580            #[cfg(target_os = "macos")]
5581            {
5582                // SETEXEC confirmation is asynchronous; the orphan record must
5583                // identify the final image, never the intermediate trampoline.
5584                let deadline = Instant::now() + Duration::from_secs(5);
5585                while crate::live_children::read_record(&record_path)
5586                    .unwrap()
5587                    .iter()
5588                    .all(|entry| entry.executable.is_none())
5589                {
5590                    assert!(Instant::now() < deadline, "module image was not confirmed");
5591                    tokio::time::sleep(Duration::from_millis(5)).await;
5592                }
5593            }
5594            let entry = crate::live_children::read_record(&record_path)
5595                .unwrap()
5596                .pop()
5597                .unwrap();
5598            let observed = subc_os::Process::open(entry.pid)
5599                .unwrap()
5600                .unwrap()
5601                .observe()
5602                .unwrap();
5603            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5604            module.drain().await.unwrap();
5605            assert_eq!(
5606                verdict,
5607                crate::live_children::IdentityVerdict::Matches,
5608                "program {program:?}: recorded {entry:?}, observed {observed:?}"
5609            );
5610        }
5611    }
5612
5613    #[cfg(unix)]
5614    fn http_fixture(
5615        dir: &std::path::Path,
5616        url: &str,
5617        threshold: u32,
5618    ) -> crate::daemon_config::ConfiguredModule {
5619        let path = dir.join("subc.jsonc");
5620        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5621            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5622            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5623            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5624            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5625        }}}).to_string()).unwrap();
5626        crate::daemon_config::load(&path)
5627            .unwrap()
5628            .unwrap()
5629            .modules
5630            .pop()
5631            .unwrap()
5632    }
5633
5634    #[cfg(unix)]
5635    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5636        timeout(Duration::from_secs(5), async {
5637            loop {
5638                if module.status().unwrap().health.status == status {
5639                    break;
5640                }
5641                sleep(Duration::from_millis(5)).await;
5642            }
5643        })
5644        .await
5645        .unwrap_or_else(|_| {
5646            panic!(
5647                "expected {status:?}, got {:?}",
5648                module.status().unwrap().health
5649            )
5650        });
5651    }
5652
5653    #[cfg(unix)]
5654    #[tokio::test]
5655    async fn http_health_status_flips_ok_failing_ok() {
5656        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5657        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5658        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5659        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5660        let serving_status = status.clone();
5661        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5662        let server = tokio::spawn(async move {
5663            loop {
5664                let (mut stream, _) = listener.accept().await.unwrap();
5665                let mut request = [0u8; 2048];
5666                let count = stream.read(&mut request).await.unwrap();
5667                assert!(count > 0, "a probe must send an HTTP request");
5668                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5669                let body = if code == 200 {
5670                    "ready"
5671                } else {
5672                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5673                };
5674                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5675                let _ = stream.write_all(response.as_bytes()).await;
5676            }
5677        });
5678        let configured = http_fixture(&dir, &url, 1000);
5679        let module =
5680            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5681                .supervise_configured_with_health(
5682                    configured.module_spec(),
5683                    true,
5684                    configured.health,
5685                    configured.drain_timeout_ms,
5686                    configured.restart,
5687                )
5688                .unwrap();
5689        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5690        status.store(503, std::sync::atomic::Ordering::SeqCst);
5691        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5692        assert!(module
5693            .status()
5694            .unwrap()
5695            .health
5696            .detail
5697            .unwrap()
5698            .contains("scratch failure"));
5699        status.store(200, std::sync::atomic::Ordering::SeqCst);
5700        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5701        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5702        let before = module.status().unwrap();
5703        let (spec, mut health) = module.configuration().unwrap();
5704        health.http = None;
5705        module
5706            .update_configuration(spec.clone(), health.clone(), Some(10))
5707            .await
5708            .unwrap();
5709        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5710        health.http = Some(url);
5711        module
5712            .update_configuration(spec, health, Some(10))
5713            .await
5714            .unwrap();
5715        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5716        assert_eq!(
5717            module.status().unwrap().pid,
5718            before.pid,
5719            "changing a probe must apply live, not restart its process"
5720        );
5721        let (spec, mut health) = module.configuration().unwrap();
5722        health.failure_threshold = 2;
5723        module
5724            .update_configuration(spec, health, Some(10))
5725            .await
5726            .unwrap();
5727        status.store(503, std::sync::atomic::Ordering::SeqCst);
5728        timeout(Duration::from_secs(5), async {
5729            while module.status().unwrap().spawn_generation == before.spawn_generation {
5730                sleep(Duration::from_millis(5)).await;
5731            }
5732        })
5733        .await
5734        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5735        module.drain().await.unwrap();
5736        server.abort();
5737    }
5738
5739    #[cfg(unix)]
5740    #[tokio::test]
5741    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5742        let dir = subc_test_support::TestTempDir::new("http-refused");
5743        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5744        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5745        drop(unused);
5746        let configured = http_fixture(&dir, &url, 2);
5747        let module =
5748            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5749                .supervise_configured_with_health(
5750                    configured.module_spec(),
5751                    true,
5752                    configured.health,
5753                    configured.drain_timeout_ms,
5754                    configured.restart,
5755                )
5756                .unwrap();
5757        let before = module.status().unwrap().spawn_generation;
5758        timeout(Duration::from_secs(5), async {
5759            loop {
5760                let status = module.status().unwrap();
5761                if status.spawn_generation > before {
5762                    assert!(status.lifetime_restarts > 0);
5763                    break;
5764                }
5765                sleep(Duration::from_millis(5)).await;
5766            }
5767        })
5768        .await
5769        .expect("sustained HTTP refusal must trigger the health restart policy");
5770        module.drain().await.unwrap();
5771    }
5772
5773    #[cfg(unix)]
5774    #[tokio::test]
5775    async fn http_health_timeout_honours_deadline() {
5776        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5777        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5778        let server = tokio::spawn(async move {
5779            let _held = listener.accept().await.unwrap();
5780            std::future::pending::<()>().await;
5781        });
5782        let error = timeout(
5783            Duration::from_secs(1),
5784            probe_http_health(&url, Duration::from_millis(10)),
5785        )
5786        .await
5787        .expect("the probe must enforce its own deadline")
5788        .unwrap_err();
5789        server.abort();
5790        assert!(error.to_string().contains("timed out"));
5791    }
5792
5793    #[cfg(unix)]
5794    #[tokio::test]
5795    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5796        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5797        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5798        let url = format!(
5799            "http://localhost:{}/healthz",
5800            listener.local_addr().unwrap().port()
5801        );
5802        let server = tokio::spawn(async move {
5803            let (mut stream, _) = listener.accept().await.unwrap();
5804            let mut request = [0u8; 2048];
5805            assert!(stream.read(&mut request).await.unwrap() > 0);
5806            stream
5807                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5808                .await
5809                .unwrap();
5810        });
5811        // The deadline only bounds a hang. A probe that never tried the IPv6
5812        // address would be refused on 127.0.0.1 and fail at once, so a longer
5813        // deadline does not weaken the assertion; one second timed out under a
5814        // loaded parallel test run.
5815        assert_eq!(
5816            probe_http_health(&url, Duration::from_secs(10))
5817                .await
5818                .unwrap()
5819                .status,
5820            HealthStatus::Ok
5821        );
5822        server.await.unwrap();
5823    }
5824
5825    #[cfg(unix)]
5826    #[tokio::test]
5827    async fn http_health_timeout_keeps_partial_status_and_body() {
5828        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5829        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5830        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5831        let server = tokio::spawn(async move {
5832            let (mut stream, _) = listener.accept().await.unwrap();
5833            let mut request = [0u8; 2048];
5834            assert!(stream.read(&mut request).await.unwrap() > 0);
5835            stream
5836                .write_all(
5837                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5838                )
5839                .await
5840                .unwrap();
5841            std::future::pending::<()>().await;
5842        });
5843        let error = probe_http_health(&url, Duration::from_secs(1))
5844            .await
5845            .unwrap_err()
5846            .to_string();
5847        server.abort();
5848        assert!(
5849            error.contains("timed out")
5850                && error.contains("503 Unavailable")
5851                && error.contains("partial diagnostic"),
5852            "{error}"
5853        );
5854    }
5855
5856    #[cfg(unix)]
5857    #[tokio::test]
5858    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5859        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5860        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5861        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5862        let server = tokio::spawn(async move {
5863            let (mut stream, _) = listener.accept().await.unwrap();
5864            let mut request = [0u8; 2048];
5865            assert!(stream.read(&mut request).await.unwrap() > 0);
5866            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5867            let response = format!(
5868                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5869                body.len()
5870            );
5871            stream.write_all(response.as_bytes()).await.unwrap();
5872        });
5873        let error = probe_http_health(&url, Duration::from_secs(1))
5874            .await
5875            .unwrap_err()
5876            .to_string();
5877        server.await.unwrap();
5878        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5879        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5880        assert!(!error.contains("not-in-diagnostic"));
5881    }
5882
5883    #[cfg(unix)]
5884    #[tokio::test]
5885    async fn http_health_real_nats_server_monitoring() {
5886        if std::process::Command::new("nats-server")
5887            .arg("--version")
5888            .env("XDG_DATA_HOME", std::env::temp_dir())
5889            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5890            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5891            .output()
5892            .is_err()
5893        {
5894            eprintln!(
5895                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5896            );
5897            return;
5898        }
5899        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5900        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5901        let port = monitor.local_addr().unwrap().port();
5902        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5903        let client_port = client.local_addr().unwrap().port();
5904        let config = dir.join("server.conf");
5905        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5906        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5907        configured.program = PathBuf::from("nats-server");
5908        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5909        drop(monitor);
5910        drop(client);
5911        let module =
5912            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5913                .supervise_configured_with_health(
5914                    configured.module_spec(),
5915                    true,
5916                    configured.health,
5917                    configured.drain_timeout_ms,
5918                    configured.restart,
5919                )
5920                .unwrap();
5921        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5922        module.drain().await.unwrap();
5923    }
5924
5925    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5926    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5927        let supervisor =
5928            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5929        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5930        let initial = ModuleSpec {
5931            module_id: "rescan-preserves-spawn-facts".to_string(),
5932            program: PathBuf::from("/spawned/module"),
5933            args: Vec::new(),
5934            env: Vec::new(),
5935            reserved: false,
5936            reserved_prefixes: Vec::new(),
5937            protocol: ModuleProtocol::Subc,
5938            overlap: Default::default(),
5939        };
5940        let module = supervisor.supervised_module(
5941            initial.clone(),
5942            supervisor.runtime_config(),
5943            snapshot,
5944            None,
5945        );
5946        let before = module.status().unwrap();
5947        let mut replacement = initial;
5948        replacement.program = PathBuf::from("/rescanned/replacement-module");
5949
5950        module
5951            .update_configuration(replacement, HealthConfig::default(), None)
5952            .await
5953            .unwrap();
5954
5955        let after = module.status().unwrap();
5956        assert_eq!(after.pid, before.pid);
5957        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5958        assert_eq!(after.spawned_from, before.spawned_from);
5959        drop(module);
5960    }
5961}
5962
5963fn unix_ms_now() -> u64 {
5964    SystemTime::now()
5965        .duration_since(UNIX_EPOCH)
5966        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5967        .unwrap_or(0)
5968}
5969
5970async fn supervise_loop(
5971    mut spec: ModuleSpec,
5972    mut runtime: SupervisorRuntimeConfig,
5973    registry: Arc<Registry>,
5974    process_liveness: Arc<SupervisorProcessLiveness>,
5975    snapshot: SharedSnapshot,
5976    mut child: Option<SupervisedChild>,
5977    mut commands: mpsc::Receiver<SupervisorCommand>,
5978) {
5979    let mut health_probe = HealthProbeRuntime::default();
5980    // Registry writes (including embedded callers) notify this module only.
5981    // Subscribe before the first refresh; watch retains changes that arrive
5982    // while commands or probes are running. No polling fallback is needed.
5983    let mut registration_changes = match registry.subscribe_module_changes(&spec.module_id) {
5984        Ok(changes) => Some(changes),
5985        Err(err) => {
5986            // A poisoned registry cannot accept further writes, so it cannot
5987            // recover via a timer. Keep exit and command handling alive.
5988            warn!(module_id = %spec.module_id, error = %err, "health prober could not subscribe to registry");
5989            None
5990        }
5991    };
5992    // All restart backoffs run here, including health and operator requests.
5993    // While one is pending the loop serves commands, so disable or drain can
5994    // cancel the replacement without spawning a process just to stop it.
5995    let mut pending_respawn: Option<PendingRespawn> = None;
5996    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5997    // before anything else so a stop that interrupted a swap runs at once.
5998    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5999    loop {
6000        #[cfg(test)]
6001        {
6002            lock_snapshot(&snapshot).unwrap().actor_select = None;
6003        }
6004        #[cfg(target_os = "macos")]
6005        if let Some(active) = child.as_mut() {
6006            active.confirm_privacy_exec().await;
6007        }
6008        if let Some(scheduled) = runtime
6009            .scheduled_respawn
6010            .lock()
6011            .unwrap_or_else(|p| p.into_inner())
6012            .take()
6013        {
6014            pending_respawn = Some(scheduled);
6015        }
6016        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
6017            pending_respawn = None;
6018            cancel_deferred_reload(
6019                &runtime,
6020                &spec.module_id,
6021                "respawn cancelled by a supervisor command",
6022            );
6023        }
6024        if child.is_none() && pending_respawn.is_none() {
6025            cancel_deferred_reload(
6026                &runtime,
6027                &spec.module_id,
6028                "respawn cancelled before a replacement was spawned",
6029            );
6030            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6031                state.respawn_pending = false;
6032                state.coalesced_restart_pending = false;
6033                if matches!(
6034                    state.state,
6035                    ModuleState::Restarting
6036                        | ModuleState::Starting
6037                        | ModuleState::Draining
6038                        | ModuleState::Unresponsive
6039                ) {
6040                    error!(module_id = %spec.module_id, state = ?state.state,
6041                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
6042                    state.state = ModuleState::Failed;
6043                    clear_current_process_facts(state);
6044                }
6045            });
6046        }
6047        if let Some(command) = requeued.pop_front() {
6048            if !handle_supervisor_command(
6049                command,
6050                &mut spec,
6051                &mut runtime,
6052                &registry,
6053                &process_liveness,
6054                &snapshot,
6055                &mut child,
6056                &mut commands,
6057                &mut requeued,
6058            )
6059            .await
6060            {
6061                return;
6062            }
6063            if child.is_some() || !respawn_still_pending(&snapshot) {
6064                pending_respawn = None;
6065                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6066                    state.respawn_pending = false
6067                });
6068            }
6069            continue;
6070        }
6071        if child.is_some() {
6072            #[cfg(test)]
6073            {
6074                lock_snapshot(&snapshot).unwrap().actor_turns += 1;
6075            }
6076            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
6077            let wake_after = health_probe.wake_after();
6078            let probe_sleep = async move {
6079                match wake_after {
6080                    Some(delay) => sleep(delay).await,
6081                    None => std::future::pending().await,
6082                }
6083            };
6084            tokio::pin!(probe_sleep);
6085            // A `protocol: "none"` child (a plain process that never registers
6086            // over the subc wire, such as nats-server) with no HTTP health
6087            // check has nothing to probe, so it waits only for its exit or a
6088            // supervisor command. A rescan that adds an HTTP check is a command.
6089            let wire_child = running_protocol(&spec, &snapshot) == ModuleProtocol::Subc;
6090            #[cfg(test)]
6091            {
6092                let registration_pending = wire_child
6093                    && registration_changes
6094                        .as_ref()
6095                        .is_some_and(|changes| changes.has_changed().unwrap());
6096                let mut state = lock_snapshot(&snapshot).unwrap();
6097                state.actor_select = (!registration_pending).then_some(ActorSelectCheckpoint {
6098                    generation: state.spawn_generation,
6099                    turn: state.actor_turns,
6100                    registered_connection: health_probe.registered_connection,
6101                    next_probe_at: health_probe.next_probe_at,
6102                    wake_after,
6103                });
6104            }
6105            let active_child = child.as_mut().expect("child checked above");
6106            tokio::select! {
6107                wait_result = active_child.wait() => {
6108                    #[cfg(test)]
6109                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6110                    // Every arm below that gives up on the CHILD must keep the
6111                    // supervision task itself alive (child = None, loop
6112                    // continues into command-serving mode). Returning here
6113                    // closes the command channel, which makes the module
6114                    // permanently unrestartable in-band: a clean child exit
6115                    // of an enabled module once wedged the fleet this way
6116                    // ('supervisor command channel is closed') and required a
6117                    // full daemon restart to recover.
6118                    let exit_report = match wait_result {
6119                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6120                        Err(err) => {
6121                            active_child.drain_stderr(&spec.module_id).await;
6122                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6123                            // Every other exit path (on_child_exit's Clean/Crash arms,
6124                            // the reload-registration-failure path) records a terminal
6125                            // before moving on. Without one here, a module whose wait()
6126                            // itself errored (e.g. already reaped) leaves no terminal
6127                            // record at all -- an empty ring reads as "nothing died".
6128                            record_wait_error_terminal(
6129                                &spec.module_id,
6130                                &runtime.terminal_ring,
6131                                &runtime.spawn_events,
6132                            );
6133                            untrack_if_registration_released(
6134                                &process_liveness,
6135                                &registry,
6136                                &spec.module_id,
6137                                &snapshot,
6138                            );
6139                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6140                            child = None;
6141                            continue;
6142                        }
6143                    };
6144                    active_child.drain_stderr(&spec.module_id).await;
6145
6146                    let next = on_child_exit(
6147                        &spec,
6148                        runtime.restart_policy,
6149                        &registry,
6150                        &snapshot,
6151                        &runtime.terminal_ring,
6152                        &runtime.spawn_events,
6153                        &runtime.child_roster,
6154                        exit_report,
6155                    ).await;
6156                    // The exit is recorded, so a daemon shutdown may stop
6157                    // waiting for this child (see `SupervisedChild::wait`).
6158                    active_child.release_roster();
6159                    match next {
6160                        NextAction::Stop { registration_released } => {
6161                            if registration_released {
6162                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6163                            }
6164                            child = None;
6165                        }
6166                        NextAction::Restart { schedule } => {
6167                            let delay = schedule.map_or(
6168                                runtime.restart_policy.delay_for_restart(0),
6169                                |schedule| schedule.delay,
6170                            );
6171                            if let Some(schedule) = schedule {
6172                                log_crash_respawn(&spec.module_id, schedule);
6173                            }
6174                            // The exited child is fully recorded at this point,
6175                            // so release it and count the backoff down in the
6176                            // command-serving branch below rather than sleeping
6177                            // here: commands cannot be received from inside this
6178                            // select arm, and an operator disable or drain that
6179                            // arrives during the backoff must cancel the pending
6180                            // respawn instead of waiting for it to spawn first.
6181                            child = None;
6182                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6183                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6184                        }
6185                    }
6186                }
6187                command = commands.recv() => {
6188                    #[cfg(test)]
6189                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6190                    let Some(command) = command else {
6191                        return;
6192                    };
6193                    if !handle_supervisor_command(
6194                        command,
6195                        &mut spec,
6196                        &mut runtime,
6197                        &registry,
6198                        &process_liveness,
6199                        &snapshot,
6200                        &mut child,
6201                        &mut commands,
6202                        &mut requeued,
6203                    ).await {
6204                        return;
6205                    }
6206                }
6207                _ = async {
6208                    match registration_changes.as_mut() {
6209                        Some(changes) => { let _ = changes.changed().await; }
6210                        None => std::future::pending().await,
6211                    }
6212                }, if wire_child => {
6213                    #[cfg(test)]
6214                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6215                }
6216                _ = &mut probe_sleep => {
6217                    #[cfg(test)]
6218                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6219                    if health_probe.due() {
6220                        run_health_probe_cycle(
6221                            &spec,
6222                            &runtime,
6223                            &registry,
6224                            &process_liveness,
6225                            &snapshot,
6226                            &mut child,
6227                        ).await;
6228                        if child.is_some() {
6229                            health_probe.schedule_next(&spec, runtime.health.cadence);
6230                        }
6231                    }
6232                }
6233            }
6234        } else if let Some(pending) = pending_respawn {
6235            tokio::select! {
6236                _ = sleep_until(pending.deadline) => {
6237                    pending_respawn = None;
6238                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6239                    // A command handled below while the backoff elapsed may
6240                    // have stopped the module; never respawn past an operator's
6241                    // disable or drain.
6242                    if !respawn_still_pending(&snapshot) {
6243                        continue;
6244                    }
6245                    // The daemon began shutting down during the backoff: the
6246                    // spawn would be refused anyway, and refusing it here
6247                    // leaves the module stopped instead of reporting a
6248                    // failed restart.
6249                    if runtime.child_roster.is_closed() {
6250                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6251                            state.state = ModuleState::Stopped;
6252                        });
6253                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6254                        continue;
6255                    }
6256                    if let Err(err) = release_dead_registration(
6257                        &registry,
6258                        runtime.forwarding.as_deref(),
6259                        &snapshot,
6260                        &spec.module_id,
6261                    ).await {
6262                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6263                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6264                        continue;
6265                    }
6266
6267                    if matches!(pending.kind, RespawnKind::Reload) {
6268                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6269                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6270                        if let Some(reply) = reply { let _ = reply.send(result); }
6271                        continue;
6272                    }
6273                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6274                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6275                        Ok(next_child) => {
6276                            child = Some(next_child);
6277                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6278                        }
6279                        Err(err) => {
6280                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6281                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6282                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6283                        }
6284                    }
6285                }
6286                command = commands.recv() => {
6287                    let Some(command) = command else {
6288                        return;
6289                    };
6290                    if !handle_supervisor_command(
6291                        command,
6292                        &mut spec,
6293                        &mut runtime,
6294                        &registry,
6295                        &process_liveness,
6296                        &snapshot,
6297                        &mut child,
6298                        &mut commands,
6299                        &mut requeued,
6300                    ).await {
6301                        return;
6302                    }
6303                    // Reconcile the pending respawn with what the command did:
6304                    // a start may already have spawned a fresh child,
6305                    // while a disable or drain moved the snapshot out of the
6306                    // state the respawn was counting down from.
6307                    if child.is_some() || !respawn_still_pending(&snapshot) {
6308                        pending_respawn = None;
6309                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6310                    }
6311                }
6312            }
6313        } else {
6314            let Some(command) = commands.recv().await else {
6315                return;
6316            };
6317            if !handle_supervisor_command(
6318                command,
6319                &mut spec,
6320                &mut runtime,
6321                &registry,
6322                &process_liveness,
6323                &snapshot,
6324                &mut child,
6325                &mut commands,
6326                &mut requeued,
6327            )
6328            .await
6329            {
6330                return;
6331            }
6332        }
6333    }
6334}
6335
6336fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6337    info!(
6338        module_id,
6339        restart_in_window = schedule.restart_in_window,
6340        delay_ms = schedule.delay.as_millis() as u64,
6341        "respawning after crash"
6342    );
6343}
6344
6345/// Whether the respawn a backoff was counting down to is still wanted. A
6346/// disable or drain handled while the backoff elapsed moves the snapshot out
6347/// of `Restarting`, and the operator's stop must win over the pending respawn,
6348/// so every sleep-then-spawn path re-validates against the live snapshot
6349/// instead of assuming the state it left behind still holds.
6350fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6351    matches!(
6352        lock_snapshot(snapshot),
6353        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6354    )
6355}
6356
6357enum NextAction {
6358    Stop {
6359        registration_released: bool,
6360    },
6361    Restart {
6362        schedule: Option<CrashRestartSchedule>,
6363    },
6364}
6365
6366#[allow(clippy::too_many_arguments)]
6367async fn handle_supervisor_command(
6368    command: SupervisorCommand,
6369    spec: &mut ModuleSpec,
6370    runtime: &mut SupervisorRuntimeConfig,
6371    registry: &Arc<Registry>,
6372    process_liveness: &SupervisorProcessLiveness,
6373    snapshot: &SharedSnapshot,
6374    child: &mut Option<SupervisedChild>,
6375    commands: &mut mpsc::Receiver<SupervisorCommand>,
6376    requeued: &mut VecDeque<SupervisorCommand>,
6377) -> bool {
6378    match command {
6379        SupervisorCommand::Drain { reply } => {
6380            // A plain stop runs no forwarding drain, so nothing reaches the
6381            // module over its connection before the wait: ask by signal.
6382            let result = drain_optional_child(
6383                &spec.module_id,
6384                spec.protocol,
6385                StopNotice::NotSent,
6386                registry,
6387                runtime.forwarding.as_deref(),
6388                snapshot,
6389                &runtime.terminal_ring,
6390                &runtime.spawn_events,
6391                child,
6392                runtime.drain_timeout,
6393                ModuleState::Stopped,
6394                None,
6395            )
6396            .await;
6397            let registration_released = result.is_ok();
6398            let _ = reply.send(result);
6399            if registration_released {
6400                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6401            }
6402            false
6403        }
6404        SupervisorCommand::Retire { reply } => {
6405            let result = async {
6406                let stop_notice = begin_forwarding_drain_if_configured(
6407                    spec,
6408                    runtime,
6409                    registry,
6410                    snapshot,
6411                    None,
6412                    RouteCloseReason::Disable,
6413                )
6414                .await?;
6415                drain_optional_child(
6416                    &spec.module_id,
6417                    spec.protocol,
6418                    stop_notice,
6419                    registry,
6420                    runtime.forwarding.as_deref(),
6421                    snapshot,
6422                    &runtime.terminal_ring,
6423                    &runtime.spawn_events,
6424                    child,
6425                    runtime.drain_timeout,
6426                    ModuleState::Stopped,
6427                    None,
6428                )
6429                .await
6430            }
6431            .await;
6432            let registration_released = result.is_ok();
6433            let _ = reply.send(result);
6434            if registration_released {
6435                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6436            }
6437            false
6438        }
6439        SupervisorCommand::Restart {
6440            drain_timeout_ms,
6441            received_at_generation,
6442            queued_at,
6443            reply,
6444        } => {
6445            // Without this line a restart that waited in the queue (behind a
6446            // health probe cycle or another command) was invisible: the log
6447            // showed only the drain timing out, minutes after the operator's call.
6448            info!(
6449                module_id = %spec.module_id,
6450                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6451                "restart command dequeued"
6452            );
6453            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6454            // caller whose own request lane rides the module being restarted: the
6455            // caller's in-flight request keeps the drain from quiescing, the drain
6456            // keeps the restart from completing, and the completion keeps the reply
6457            // from releasing the caller — so the drain always timed out and cut the
6458            // initiator with a GOODBYE, even on a healthy module. Replying once the
6459            // restart is validated lets a self-lane caller settle, which is exactly
6460            // what makes the drain succeed. Completion is observable via
6461            // supervisor.list / module status; a post-ack failure lands the module
6462            // in a visible terminal state below rather than in a reply nobody can
6463            // receive.
6464            let validation = match lock_snapshot(snapshot) {
6465                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6466                    module_id: spec.module_id.clone(),
6467                }),
6468                Ok(_) => Ok(()),
6469                Err(err) => Err(err),
6470            };
6471            let initiated = validation.is_ok();
6472            let _ = reply.send(validation);
6473            // A restart asks for a fresh process. Commands run one at a time,
6474            // so a restart queued behind another restart (two operator calls
6475            // in quick succession) is dequeued the moment the first one has
6476            // spawned its replacement -- before that process has sent HELLO.
6477            // Running it would drain and kill the process the first restart
6478            // just produced, which is the opposite of what both callers asked
6479            // for. If a process spawned after this request was received is
6480            // still supervised, the request is already satisfied. Not when the
6481            // configuration changed since that spawn: then the newer process
6482            // predates the spec this restart may exist to apply.
6483            let satisfied_by_generation = if initiated && child.is_some() {
6484                lock_snapshot(snapshot).ok().and_then(|state| {
6485                    (state.spawn_generation > received_at_generation
6486                        && !state.configuration_updated_since_spawn)
6487                        .then_some(state.spawn_generation)
6488                })
6489            } else {
6490                None
6491            };
6492            let satisfied_by_pending = initiated
6493                && child.is_none()
6494                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6495                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6496                    if pending {
6497                        state.coalesced_restart_pending = true;
6498                    }
6499                    pending
6500                });
6501            if satisfied_by_pending {
6502                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6503            } else if let Some(generation) = satisfied_by_generation {
6504                info!(
6505                    module_id = %spec.module_id,
6506                    received_at_generation,
6507                    "restart already satisfied by generation {generation}; not restarting again"
6508                );
6509            } else if initiated {
6510                // Precedence: this restart's operator override, else the module's
6511                // configured budget (already resolved into the runtime).
6512                let drain_timeout = drain_timeout_ms
6513                    .map(Duration::from_millis)
6514                    .unwrap_or(runtime.drain_timeout);
6515                if let Err(err) = restart_child(
6516                    spec,
6517                    runtime,
6518                    registry,
6519                    process_liveness,
6520                    snapshot,
6521                    child,
6522                    drain_timeout,
6523                )
6524                .await
6525                {
6526                    warn!(
6527                        module_id = %spec.module_id,
6528                        error = %err,
6529                        "operator restart failed after initiation ack; module state carries the outcome"
6530                    );
6531                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6532                        state.state = ModuleState::Failed;
6533                        clear_current_process_facts(state);
6534                    });
6535                }
6536            }
6537            true
6538        }
6539        SupervisorCommand::Reload { reply } => {
6540            let result =
6541                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6542            if result.is_ok()
6543                && runtime
6544                    .scheduled_respawn
6545                    .lock()
6546                    .unwrap_or_else(|p| p.into_inner())
6547                    .as_ref()
6548                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6549            {
6550                *runtime
6551                    .deferred_reload_reply
6552                    .lock()
6553                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6554            } else {
6555                let _ = reply.send(result);
6556            }
6557            true
6558        }
6559        SupervisorCommand::SetEnabled { enabled, reply } => {
6560            let result = set_child_enabled(
6561                spec,
6562                runtime,
6563                registry,
6564                process_liveness,
6565                snapshot,
6566                child,
6567                enabled,
6568            )
6569            .await;
6570            let _ = reply.send(result);
6571            true
6572        }
6573        SupervisorCommand::UpdateConfiguration {
6574            spec: next_spec,
6575            health,
6576            drain_timeout_ms,
6577            reply,
6578        } => {
6579            if let Some(handle) = &runtime.supervisor_handle {
6580                handle.apply_identity_configuration(&next_spec);
6581            }
6582            *spec = next_spec;
6583            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6584                state.configuration_updated_since_spawn = true;
6585            });
6586            let health_changed = runtime.health != health;
6587            runtime.health = health;
6588            // Reset the cadence and old endpoint's failure streak on a live
6589            // health-policy change rather than waiting for its old deadline.
6590            if health_changed {
6591                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6592                    state.health = ModuleHealthStatus::default();
6593                });
6594            }
6595            runtime.drain_timeout = drain_timeout_ms
6596                .map(Duration::from_millis)
6597                .unwrap_or(runtime.default_drain_timeout);
6598            *runtime
6599                .effective_drain_timeout
6600                .lock()
6601                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6602            let _ = reply.send(());
6603            true
6604        }
6605        SupervisorCommand::Swap {
6606            ready_timeout,
6607            reply,
6608        } => {
6609            let end = swap::run_swap(
6610                spec,
6611                runtime,
6612                registry,
6613                process_liveness,
6614                snapshot,
6615                child,
6616                commands,
6617                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6618                reply,
6619            )
6620            .await;
6621            requeued.extend(end.requeue);
6622            true
6623        }
6624    }
6625}
6626
6627async fn restart_child(
6628    spec: &ModuleSpec,
6629    runtime: &SupervisorRuntimeConfig,
6630    registry: &Registry,
6631    process_liveness: &SupervisorProcessLiveness,
6632    snapshot: &SharedSnapshot,
6633    child: &mut Option<SupervisedChild>,
6634    drain_timeout: Duration,
6635) -> Result<(), SuperviseError> {
6636    // Restart cycles a running module; it must not silently start a disabled one.
6637    if !lock_snapshot(snapshot)?.enabled {
6638        return Err(SuperviseError::Disabled {
6639            module_id: spec.module_id.clone(),
6640        });
6641    }
6642    let stop_notice = begin_forwarding_drain_with_timeout(
6643        spec,
6644        runtime,
6645        registry,
6646        snapshot,
6647        None,
6648        RouteCloseReason::Restart,
6649        drain_timeout,
6650    )
6651    .await?;
6652
6653    if child.is_some() {
6654        drain_optional_child(
6655            &spec.module_id,
6656            spec.protocol,
6657            stop_notice,
6658            registry,
6659            runtime.forwarding.as_deref(),
6660            snapshot,
6661            &runtime.terminal_ring,
6662            &runtime.spawn_events,
6663            child,
6664            drain_timeout,
6665            ModuleState::Restarting,
6666            Some(true),
6667        )
6668        .await?;
6669    } else {
6670        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6671            state.enabled = true;
6672            state.state = ModuleState::Restarting;
6673            clear_current_process_facts(state);
6674        })?;
6675        release_dead_registration(
6676            registry,
6677            runtime.forwarding.as_deref(),
6678            snapshot,
6679            &spec.module_id,
6680        )
6681        .await?;
6682    }
6683
6684    reset_restart_count(snapshot, &spec.module_id)?;
6685    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6686    schedule_respawn(
6687        runtime,
6688        snapshot,
6689        &spec.module_id,
6690        runtime.restart_policy.backoff,
6691        RespawnKind::Spawn,
6692    )
6693}
6694
6695async fn reload_child(
6696    spec: &ModuleSpec,
6697    runtime: &SupervisorRuntimeConfig,
6698    registry: &Registry,
6699    process_liveness: &SupervisorProcessLiveness,
6700    snapshot: &SharedSnapshot,
6701    child: &mut Option<SupervisedChild>,
6702) -> Result<(), SuperviseError> {
6703    // Reload cycles a running module; it must not silently start a disabled one.
6704    if !lock_snapshot(snapshot)?.enabled {
6705        return Err(SuperviseError::Disabled {
6706            module_id: spec.module_id.clone(),
6707        });
6708    }
6709    let stop_notice = begin_forwarding_drain(
6710        spec,
6711        runtime,
6712        registry,
6713        snapshot,
6714        Some(true),
6715        RouteCloseReason::Reload,
6716    )
6717    .await?;
6718
6719    if child.is_some() {
6720        drain_optional_child(
6721            &spec.module_id,
6722            spec.protocol,
6723            stop_notice,
6724            registry,
6725            runtime.forwarding.as_deref(),
6726            snapshot,
6727            &runtime.terminal_ring,
6728            &runtime.spawn_events,
6729            child,
6730            runtime.drain_timeout,
6731            ModuleState::Restarting,
6732            Some(true),
6733        )
6734        .await?;
6735    } else {
6736        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6737            state.enabled = true;
6738            state.state = ModuleState::Restarting;
6739            clear_current_process_facts(state);
6740        })?;
6741        release_dead_registration(
6742            registry,
6743            runtime.forwarding.as_deref(),
6744            snapshot,
6745            &spec.module_id,
6746        )
6747        .await?;
6748    }
6749
6750    reset_restart_count(snapshot, &spec.module_id)?;
6751    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6752    schedule_respawn(
6753        runtime,
6754        snapshot,
6755        &spec.module_id,
6756        runtime.restart_policy.backoff,
6757        RespawnKind::Reload,
6758    )
6759}
6760
6761async fn finish_reload_child(
6762    spec: &ModuleSpec,
6763    runtime: &SupervisorRuntimeConfig,
6764    registry: &Registry,
6765    process_liveness: &SupervisorProcessLiveness,
6766    snapshot: &SharedSnapshot,
6767    child: &mut Option<SupervisedChild>,
6768) -> Result<(), SuperviseError> {
6769    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6770    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6771        Ok(next_child) => next_child,
6772        Err(err) => {
6773            return handle_reload_spawn_failure(
6774                spec,
6775                runtime,
6776                process_liveness,
6777                snapshot,
6778                child,
6779                format!("new child failed to spawn: {err}"),
6780            )
6781            .await;
6782        }
6783    };
6784    *child = Some(next_child);
6785
6786    let wait_outcome = {
6787        let active_child = child.as_mut().expect("new reload child was just stored");
6788        wait_for_registration_after_reload(
6789            registry,
6790            &spec.module_id,
6791            snapshot,
6792            active_child,
6793            REGISTRY_RELEASE_TIMEOUT,
6794        )
6795        .await?
6796    };
6797
6798    match wait_outcome {
6799        RegistrationWaitOutcome::Registered => {
6800            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6801            Ok(())
6802        }
6803        RegistrationWaitOutcome::Exited(exit_report) => {
6804            if let Some(active_child) = child.as_mut() {
6805                active_child.drain_stderr(&spec.module_id).await;
6806            }
6807            // Keep the reaped child's roster guard until its terminal is written.
6808            // Shutdown waits on that guard, not on the child Option used for respawn.
6809            let mut exited_child = child.take().expect("exited reload child is still stored");
6810            #[cfg(test)]
6811            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6812                gate.reached.notify_one();
6813                gate.resume.notified().await;
6814            }
6815            let result = handle_reload_child_registration_failure(
6816                spec,
6817                runtime,
6818                registry,
6819                process_liveness,
6820                snapshot,
6821                child,
6822                ReloadRegistrationFailure {
6823                    exit_report: registration_failure_exit_report(exit_report),
6824                    reason: exited_child
6825                        .spawn_failure
6826                        .clone()
6827                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6828                },
6829            )
6830            .await;
6831            exited_child.release_roster();
6832            result
6833        }
6834        RegistrationWaitOutcome::TimedOut => {
6835            let mut timed_out_child = child
6836                .take()
6837                .expect("timed-out reload child is still running");
6838            timed_out_child
6839                .start_kill()
6840                .map_err(|source| SuperviseError::Kill {
6841                    module_id: spec.module_id.clone(),
6842                    source,
6843                })?;
6844            let status = timed_out_child
6845                .wait()
6846                .await
6847                .map_err(|source| SuperviseError::Wait {
6848                    module_id: spec.module_id.clone(),
6849                    source,
6850                })?;
6851            timed_out_child.drain_stderr(&spec.module_id).await;
6852            handle_reload_child_registration_failure(
6853                spec,
6854                runtime,
6855                registry,
6856                process_liveness,
6857                snapshot,
6858                child,
6859                ReloadRegistrationFailure {
6860                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6861                        snapshot,
6862                        &timed_out_child,
6863                        &status,
6864                    )),
6865                    reason: format!(
6866                        "new child did not register within {:?}",
6867                        REGISTRY_RELEASE_TIMEOUT
6868                    ),
6869                },
6870            )
6871            .await
6872        }
6873    }
6874}
6875
6876async fn set_child_enabled(
6877    spec: &ModuleSpec,
6878    runtime: &SupervisorRuntimeConfig,
6879    registry: &Registry,
6880    process_liveness: &SupervisorProcessLiveness,
6881    snapshot: &SharedSnapshot,
6882    child: &mut Option<SupervisedChild>,
6883    enabled: bool,
6884) -> Result<bool, SuperviseError> {
6885    let (current_enabled, current_state, respawn_pending) = {
6886        let state = lock_snapshot(snapshot)?;
6887        (state.enabled, state.state, state.respawn_pending)
6888    };
6889    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6890    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6891    // clean (Stopped) has no live process and no other in-band recovery — the
6892    // operator's start is the explicit recovery act and resets the budget. Without
6893    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6894    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6895    // the one providing every agent's shell.
6896    let revive_terminal = enabled
6897        && current_enabled
6898        && child.is_none()
6899        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6900            || (current_state == ModuleState::Restarting && !respawn_pending));
6901    if current_enabled == enabled && !revive_terminal {
6902        return Ok(false);
6903    }
6904
6905    if enabled {
6906        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6907            state.enabled = true;
6908            state.state = ModuleState::Starting;
6909            clear_current_process_facts(state);
6910        })?;
6911        #[cfg(test)]
6912        if runtime.test_seed_stale_facts_before_enable_spawn {
6913            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6914                state.process_alive = true;
6915                state.pid = Some(41);
6916                state.spawned_at_ms = Some(42);
6917                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6918                state.spawned_file_identity = Some(SpawnedFileIdentity {
6919                    device: 43,
6920                    inode: 44,
6921                });
6922            })?;
6923        }
6924        release_dead_registration(
6925            registry,
6926            runtime.forwarding.as_deref(),
6927            snapshot,
6928            &spec.module_id,
6929        )
6930        .await?;
6931        reset_restart_count(snapshot, &spec.module_id)?;
6932        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6933        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6934            Ok(next_child) => next_child,
6935            Err(err) => {
6936                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6937                    state.state = ModuleState::Failed;
6938                    clear_current_process_facts(state);
6939                }) {
6940                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6941                }
6942                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6943                return Err(err);
6944            }
6945        };
6946        *child = Some(next_child);
6947        debug!(module_id = %spec.module_id, "supervised module enabled");
6948        Ok(true)
6949    } else {
6950        let stop_notice = begin_forwarding_drain_if_configured(
6951            spec,
6952            runtime,
6953            registry,
6954            snapshot,
6955            Some(false),
6956            RouteCloseReason::Disable,
6957        )
6958        .await?;
6959        drain_optional_child(
6960            &spec.module_id,
6961            spec.protocol,
6962            stop_notice,
6963            registry,
6964            runtime.forwarding.as_deref(),
6965            snapshot,
6966            &runtime.terminal_ring,
6967            &runtime.spawn_events,
6968            child,
6969            runtime.drain_timeout,
6970            ModuleState::Disabled,
6971            Some(false),
6972        )
6973        .await?;
6974        debug!(module_id = %spec.module_id, "supervised module disabled");
6975        Ok(true)
6976    }
6977}
6978
6979#[allow(clippy::too_many_arguments)]
6980async fn on_child_exit(
6981    spec: &ModuleSpec,
6982    policy: RestartPolicy,
6983    registry: &Registry,
6984    snapshot: &SharedSnapshot,
6985    terminal_ring: &Arc<Mutex<TerminalRing>>,
6986    spawn_events: &SpawnEventFeed,
6987    roster: &ChildRoster,
6988    exit_report: ExitReport,
6989) -> NextAction {
6990    // Once the daemon has begun shutting down, no exit is a crash to recover
6991    // from: the module is exiting because the daemon is going away (EOF on its
6992    // connection, or a service manager signalling the whole cgroup). Record it
6993    // as such and never schedule a respawn, which would only start a process
6994    // for the shutdown to end again.
6995    if roster.is_closed() {
6996        return on_child_exit_during_daemon_shutdown(
6997            spec,
6998            registry,
6999            snapshot,
7000            terminal_ring,
7001            spawn_events,
7002            exit_report,
7003        )
7004        .await;
7005    }
7006    // Every stop the supervisor itself asks for (operator stop, disable,
7007    // restart, reload, swap, a health restart, a drain that runs out of budget)
7008    // takes the child out of the supervise loop and reaps it in
7009    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
7010    // that reaches this point was not requested by the daemon.
7011    //
7012    // For a subc-wire module a clean exit is still a stop: those modules are
7013    // written to re-raise SIGTERM, so a stray outside signal already reads as a
7014    // crash, and exiting 0 is a deliberate choice the module made. A
7015    // `protocol: "none"` module is a stock program we cannot change, and many
7016    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
7017    // stop would leave the module down for good after any stray signal, so it
7018    // goes through the crash path instead: it spends restart budget, respawns
7019    // with the crash backoff, and ends `failed` when the budget runs out.
7020    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
7021        && running_protocol(spec, snapshot) == ModuleProtocol::None;
7022    match exit_report.kind {
7023        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
7024            info!(
7025                module_id = %spec.module_id,
7026                exit_code = ?exit_report.code,
7027                exit_signal = ?exit_report.signal,
7028                "supervised module exited cleanly"
7029            );
7030            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7031                state.state = ModuleState::Stopped;
7032                clear_current_process_facts(state);
7033                state.last_exit = Some(exit_report.clone());
7034            }) {
7035                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
7036            }
7037            record_terminal(
7038                &spec.module_id,
7039                terminal_ring,
7040                spawn_events,
7041                &exit_report,
7042                TerminalDisposition::Stopped,
7043            );
7044            let registration_released = match wait_for_registration_release(
7045                registry,
7046                &spec.module_id,
7047                REGISTRY_RELEASE_TIMEOUT,
7048            )
7049            .await
7050            {
7051                Ok(()) => true,
7052                Err(err) => {
7053                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
7054                    false
7055                }
7056            };
7057            NextAction::Stop {
7058                registration_released,
7059            }
7060        }
7061        ExitKind::Clean | ExitKind::Crash => {
7062            if unrequested_clean_exit_of_protocol_none {
7063                warn!(
7064                    module_id = %spec.module_id,
7065                    exit_code = ?exit_report.code,
7066                    exit_signal = ?exit_report.signal,
7067                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
7068                );
7069            } else {
7070                warn!(
7071                    module_id = %spec.module_id,
7072                    exit_code = ?exit_report.code,
7073                    exit_signal = ?exit_report.signal,
7074                    "supervised module exited abnormally (crash)"
7075                );
7076            }
7077            let mut restart_schedule = None;
7078            let mut disposition = TerminalDisposition::Disabled;
7079            // Set only when the budget is what stopped the module, so the
7080            // terminal record says which limit was hit rather than leaving
7081            // `failed` to be read as "crashed once, badly".
7082            let mut disposition_detail = lock_snapshot(snapshot)
7083                .ok()
7084                .and_then(|mut state| state.spawn_failure.take());
7085            let now = Instant::now();
7086            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7087                clear_current_process_facts(state);
7088                state.last_exit = Some(exit_report.clone());
7089                if state.enabled {
7090                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
7091                        state.state = ModuleState::Restarting;
7092                        restart_schedule = Some(schedule);
7093                        disposition = TerminalDisposition::Restarting;
7094                    } else {
7095                        disposition = TerminalDisposition::Failed;
7096                        let budget = policy.budget_exhausted_detail();
7097                        disposition_detail =
7098                            Some(disposition_detail.take().map_or_else(
7099                                || budget.clone(),
7100                                |cause| format!("{cause}; {budget}"),
7101                            ));
7102                    }
7103                } else {
7104                    state.state = ModuleState::Disabled;
7105                    disposition = TerminalDisposition::Disabled;
7106                }
7107            }) {
7108                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7109                return NextAction::Stop {
7110                    registration_released: false,
7111                };
7112            }
7113            if disposition == TerminalDisposition::Failed {
7114                // The window is in the message, not only in the fields: this line
7115                // is read in a scrollback where a bare `max_restarts=3` reads as a
7116                // lifetime cap and sends the operator looking for three crashes
7117                // that never happened together.
7118                error!(
7119                    module_id = %spec.module_id,
7120                    max_restarts = policy.max_restarts,
7121                    window_secs = policy.window.as_secs(),
7122                    "module stopped: {}",
7123                    policy.budget_exhausted_detail()
7124                );
7125            }
7126            let budget_exhausted = disposition == TerminalDisposition::Failed;
7127            let record_exit = || {
7128                record_terminal_with_detail(
7129                    &spec.module_id,
7130                    terminal_ring,
7131                    spawn_events,
7132                    &exit_report,
7133                    disposition,
7134                    disposition_detail,
7135                );
7136            };
7137            if budget_exhausted {
7138                // Publish Failed only after its terminal record is available.
7139                // Recording takes the event-feed lock, then ring -> journal
7140                // writer (with file I/O), all without the hot snapshot lock.
7141                // No lock is held when the final snapshot update runs, nor
7142                // across the registration-release await below. Commands and
7143                // health actions run on this same supervisor task, so none can
7144                // act on the old state during the write; process facts already
7145                // say the child is dead to concurrent liveness readers.
7146                // A journal error is retained in history, not returned. If
7147                // recording panics, still publish Failed before resuming the
7148                // original unwind rather than leaving a dead child Running.
7149                let recorded = std::panic::catch_unwind(std::panic::AssertUnwindSafe(record_exit));
7150                if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7151                    state.state = ModuleState::Failed;
7152                }) {
7153                    error!(module_id = %spec.module_id, error = %err, "failed to publish exhausted restart budget");
7154                }
7155                if let Err(panic) = recorded {
7156                    std::panic::resume_unwind(panic);
7157                }
7158            } else {
7159                record_exit();
7160            }
7161
7162            if let Some(schedule) = restart_schedule {
7163                NextAction::Restart {
7164                    schedule: Some(schedule),
7165                }
7166            } else {
7167                let registration_released = match wait_for_registration_release(
7168                    registry,
7169                    &spec.module_id,
7170                    REGISTRY_RELEASE_TIMEOUT,
7171                )
7172                .await
7173                {
7174                    Ok(()) => true,
7175                    Err(err) => {
7176                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7177                        false
7178                    }
7179                };
7180                NextAction::Stop {
7181                    registration_released,
7182                }
7183            }
7184        }
7185        ExitKind::DeliberateSeverance => {
7186            warn!(
7187                module_id = %spec.module_id,
7188                exit_code = ?exit_report.code,
7189                exit_signal = ?exit_report.signal,
7190                "supervised module exited after deliberate connection severance"
7191            );
7192            let mut should_restart = false;
7193            let mut disposition = TerminalDisposition::Disabled;
7194            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7195                clear_current_process_facts(state);
7196                state.last_exit = Some(exit_report.clone());
7197                state.lifetime_restarts += 1;
7198                if state.enabled {
7199                    state.state = ModuleState::Restarting;
7200                    should_restart = true;
7201                    disposition = TerminalDisposition::Restarting;
7202                } else {
7203                    state.state = ModuleState::Disabled;
7204                }
7205            }) {
7206                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7207                return NextAction::Stop {
7208                    registration_released: false,
7209                };
7210            }
7211            record_terminal(
7212                &spec.module_id,
7213                terminal_ring,
7214                spawn_events,
7215                &exit_report,
7216                disposition,
7217            );
7218
7219            if should_restart {
7220                NextAction::Restart { schedule: None }
7221            } else {
7222                let registration_released = match wait_for_registration_release(
7223                    registry,
7224                    &spec.module_id,
7225                    REGISTRY_RELEASE_TIMEOUT,
7226                )
7227                .await
7228                {
7229                    Ok(()) => true,
7230                    Err(err) => {
7231                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7232                        false
7233                    }
7234                };
7235                NextAction::Stop {
7236                    registration_released,
7237                }
7238            }
7239        }
7240    }
7241}
7242
7243async fn on_child_exit_during_daemon_shutdown(
7244    spec: &ModuleSpec,
7245    registry: &Registry,
7246    snapshot: &SharedSnapshot,
7247    terminal_ring: &Arc<Mutex<TerminalRing>>,
7248    spawn_events: &SpawnEventFeed,
7249    exit_report: ExitReport,
7250) -> NextAction {
7251    info!(
7252        module_id = %spec.module_id,
7253        exit_code = ?exit_report.code,
7254        exit_signal = ?exit_report.signal,
7255        exit_kind = ?exit_report.kind,
7256        "supervised module exited during daemon shutdown; not restarting it"
7257    );
7258    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7259        state.state = ModuleState::Stopped;
7260        clear_current_process_facts(state);
7261        state.last_exit = Some(exit_report.clone());
7262    }) {
7263        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7264    }
7265    record_terminal(
7266        &spec.module_id,
7267        terminal_ring,
7268        spawn_events,
7269        &exit_report,
7270        TerminalDisposition::DaemonShutdown,
7271    );
7272    let registration_released =
7273        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7274            .await
7275            .is_ok();
7276    NextAction::Stop {
7277        registration_released,
7278    }
7279}
7280
7281fn record_wait_error_terminal(
7282    module_id: &str,
7283    terminal_ring: &Arc<Mutex<TerminalRing>>,
7284    spawn_events: &SpawnEventFeed,
7285) {
7286    record_terminal(
7287        module_id,
7288        terminal_ring,
7289        spawn_events,
7290        &wait_error_exit_report(),
7291        TerminalDisposition::Failed,
7292    );
7293}
7294
7295fn record_terminal(
7296    module_id: &str,
7297    terminal_ring: &Arc<Mutex<TerminalRing>>,
7298    spawn_events: &SpawnEventFeed,
7299    exit_report: &ExitReport,
7300    disposition: TerminalDisposition,
7301) {
7302    record_terminal_with_detail(
7303        module_id,
7304        terminal_ring,
7305        spawn_events,
7306        exit_report,
7307        disposition,
7308        None,
7309    );
7310}
7311
7312/// The ring lock is held only to capture the read (see
7313/// `TerminalJournal::capture_read`), so this module's exits keep recording
7314/// while the journal files are read. Blocking: it reads files.
7315fn durable_terminal_history_of(
7316    terminal_ring: &Mutex<TerminalRing>,
7317    module_id: &str,
7318) -> subc_control::TerminalHistory {
7319    let read = terminal_ring
7320        .lock()
7321        .unwrap_or_else(|p| p.into_inner())
7322        .capture_durable_history();
7323    read.read(module_id)
7324}
7325
7326fn record_terminal_with_detail(
7327    module_id: &str,
7328    terminal_ring: &Arc<Mutex<TerminalRing>>,
7329    spawn_events: &SpawnEventFeed,
7330    exit_report: &ExitReport,
7331    disposition: TerminalDisposition,
7332    disposition_detail: Option<String>,
7333) {
7334    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7335    let record = TerminalRecord {
7336        exit_code: exit_report.code,
7337        exit_signal: exit_report.signal,
7338        at_ms: exit_report.at_ms,
7339        disposition,
7340        exit_kind: exit_report.kind.into(),
7341        disposition_detail,
7342    };
7343    terminal_ring
7344        .lock()
7345        .unwrap_or_else(|poisoned| poisoned.into_inner())
7346        .record_exit(module_id, record);
7347}
7348
7349fn untrack_if_registration_released(
7350    process_liveness: &SupervisorProcessLiveness,
7351    registry: &Registry,
7352    module_id: &str,
7353    snapshot: &SharedSnapshot,
7354) {
7355    match registry.get_module(module_id) {
7356        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7357        Ok(Some(_)) => {}
7358        Err(err) => {
7359            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7360        }
7361    }
7362}
7363
7364/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7365/// then apply the module's configured entries minus daemon-private capture keys.
7366///
7367/// Separated from `spawn_child` only so it can be asserted without spawning a
7368/// process — a duplicate of this logic in a test would pass while the real one
7369/// drifted, which is the defect class this function exists to avoid.
7370/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7371/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7372/// either and the argument would stop a stock binary from starting at all.
7373/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7374///
7375/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7376/// through [`apply_wire_spawn_args_for_role`].
7377#[cfg(test)]
7378fn apply_wire_spawn_args(
7379    command: &mut Command,
7380    spec: &ModuleSpec,
7381    connection_file_path: Option<&std::path::Path>,
7382    handle: Option<&SupervisorHandle>,
7383) -> Result<Option<NonceHandoff>, SuperviseError> {
7384    apply_wire_spawn_args_for_role(
7385        command,
7386        spec,
7387        connection_file_path,
7388        handle,
7389        SpawnRole::Plain,
7390    )
7391}
7392
7393/// The read end of a spawn's launch-nonce pipe, prepared by
7394/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7395/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7396/// handoff and keeps only the environment copy.
7397#[cfg(unix)]
7398type NonceHandoff = subc_os::LaunchNonceHandoff;
7399#[cfg(not(unix))]
7400type NonceHandoff = std::convert::Infallible;
7401
7402/// Prepare wire identity for a plain spawn or a swap candidate.
7403///
7404/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7405/// a separate candidate token so the still-serving incumbent and its consumers
7406/// keep their nonce. Both records are installed before the process exists, so
7407/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7408///
7409/// On Unix the nonce is delivered only through a pipe. It is written into
7410/// a pipe whose read end the child gets as descriptor 3, named by
7411/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7412/// process of the same user cannot read it with `ps eww`. That handoff is
7413/// returned rather than installed here, because installing it replaces
7414/// whatever the child has at descriptor 3 and so must be the last pre-exec
7415/// step, after the Linux cgroup placement that the caller registers later.
7416/// Windows retains the environment handoff until restricted handle inheritance
7417/// can be implemented outside std's process primitives.
7418fn apply_wire_spawn_args_for_role(
7419    command: &mut Command,
7420    spec: &ModuleSpec,
7421    connection_file_path: Option<&std::path::Path>,
7422    handle: Option<&SupervisorHandle>,
7423    role: SpawnRole,
7424) -> Result<Option<NonceHandoff>, SuperviseError> {
7425    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7426    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7427    // included: a daemon started from a module's process tree inherits it,
7428    // and passing it on would point the child at a descriptor it does not
7429    // have.
7430    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7431    // Remove inherited or configured copies too: withholding must mean absent.
7432    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7433    if spec.protocol == ModuleProtocol::None {
7434        return Ok(None);
7435    }
7436    if let Some(connection_file_path) = connection_file_path {
7437        command.arg(SUBC_ARG).arg(connection_file_path);
7438    }
7439
7440    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7441    // route.open attestation. Reserved modules additionally use the same nonce
7442    // for HELLO id-squatting protection. A respawn rotates both records.
7443    let nonce = generate_launch_nonce()?;
7444    if let Some(handle) = handle {
7445        match role {
7446            SpawnRole::Plain => {
7447                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7448                if spec.reserved {
7449                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7450                }
7451            }
7452            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7453        }
7454    }
7455    #[cfg(unix)]
7456    let handoff = {
7457        let handoff =
7458            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7459                program: spec.program.clone(),
7460                source,
7461                cgroup_path: None,
7462            })?;
7463        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7464        Some(handoff)
7465    };
7466    #[cfg(not(unix))]
7467    let handoff = None;
7468    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7469    // handle to this child without leaking it to concurrently spawned processes.
7470    #[cfg(not(unix))]
7471    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7472    Ok(handoff)
7473}
7474
7475fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7476    command.env_remove(CK_LOG_ENV);
7477    // The spawn role is the supervisor's to set, and only on a swap candidate
7478    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7479    // it, is what makes it absent on a plain spawn: the daemon's own
7480    // environment could carry it, and so could a spec built outside daemon
7481    // config (config refuses it as an `env` key). A module reading it on a
7482    // plain restart would pick the long swap budget and leave callers waiting.
7483    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7484    for (key, value) in &spec.env {
7485        // cortexkit-log currently exposes retention only as a Rust struct, not
7486        // environment names. These values are daemon-private sink metadata and
7487        // must never become a public child-process contract by being inherited.
7488        if matches!(
7489            key.as_str(),
7490            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7491        ) || key == SUBC_SPAWN_ROLE_ENV
7492        {
7493            continue;
7494        }
7495        command.env(key, value);
7496    }
7497}
7498
7499/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7500/// of a blue/green swap.
7501#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7502enum SpawnRole {
7503    Plain,
7504    SwapCandidate,
7505}
7506
7507/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7508/// `apply_child_env` has already removed the variable for every spawn.
7509fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7510    if role == SpawnRole::SwapCandidate {
7511        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7512    }
7513}
7514
7515fn spawn_child(
7516    spec: &ModuleSpec,
7517    connection_file_path: Option<&std::path::Path>,
7518    handle: Option<&SupervisorHandle>,
7519    ring: &Arc<Mutex<StderrRing>>,
7520    capture_logs_dir: Option<&std::path::Path>,
7521    roster: &ChildRoster,
7522    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7523) -> Result<SupervisedChild, SuperviseError> {
7524    spawn_child_in_slot(
7525        spec,
7526        connection_file_path,
7527        handle,
7528        ring,
7529        capture_logs_dir,
7530        roster,
7531        #[cfg(target_os = "linux")]
7532        cgroup_placement,
7533        SpawnRole::Plain,
7534        false,
7535    )
7536}
7537
7538/// Spawn one process of `spec` into a slot.
7539///
7540/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7541/// A swap candidate needs a different cgroup from the process it is replacing,
7542/// which is still alive: in the same cgroup the two would be one kill domain,
7543/// and killing a failed candidate could take the incumbent with it.
7544///
7545/// The stderr capture file is `<module_id>.stderr.log` for every process of
7546/// the module, whichever slot it is in, because that is the one file
7547/// `ck module logs` reads. During a swap's overlap both processes append to it;
7548/// the daemon writes whole lines, so the two interleave by line, which is also
7549/// the merged view an operator wants while a swap runs.
7550#[allow(clippy::too_many_arguments)]
7551fn spawn_child_in_slot(
7552    spec: &ModuleSpec,
7553    connection_file_path: Option<&std::path::Path>,
7554    handle: Option<&SupervisorHandle>,
7555    ring: &Arc<Mutex<StderrRing>>,
7556    capture_logs_dir: Option<&std::path::Path>,
7557    roster: &ChildRoster,
7558    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7559    role: SpawnRole,
7560    alternate_slot: bool,
7561) -> Result<SupervisedChild, SuperviseError> {
7562    if roster.is_closed() {
7563        return Err(SuperviseError::Spawn {
7564            program: spec.program.clone(),
7565            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7566            cgroup_path: None,
7567        });
7568    }
7569    #[cfg(target_os = "linux")]
7570    let cgroup_name = {
7571        // Slot names alone are not kill domains: a retired incumbent may still
7572        // be draining when a later enable/restart spawns into the same slot.
7573        // Decimal entropy keeps the suffix unambiguous; Placement performs
7574        // the module-id escaping and constructs the filesystem path.
7575        if cgroup_placement.is_none() {
7576            swap::cgroup_name(&spec.module_id, alternate_slot)
7577        } else {
7578            let nonce = generate_launch_nonce()?;
7579            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7580            // Leave room for byte escaping and the suffix under NAME_MAX. The
7581            // label is only for humans; the nonce identifies the kill domain.
7582            let mut end = spec.module_id.len().min(64);
7583            while !spec.module_id.is_char_boundary(end) {
7584                end -= 1;
7585            }
7586            format!(
7587                "{}_{suffix}",
7588                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7589            )
7590        }
7591    };
7592    #[cfg(not(target_os = "linux"))]
7593    let _ = alternate_slot;
7594    #[cfg(target_os = "macos")]
7595    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7596    #[cfg(not(target_os = "macos"))]
7597    let mut command = Command::new(&spec.program);
7598    command.args(&spec.args);
7599    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7600    // that is the whole of the intent, so remove that one key rather than the
7601    // environment.
7602    //
7603    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7604    // and took the POSIX environment with it. Modules spawned that way had no
7605    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7606    // logging:
7607    //
7608    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7609    //     both unset it fell back to the temp dir alone and `ck` could not find
7610    //     a daemon running on the same machine from inside any module's process
7611    //     tree — reporting a path the file has never lived at, which reads as
7612    //     "the daemon did not write its file".
7613    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7614    //     the RELATIVE `.local/share`, so a module deriving its own store path
7615    //     resolved it against its own CWD. That is the store-fragmentation
7616    //     defect the daemon already refuses in config (`parse_doc` rejects a
7617    //     relative `storage.data_home`) arriving by derivation instead.
7618    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7619    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7620    //     quietly rather than erroring.
7621    //
7622    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7623    // offered one candidate under /tmp while the file sat in /run/user/1000.
7624    //
7625    // A configured module is unaffected either way: `module_spec()` puts the
7626    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7627    // wins over anything ambient.
7628    apply_child_env(&mut command, spec);
7629    apply_spawn_role(&mut command, role);
7630    let nonce_handoff =
7631        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7632
7633    #[cfg(target_os = "linux")]
7634    let cgroup_path = cgroup_placement
7635        .map(|placement| placement.module_path(&cgroup_name))
7636        .transpose()
7637        .map_err(|source| SuperviseError::Cgroup {
7638            module_id: spec.module_id.clone(),
7639            source,
7640        })?;
7641    #[cfg(not(target_os = "linux"))]
7642    let cgroup_path: Option<PathBuf> = None;
7643    #[cfg(target_os = "linux")]
7644    if let Some(path) = &cgroup_path {
7645        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7646            if let Some(placement) = cgroup_placement {
7647                remove_module_cgroup(placement, &cgroup_name);
7648            }
7649            return Err(error);
7650        }
7651    }
7652
7653    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7654        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7655        match ChildOutputSink::open(&path, capture_retention(spec)) {
7656            Ok(sink) => sink,
7657            Err(error) => {
7658                warn!(
7659                    module_id = %spec.module_id,
7660                    path = %path.display(),
7661                    error = %error,
7662                    "could not open child output capture file; forwarding to stderr"
7663                );
7664                ChildOutputSink::Stderr
7665            }
7666        }
7667    } else {
7668        ChildOutputSink::Stderr
7669    };
7670
7671    command.stdout(Stdio::piped());
7672    command.stderr(Stdio::piped());
7673    command.kill_on_drop(true);
7674    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7675    // before exec). In the daemon's group, a service manager that kills the
7676    // job's process group when the daemon exits (launchd's default) killed
7677    // every module at the same moment its control connection closed, so no
7678    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7679    // module is reached only by the daemon: the EOF it sees when its
7680    // connection closes, and the bounded stop in `child_roster` for anything
7681    // still running after that. On Linux this composes with the cgroup
7682    // placement above: that is a pre_exec write to cgroup.procs, std performs
7683    // setpgid in the child before running pre_exec callbacks, and the two
7684    // change independent process attributes.
7685    //
7686    // stdin is /dev/null because a process outside the terminal's foreground
7687    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7688    // by hand would otherwise hand down. Under a service manager stdin is
7689    // already /dev/null.
7690    #[cfg(unix)]
7691    command.process_group(0);
7692    command.stdin(Stdio::null());
7693    // The LAST pre-exec step, after the cgroup placement above: installing the
7694    // nonce at descriptor 3 replaces whatever the child had there, which could
7695    // be the descriptor an earlier step writes through.
7696    #[cfg(unix)]
7697    if let Some(handoff) = nonce_handoff {
7698        handoff.install_last(command.as_std_mut());
7699    }
7700    #[cfg(not(unix))]
7701    let _ = nonce_handoff;
7702
7703    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7704    // cannot run a single instruction -- and therefore cannot spawn a
7705    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7706    // other two steps and why the window matters.
7707    #[cfg(windows)]
7708    subc_jobobject::suspend_on_create_async(&mut command);
7709    let mut child = match command.spawn() {
7710        Ok(child) => child,
7711        Err(source) => {
7712            #[cfg(target_os = "linux")]
7713            if let Some(placement) = cgroup_placement {
7714                remove_module_cgroup(placement, &cgroup_name);
7715            }
7716            return Err(SuperviseError::Spawn {
7717                program: spec.program.clone(),
7718                source,
7719                cgroup_path,
7720            });
7721        }
7722    };
7723    // The parent must close its writer now: the acknowledgement pipe reports EOF
7724    // only when every writer is gone, and the child's copy closes when the
7725    // trampoline replaces itself with the module. Command holds only an integer
7726    // in its pre_exec callback, not another writer.
7727    #[cfg(target_os = "macos")]
7728    drop(exec_ack);
7729
7730    // Containment, steps 2 and 3: assign while suspended, then resume.
7731    #[cfg(windows)]
7732    let job = contain_spawned_child(&child, spec)?;
7733    let spawned_at_ms = unix_ms_now();
7734    let spawned_from = spec.program.clone();
7735    let spawned_file_identity = spawned_file_identity(&spawned_from);
7736    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7737        program: spec.program.clone(),
7738        source: io::Error::other("spawned child exposed no live pid"),
7739        cgroup_path: cgroup_path.clone(),
7740    })?;
7741    let process_start_time = crate::provenance::process_start_time(pid);
7742    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7743    #[cfg(all(test, target_os = "macos"))]
7744    privacy_exec_boundary_tests::before_image_sample(spec, pid);
7745    // Unix spawn returns after exec's error pipe closes. The kernel image is
7746    // therefore the executable to compare during a future orphan sweep: PATH
7747    // lookup and shebang interpretation may select a different file from the
7748    // configured program. Keep the literal program's identity for provenance,
7749    // but never use it as proof that a recorded pid may be signalled.
7750    let recorded_image = observe_spawned_image(pid);
7751    // spawn() confirms only the first exec, into the trampoline. Never persist
7752    // the trampoline image; the asynchronous acknowledgement publishes the
7753    // module image once the trampoline has replaced itself with the module.
7754    #[cfg(target_os = "macos")]
7755    let recorded_image = if privacy_exec.is_some() {
7756        None
7757    } else {
7758        recorded_image
7759    };
7760    #[cfg(target_os = "linux")]
7761    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7762    #[cfg(not(target_os = "linux"))]
7763    let recorded_cgroup_name = None;
7764    let roster_guard = roster.admit(
7765        spec.module_id.clone(),
7766        pid,
7767        spec.protocol,
7768        process_start_time,
7769        crate::child_roster::RecordedIdentity {
7770            start_time: recorded_image.map(|image| image.start_time),
7771            executable: recorded_image
7772                .and_then(|image| image.executable)
7773                .map(crate::live_children::ExecutableIdentity::from),
7774            cgroup_name: recorded_cgroup_name,
7775            #[cfg(target_os = "linux")]
7776            cgroup_placement: cgroup_placement.cloned(),
7777        },
7778    );
7779    // The check at the top of this function can pass just before daemon
7780    // shutdown begins, and the process is only in the roster from here on.
7781    // The shutdown stop returns as soon as it finds the roster empty, so a
7782    // process admitted after that look would outlive the daemon. The roster
7783    // is closed before the stop first reads it and admission happens under
7784    // the roster's lock, so either the stop sees this process or this check
7785    // sees the roster closed: end the process now rather than start a module
7786    // the daemon is about to stop.
7787    if roster.is_closed() {
7788        // This child was never admitted, so there is no module protocol shutdown to wait for.
7789        #[cfg(target_os = "linux")]
7790        kill_module_cgroup(cgroup_placement, &cgroup_name);
7791        if let Err(error) = child.start_kill() {
7792            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7793        }
7794        #[cfg(target_os = "linux")]
7795        if let Some(placement) = cgroup_placement {
7796            // This spawn was never admitted, so shutdown has no roster entry
7797            // to await. Do not detach its cleanup: the runtime could exit
7798            // before that task reaps the rejected child and removes its group.
7799            while matches!(child.try_wait(), Ok(None)) {
7800                std::thread::yield_now();
7801            }
7802            if matches!(
7803                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7804                subc_cgroup::KillOutcome::Killed
7805            ) {
7806                if let Ok(path) = placement.module_path(&cgroup_name) {
7807                    while std::fs::read_to_string(path.join("cgroup.events"))
7808                        .ok()
7809                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7810                    {
7811                        std::thread::yield_now();
7812                    }
7813                }
7814            }
7815            remove_module_cgroup(placement, &cgroup_name);
7816        }
7817        drop(roster_guard);
7818        return Err(SuperviseError::Spawn {
7819            program: spec.program.clone(),
7820            source: io::Error::other(
7821                "the daemon began shutting down while this process was starting; ended it",
7822            ),
7823            cgroup_path,
7824        });
7825    }
7826
7827    let stdout_pump = match child.stdout.take() {
7828        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7829        None => {
7830            warn!(
7831                module_id = %spec.module_id,
7832                "spawned child exposed no stdout pipe; file capture will be incomplete"
7833            );
7834            None
7835        }
7836    };
7837    let stderr_pump = match child.stderr.take() {
7838        Some(stderr) => {
7839            let generation = ring
7840                .lock()
7841                .unwrap_or_else(|poisoned| poisoned.into_inner())
7842                .begin_process();
7843            Some(StderrPump {
7844                task: tokio::spawn(pump_stderr_to(
7845                    stderr,
7846                    Arc::clone(ring),
7847                    generation,
7848                    output_sink,
7849                )),
7850                generation,
7851            })
7852        }
7853        None => {
7854            // Spawning succeeded but the pipe did not materialise. Recording it as
7855            // uncaptured keeps the tail honest: the alternative is an empty tail
7856            // that reads as a module which printed nothing.
7857            ring.lock()
7858                .unwrap_or_else(|poisoned| poisoned.into_inner())
7859                .mark_not_captured("stderr pipe was not available on spawn");
7860            warn!(
7861                module_id = %spec.module_id,
7862                "spawned child exposed no stderr pipe; tail will be unavailable"
7863            );
7864            None
7865        }
7866    };
7867
7868    Ok(SupervisedChild {
7869        child,
7870        protocol: spec.protocol,
7871        #[cfg(target_os = "linux")]
7872        module_id: cgroup_name,
7873        #[cfg(target_os = "linux")]
7874        cgroup_placement: cgroup_placement.cloned(),
7875        #[cfg(windows)]
7876        job,
7877        stdout_pump,
7878        stderr_pump,
7879        stderr_ring: Arc::clone(ring),
7880        spawned_at_ms,
7881        spawned_from,
7882        spawned_file_identity,
7883        process_start_time,
7884        process_identity,
7885        pid,
7886        roster_guard: Some(roster_guard),
7887        #[cfg(target_os = "macos")]
7888        privacy_exec,
7889        #[cfg(target_os = "macos")]
7890        report_ready: Arc::new(OnceLock::new()),
7891        spawn_failure: None,
7892    })
7893}
7894
7895#[cfg(target_os = "linux")]
7896pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7897    use subc_cgroup::KillOutcome;
7898    match subc_cgroup::kill_module(placement, module_id) {
7899        KillOutcome::Killed => {}
7900        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7901            debug!(
7902                module_id,
7903                "cgroup tree kill unavailable; using direct-child kill"
7904            );
7905        }
7906        KillOutcome::IoError { path, error } => {
7907            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7908        }
7909    }
7910}
7911
7912/// Contain a freshly spawned Windows child and start it.
7913///
7914/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7915/// child assigned **while it is still suspended** (step 1 is
7916/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7917///
7918/// A child that is never resumed hangs forever holding a pid, so a resume
7919/// failure kills the child and fails the spawn rather than returning a
7920/// `SupervisedChild` that can never run.
7921///
7922/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7923/// it did before this existed, whereas refusing to start one would be a new
7924/// outage. It is logged at warn because it means a helper process could leak.
7925#[cfg(windows)]
7926fn contain_spawned_child(
7927    child: &Child,
7928    spec: &ModuleSpec,
7929) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7930    let module_id = spec.module_id.as_str();
7931    let Some(pid) = child.id() else {
7932        // The child exited between spawn and here. Its tree, if it made one,
7933        // needs no containment: nothing is left to contain.
7934        warn!(
7935            module_id,
7936            "spawned child had already exited before containment; no job object attached"
7937        );
7938        return Ok(None);
7939    };
7940
7941    let job = match subc_jobobject::JobObject::new() {
7942        Ok(job) => job,
7943        Err(source) => {
7944            warn!(
7945                module_id,
7946                error = %source,
7947                "could not create a job object; this module's helper processes will not be \
7948                 reaped on teardown"
7949            );
7950            // Resume regardless: leaving the child suspended would turn a
7951            // containment gap into a hung module.
7952            resume_suspended_child(pid, spec)?;
7953            return Ok(None);
7954        }
7955    };
7956
7957    if let Err(source) = job.assign(child) {
7958        warn!(
7959            module_id,
7960            error = %source,
7961            "could not assign the child to its job object; this module's helper processes \
7962             will not be reaped on teardown"
7963        );
7964        resume_suspended_child(pid, spec)?;
7965        return Ok(None);
7966    }
7967
7968    resume_suspended_child(pid, spec)?;
7969    Ok(Some(job))
7970}
7971
7972/// Resume a suspended child, killing it if it cannot be started.
7973///
7974/// A suspended process holds a pid and does nothing, so there is no useful
7975/// state to return: the caller gets an error and the spawn fails.
7976#[cfg(windows)]
7977fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7978    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7979        // Kill it here rather than leaving a suspended process for the caller
7980        // to notice; `kill_on_drop` would eventually do this, but the module
7981        // would have been reported as running in between.
7982        let _ = std::process::Command::new("taskkill.exe")
7983            .args(["/PID", &pid.to_string(), "/T", "/F"])
7984            .stdin(Stdio::null())
7985            .stdout(Stdio::null())
7986            .stderr(Stdio::null())
7987            .status();
7988        return Err(SuperviseError::Spawn {
7989            program: spec.program.clone(),
7990            source,
7991            cgroup_path: None,
7992        });
7993    }
7994    Ok(())
7995}
7996
7997#[cfg(target_os = "linux")]
7998fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7999    match placement.remove_module(module_id) {
8000        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
8001        Err(error) => warn!(
8002            module_id,
8003            error = %error,
8004            "could not remove module cgroup after process exit; continuing teardown"
8005        ),
8006    }
8007}
8008
8009#[cfg(target_os = "linux")]
8010async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
8011    // Reaping the direct child is not proof its descendants exited. End the
8012    // residual tree and wait for the kernel's population fact before rmdir;
8013    // otherwise a successful parent wait leaks a directory on each restart.
8014    if matches!(
8015        subc_cgroup::kill_module(Some(placement), module_id),
8016        subc_cgroup::KillOutcome::Killed
8017    ) {
8018        if let Ok(path) = placement.module_path(module_id) {
8019            while std::fs::read_to_string(path.join("cgroup.events"))
8020                .ok()
8021                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
8022            {
8023                sleep(Duration::from_millis(1)).await;
8024            }
8025        }
8026    }
8027    remove_module_cgroup(placement, module_id);
8028}
8029
8030#[cfg(target_os = "linux")]
8031fn apply_cgroup_placement(
8032    command: &mut Command,
8033    spec: &ModuleSpec,
8034    path: &std::path::Path,
8035) -> Result<(), SuperviseError> {
8036    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
8037        module_id: spec.module_id.clone(),
8038        source,
8039    })
8040}
8041
8042fn capture_retention(spec: &ModuleSpec) -> Retention {
8043    let defaults = Retention::default();
8044    let value = |name: &str| {
8045        spec.env
8046            .iter()
8047            .rev()
8048            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
8049    };
8050    Retention {
8051        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
8052            .and_then(|value| value.parse().ok())
8053            .unwrap_or(defaults.max_file_mb),
8054        keep: value(CAPTURE_KEEP_ENV)
8055            .and_then(|value| value.parse().ok())
8056            .unwrap_or(defaults.keep),
8057        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
8058            .and_then(|value| value.parse().ok())
8059            .unwrap_or(defaults.max_age_days),
8060    }
8061}
8062
8063/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
8064/// module's registration to the exact process the supervisor spawned.
8065fn generate_launch_nonce() -> Result<String, SuperviseError> {
8066    let mut bytes = [0u8; 32];
8067    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
8068        reason: source.to_string(),
8069    })?;
8070    let mut hex = String::with_capacity(64);
8071    for b in bytes {
8072        use std::fmt::Write;
8073        let _ = write!(hex, "{b:02x}");
8074    }
8075    Ok(hex)
8076}
8077
8078/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
8079/// signal about how many leading bytes matched.
8080fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
8081    if a.len() != b.len() {
8082        return false;
8083    }
8084    let mut diff = 0u8;
8085    for (x, y) in a.iter().zip(b.iter()) {
8086        diff |= x ^ y;
8087    }
8088    diff == 0
8089}
8090
8091/// The kernel's image after an acknowledged exec, shared by ordinary launches
8092/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
8093/// not identities inferred from a configured pathname.
8094fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
8095    subc_os::Process::open(pid)
8096        .ok()
8097        .flatten()
8098        .and_then(|process| process.observe())
8099}
8100
8101fn spawn_and_mark_running(
8102    spec: &ModuleSpec,
8103    runtime: &SupervisorRuntimeConfig,
8104    snapshot: &SharedSnapshot,
8105) -> Result<SupervisedChild, SuperviseError> {
8106    let child = spawn_child(
8107        spec,
8108        runtime.connection_file_path.as_deref(),
8109        runtime.supervisor_handle.as_ref(),
8110        &runtime.stderr_ring,
8111        runtime.capture_logs_dir.as_deref(),
8112        &runtime.child_roster,
8113        #[cfg(target_os = "linux")]
8114        runtime.cgroup_placement.as_ref(),
8115    )?;
8116    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
8117    Ok(child)
8118}
8119
8120enum RegistrationWaitOutcome {
8121    Registered,
8122    Exited(ExitReport),
8123    TimedOut,
8124}
8125
8126struct ReloadRegistrationFailure {
8127    exit_report: ExitReport,
8128    reason: String,
8129}
8130
8131#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8132enum BusyGaugeObservation {
8133    Quiescent,
8134    Busy,
8135    Omitted,
8136}
8137
8138fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8139    let Some(metrics) = metrics.and_then(Value::as_object) else {
8140        return BusyGaugeObservation::Omitted;
8141    };
8142    let mut sum = 0u128;
8143    for gauge in gauges {
8144        let Some(value) = metrics.get(gauge) else {
8145            return BusyGaugeObservation::Omitted;
8146        };
8147        let Some(value) = value.as_u64() else {
8148            return BusyGaugeObservation::Busy;
8149        };
8150        sum = sum.saturating_add(u128::from(value));
8151    }
8152    if sum == 0 {
8153        BusyGaugeObservation::Quiescent
8154    } else {
8155        BusyGaugeObservation::Busy
8156    }
8157}
8158
8159fn declared_busy_gauges(
8160    registry: &Registry,
8161    module_id: &str,
8162) -> Result<Vec<String>, SuperviseError> {
8163    busy_gauges_of(
8164        registry
8165            .get_module(module_id)
8166            .map_err(SuperviseError::Registry)?,
8167    )
8168}
8169
8170/// [`declared_busy_gauges`] for the registration a connection holds, in any
8171/// slot: after cutover the incumbent is no longer the id's active
8172/// registration, and its own manifest is the one that names its gauges.
8173fn declared_busy_gauges_for_connection(
8174    registry: &Registry,
8175    connection_id: ConnectionId,
8176) -> Result<Vec<String>, SuperviseError> {
8177    busy_gauges_of(
8178        registry
8179            .get_module_by_connection(connection_id)
8180            .map_err(SuperviseError::Registry)?,
8181    )
8182}
8183
8184fn busy_gauges_of(
8185    registration: Option<crate::registry::ModuleRegistration>,
8186) -> Result<Vec<String>, SuperviseError> {
8187    let Some(registration) = registration else {
8188        return Ok(Vec::new());
8189    };
8190    let Some(self_signals) = registration.manifest.self_signals else {
8191        return Ok(Vec::new());
8192    };
8193
8194    let mut gauges = Vec::new();
8195    for declaration in self_signals {
8196        if declaration.kind != SelfSignalKind::Busy {
8197            continue;
8198        }
8199        match declaration.anchored_to {
8200            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8201                gauges.extend(declared)
8202            }
8203            _ => {
8204                // An invalid Busy anchor is fail-safe: the empty name cannot be
8205                // present in a conforming health report, so this drain stays busy.
8206                gauges.push(String::new());
8207            }
8208        }
8209    }
8210    Ok(gauges)
8211}
8212
8213/// Wait for `endpoint` to have nothing in flight and, when the module declares
8214/// busy gauges, for a health probe to report them quiet. The probe is addressed
8215/// by `scope`: a swap's superseded incumbent must be asked about its own
8216/// gauges, and by module id the probe would reach the promoted candidate.
8217async fn wait_for_forwarding_quiescence(
8218    forwarding: &ForwardingTable,
8219    module_id: &str,
8220    runtime: &SupervisorRuntimeConfig,
8221    endpoint: crate::ModuleEndpointId,
8222    deadline: Instant,
8223    busy_gauges: &[String],
8224    scope: DrainScope,
8225) -> Result<bool, SuperviseError> {
8226    let mut gauges_quiescent = busy_gauges.is_empty();
8227    let mut next_probe_at = Instant::now();
8228    let mut omission_counted = false;
8229
8230    loop {
8231        let now = Instant::now();
8232        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8233            let report = match scope {
8234                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8235                DrainScope::Endpoint(endpoint) => {
8236                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8237                }
8238            };
8239            gauges_quiescent = match report {
8240                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8241                    BusyGaugeObservation::Quiescent => true,
8242                    BusyGaugeObservation::Busy => false,
8243                    BusyGaugeObservation::Omitted => {
8244                        if !omission_counted {
8245                            forwarding
8246                                .counters()
8247                                .increment_drains_with_undeclared_gauge();
8248                            omission_counted = true;
8249                        }
8250                        false
8251                    }
8252                },
8253                Err(err) => {
8254                    warn!(
8255                        module_id,
8256                        error = %err,
8257                        "drain health.check did not produce declared busy gauges; treating module as busy"
8258                    );
8259                    false
8260                }
8261            };
8262            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8263        }
8264
8265        let in_flight = forwarding
8266            .endpoint_in_flight_count(endpoint)
8267            .map_err(SuperviseError::Forwarding)?;
8268        if in_flight == 0 && gauges_quiescent {
8269            return Ok(true);
8270        }
8271
8272        let now = Instant::now();
8273        if now >= deadline {
8274            return Ok(false);
8275        }
8276        let mut wait = deadline
8277            .saturating_duration_since(now)
8278            .min(REGISTRY_RELEASE_POLL);
8279        if !busy_gauges.is_empty() {
8280            wait = wait.min(next_probe_at.saturating_duration_since(now));
8281        }
8282        sleep(wait).await;
8283    }
8284}
8285
8286/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8287///
8288/// `Ok` is always honest and passed straight through -- the wait actually measured
8289/// in-flight state. `Err` means the wait produced no measurement at all (the
8290/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8291/// constant: the drain did not complete. Never recomputed from route state, never a
8292/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8293fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8294    match wait_result {
8295        Ok(drained) => *drained,
8296        Err(_) => false,
8297    }
8298}
8299
8300fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8301    for released in released_routes {
8302        let frame = match Frame::build_with_version(
8303            released.negotiated_ver,
8304            FrameType::Goodbye,
8305            control_flags(),
8306            released.channel,
8307            released.epoch,
8308            0,
8309            Vec::new(),
8310        ) {
8311            Ok(frame) => frame,
8312            Err(err) => {
8313                warn!(
8314                    route_channel = released.channel,
8315                    error = %err,
8316                    "failed to build supervisor drain route GOODBYE frame"
8317                );
8318                continue;
8319            }
8320        };
8321        if !released.close_on_delivery_failure() {
8322            crate::forwarding::send_module_route_goodbye(
8323                &forwarding.counters(),
8324                &released.sink,
8325                frame,
8326                released.module_id.as_deref(),
8327                "supervisor drain",
8328            );
8329            continue;
8330        }
8331        if let Err(err) = released.sink.try_send(frame) {
8332            warn!(
8333                target_connection_id = released.connection_id.get(),
8334                route_channel = released.channel,
8335                error = %err,
8336                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8337            );
8338            let _ = forwarding.escalate_client_delivery_failure(
8339                released.connection_id,
8340                released.channel,
8341                released.epoch,
8342                CloseReason::new(
8343                    "route_goodbye_delivery_failed",
8344                    format!(
8345                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8346                        released.channel
8347                    ),
8348                ),
8349                crate::forwarding::UndeliveredFrame {
8350                    module_id: released.module_id.as_deref(),
8351                    sink: &released.sink,
8352                },
8353            );
8354        }
8355    }
8356}
8357
8358fn send_module_draining(
8359    module_id: &str,
8360    reason: RouteCloseReason,
8361    deadline_ms: u64,
8362    target: &ModuleDrainTarget,
8363) {
8364    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8365        reason,
8366        deadline_ms,
8367    }) {
8368        Ok(body) => body,
8369        Err(err) => {
8370            warn!(
8371                module_id,
8372                error = %err,
8373                "failed to encode module draining command"
8374            );
8375            return;
8376        }
8377    };
8378    let frame = match Frame::build_with_version(
8379        target.negotiated_ver,
8380        FrameType::Push,
8381        control_flags(),
8382        0,
8383        0,
8384        0,
8385        body,
8386    ) {
8387        Ok(frame) => frame,
8388        Err(err) => {
8389            warn!(
8390                module_id,
8391                error = %err,
8392                "failed to build module draining command frame"
8393            );
8394            return;
8395        }
8396    };
8397    if let Err(err) = target.sink.try_send(frame) {
8398        warn!(
8399            module_id,
8400            target_connection_id = target.endpoint.connection_id.get(),
8401            error = %err,
8402            "module draining command was not delivered to peer"
8403        );
8404    }
8405}
8406
8407/// The channel-0 GOODBYE that tells a module its stop is planned.
8408fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8409    match Frame::build_with_version(
8410        negotiated_ver,
8411        FrameType::Goodbye,
8412        control_flags(),
8413        0,
8414        0,
8415        0,
8416        Vec::new(),
8417    ) {
8418        Ok(frame) => Some(frame),
8419        Err(err) => {
8420            warn!(
8421                module_id,
8422                error = %err,
8423                "failed to build module GOODBYE frame"
8424            );
8425            None
8426        }
8427    }
8428}
8429
8430/// Send every registered module connection its module GOODBYE at daemon
8431/// shutdown, then request that connection's close.
8432///
8433/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8434/// before EOF, so the GOODBYE must reach the socket before the close. A close
8435/// request does not wait for the connection's queued frames: its writer gets a
8436/// bounded grace after the close, is aborted if it overruns it, and the daemon
8437/// process may exit before that grace ends. So with `wait_for_flush`, each
8438/// connection is closed only after its writer has acknowledged writing the
8439/// GOODBYE, or once a short shared budget runs out, so one module that is not
8440/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8441/// are only queued, for a shutdown the operator has told to stop waiting.
8442/// A connection that is already gone is skipped.
8443#[cfg(unix)]
8444async fn send_module_goodbyes_for_daemon_shutdown(
8445    forwarding: &Arc<ForwardingTable>,
8446    reason: &CloseReason,
8447    wait_for_flush: bool,
8448) {
8449    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8450    let targets = match forwarding.module_connections() {
8451        Ok(targets) => targets,
8452        Err(err) => {
8453            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8454            return;
8455        }
8456    };
8457    let deadline = Instant::now() + GOODBYE_BUDGET;
8458    let mut sends = tokio::task::JoinSet::new();
8459    for target in targets {
8460        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8461            continue;
8462        };
8463        if !wait_for_flush {
8464            if let Err(err) = target.sink.try_send(frame) {
8465                debug!(
8466                    module_id = %target.module_id,
8467                    error = %err,
8468                    "shutdown module GOODBYE was not queued"
8469                );
8470            }
8471            continue;
8472        }
8473        let forwarding = Arc::clone(forwarding);
8474        let reason = reason.clone();
8475        sends.spawn(async move {
8476            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8477                Ok(Ok(())) => {}
8478                Ok(Err(err)) => debug!(
8479                    module_id = %target.module_id,
8480                    error = %err,
8481                    "module connection closed before its shutdown GOODBYE was written"
8482                ),
8483                Err(_) => warn!(
8484                    module_id = %target.module_id,
8485                    budget = ?GOODBYE_BUDGET,
8486                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8487                ),
8488            }
8489            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8490        });
8491    }
8492    // Every task ends by the shared deadline, so this wait is bounded too.
8493    while sends.join_next().await.is_some() {}
8494}
8495
8496fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8497    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8498        return;
8499    };
8500    if let Err(err) = target.sink.try_send(frame) {
8501        warn!(
8502            module_id,
8503            target_connection_id = target.endpoint.connection_id.get(),
8504            error = %err,
8505            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8506        );
8507        forwarding.request_connection_close(
8508            target.endpoint.connection_id,
8509            CloseReason::new(
8510                "module_goodbye_delivery_failed",
8511                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8512            ),
8513        );
8514    }
8515}
8516
8517#[derive(Clone, Copy)]
8518struct ForwardingDrainContext<'a> {
8519    spec: &'a ModuleSpec,
8520    runtime: &'a SupervisorRuntimeConfig,
8521    registry: &'a Registry,
8522    scope: DrainScope,
8523}
8524
8525/// Which process a forwarding drain addresses.
8526#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8527enum DrainScope {
8528    /// Whatever endpoint is active for the module id: every plain stop,
8529    /// restart and reload. Also moves the module's state to `Draining`.
8530    Active,
8531    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8532    /// module id would resolve to the promoted candidate and leave neither
8533    /// process routable. The module's state is left alone, since the promoted
8534    /// candidate is what it describes and that process is running.
8535    Endpoint(crate::ModuleEndpointId),
8536}
8537
8538/// Whether a child being drained has already been asked to stop by the time
8539/// its drain wait starts.
8540///
8541/// The drain wait is the same budget whatever this says. What it decides is
8542/// whether the supervisor must ask by signal before that wait begins: a child
8543/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8544/// healthy or not.
8545#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8546enum StopNotice {
8547    /// The module was sent `module.draining` and a module GOODBYE over its own
8548    /// registered connection, and stops itself.
8549    SentOverConnection,
8550    /// The forwarding drain found no registered connection for the module: a
8551    /// subc child spawned moments ago that has not sent HELLO yet, or a
8552    /// `protocol: "none"` child, which never registers.
8553    NoConnection,
8554    /// This path sends nothing over the module's connection: the supervisor has
8555    /// no forwarding table, or the caller stops the child without a forwarding
8556    /// drain.
8557    NotSent,
8558}
8559
8560async fn begin_forwarding_drain(
8561    spec: &ModuleSpec,
8562    runtime: &SupervisorRuntimeConfig,
8563    registry: &Registry,
8564    snapshot: &SharedSnapshot,
8565    enabled: Option<bool>,
8566    reason: RouteCloseReason,
8567) -> Result<StopNotice, SuperviseError> {
8568    let Some(forwarding) = runtime.forwarding.as_ref() else {
8569        return Err(SuperviseError::ReloadUnavailable {
8570            module_id: spec.module_id.clone(),
8571            reason: "supervisor was not configured with a forwarding table".to_string(),
8572        });
8573    };
8574
8575    begin_forwarding_drain_with(
8576        forwarding,
8577        ForwardingDrainContext {
8578            spec,
8579            runtime,
8580            registry,
8581            scope: DrainScope::Active,
8582        },
8583        snapshot,
8584        enabled,
8585        reason,
8586        runtime.drain_timeout,
8587    )
8588    .await
8589}
8590
8591async fn begin_forwarding_drain_if_configured(
8592    spec: &ModuleSpec,
8593    runtime: &SupervisorRuntimeConfig,
8594    registry: &Registry,
8595    snapshot: &SharedSnapshot,
8596    enabled: Option<bool>,
8597    reason: RouteCloseReason,
8598) -> Result<StopNotice, SuperviseError> {
8599    begin_forwarding_drain_with_timeout(
8600        spec,
8601        runtime,
8602        registry,
8603        snapshot,
8604        enabled,
8605        reason,
8606        runtime.drain_timeout,
8607    )
8608    .await
8609}
8610
8611/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8612/// budget, for paths where the operator overrides the module's configured one
8613/// (`supervisor.restart{drain_timeout_ms}`).
8614async fn begin_forwarding_drain_with_timeout(
8615    spec: &ModuleSpec,
8616    runtime: &SupervisorRuntimeConfig,
8617    registry: &Registry,
8618    snapshot: &SharedSnapshot,
8619    enabled: Option<bool>,
8620    reason: RouteCloseReason,
8621    drain_timeout: Duration,
8622) -> Result<StopNotice, SuperviseError> {
8623    let Some(forwarding) = runtime.forwarding.as_ref() else {
8624        return Ok(StopNotice::NotSent);
8625    };
8626
8627    begin_forwarding_drain_with(
8628        forwarding,
8629        ForwardingDrainContext {
8630            spec,
8631            runtime,
8632            registry,
8633            scope: DrainScope::Active,
8634        },
8635        snapshot,
8636        enabled,
8637        reason,
8638        drain_timeout,
8639    )
8640    .await
8641}
8642
8643async fn begin_forwarding_drain_with(
8644    forwarding: &ForwardingTable,
8645    context: ForwardingDrainContext<'_>,
8646    snapshot: &SharedSnapshot,
8647    enabled: Option<bool>,
8648    reason: RouteCloseReason,
8649    drain_timeout: Duration,
8650) -> Result<StopNotice, SuperviseError> {
8651    let ForwardingDrainContext {
8652        spec,
8653        runtime,
8654        registry,
8655        scope,
8656    } = context;
8657    debug_assert_ne!(reason, RouteCloseReason::Crash);
8658    let terminal = matches!(reason, RouteCloseReason::Disable);
8659    let drain_started_at = Instant::now();
8660    let drain_deadline = drain_started_at + drain_timeout;
8661    let deadline_ms =
8662        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8663    let busy_gauges = match scope {
8664        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8665        DrainScope::Endpoint(endpoint) => {
8666            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8667        }
8668    };
8669
8670    // Admission gate first: route.open/commit and route REQUEST admission are closed
8671    // before the first quiescence check, so the outstanding count can only fall.
8672    let gate_started = Instant::now();
8673    let drain_target = match scope {
8674        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8675        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8676    }
8677    .map_err(SuperviseError::Forwarding)?;
8678    // The instant admission closed, and how long taking the forwarding write
8679    // lock to close it took. The timeout line reports only the quiescence
8680    // wait, so without this a drain that started late looked like one that
8681    // started on time.
8682    info!(
8683        module_id = %spec.module_id,
8684        ?reason,
8685        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8686        connected = drain_target.is_some(),
8687        "module drain began; route admission closed"
8688    );
8689    if scope == DrainScope::Active {
8690        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8691            state.state = ModuleState::Draining;
8692            state.draining_to_replace =
8693                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8694            if let Some(enabled) = enabled {
8695                state.enabled = enabled;
8696            }
8697        })?;
8698    }
8699
8700    let Some(target) = drain_target.as_ref() else {
8701        // Nothing was sent: the module has no registered connection to carry
8702        // `module.draining` or a GOODBYE. The caller must not assume the child
8703        // was asked to stop.
8704        return Ok(StopNotice::NoConnection);
8705    };
8706    {
8707        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8708        let routes = forwarding
8709            .endpoint_routes(target.endpoint)
8710            .map_err(SuperviseError::Forwarding)?;
8711        let routes_notified = routes.len();
8712        crate::control::send_route_control_pushes(
8713            forwarding,
8714            routes.clone(),
8715            ClientControlPush::RouteClosing {
8716                module_id: spec.module_id.clone(),
8717                channels: Vec::new(),
8718                reason,
8719            },
8720        );
8721        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8722
8723        // `route.closing` was just sent above: from here on every return path,
8724        // including an early one, MUST send `route.closed` before propagating
8725        // anything else. A client holds `closing` as a promise that a verdict is
8726        // coming; leaving early without `closed` strands it waiting forever, since
8727        // `closing` carries no timeout of its own.
8728        let wait_result = wait_for_forwarding_quiescence(
8729            forwarding,
8730            &spec.module_id,
8731            runtime,
8732            target.endpoint,
8733            drain_deadline,
8734            &busy_gauges,
8735            scope,
8736        )
8737        .await;
8738        let drained = drained_after_quiescence_wait(&wait_result);
8739        if let Err(err) = &wait_result {
8740            error!(
8741                module_id = %spec.module_id,
8742                ?reason,
8743                error = %err,
8744                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8745            );
8746        } else if !drained {
8747            // Name what the drain waited on. Without it the line says only that
8748            // something did not settle, and "one wedged call" and "every
8749            // session's held stream" read the same; the first is a module bug,
8750            // the second is a module that should end its streams on
8751            // module.draining. Read before teardown releases the routes.
8752            let holdouts = forwarding
8753                .endpoint_drain_holdouts(target.endpoint)
8754                .unwrap_or_default();
8755            warn!(
8756                module_id = %spec.module_id,
8757                waited = ?drain_timeout,
8758                ?reason,
8759                held_requests = holdouts.requests,
8760                held_routes = holdouts.routes,
8761                total_routes = holdouts.total_routes,
8762                top_connections = ?holdouts.top_connections,
8763                // `module_channel:corr`, so the module can find each held request
8764                // in its own log; capped, so `held_requests` is the full count.
8765                held = %holdouts
8766                    .held
8767                    .iter()
8768                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8769                    .collect::<Vec<_>>()
8770                    .join(","),
8771                "route drain timed out before request quiescence; forcing teardown"
8772            );
8773        }
8774        crate::control::send_route_control_pushes(
8775            forwarding,
8776            routes,
8777            ClientControlPush::RouteClosed {
8778                module_id: spec.module_id.clone(),
8779                channels: Vec::new(),
8780                reason,
8781                drained,
8782                abandoned: target.abandoned_bindings.len() as u32,
8783                excluded_subscriptions: target.excluded_subscriptions,
8784                terminal: Some(terminal),
8785            },
8786        );
8787        wait_result?;
8788
8789        // `route.closed` has now been sent unconditionally above. From here the
8790        // remaining steps are cleanup (route + module GOODBYE) rather than a
8791        // promise the client is waiting on, but a lock-poisoned
8792        // `release_module_endpoint_routes` would otherwise skip the module
8793        // GOODBYE silently too -- send it before propagating the error.
8794        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8795            Ok(routes) => routes,
8796            Err(err) => {
8797                warn!(
8798                    module_id = %spec.module_id,
8799                    ?reason,
8800                    error = %err,
8801                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8802                );
8803                send_module_goodbye(&spec.module_id, forwarding, target);
8804                return Err(SuperviseError::Forwarding(err));
8805            }
8806        };
8807        let route_goodbye_count = released_routes.len();
8808        send_route_goodbyes(forwarding, released_routes);
8809        send_module_goodbye(&spec.module_id, forwarding, target);
8810
8811        // The drain's happy path was previously silent: every emission above is
8812        // best-effort with only its failure arm logged, so "were consumers told"
8813        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8814        // hang where the open question was exactly whether teardown notice went
8815        // out). One summary line makes that class decidable in one grep.
8816        info!(
8817            module_id = %spec.module_id,
8818            ?reason,
8819            routes_notified,
8820            route_goodbyes = route_goodbye_count,
8821            abandoned_reservations = target.abandoned_bindings.len(),
8822            excluded_subscriptions = target.excluded_subscriptions,
8823            drained,
8824            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8825        );
8826    }
8827
8828    Ok(StopNotice::SentOverConnection)
8829}
8830
8831/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8832/// the only slot a plain (non-swap) spawn can register into.
8833async fn wait_for_registration_after_reload(
8834    registry: &Registry,
8835    module_id: &str,
8836    snapshot: &SharedSnapshot,
8837    child: &mut SupervisedChild,
8838    wait: Duration,
8839) -> Result<RegistrationWaitOutcome, SuperviseError> {
8840    wait_for_slot_registration(
8841        registry,
8842        crate::registry::RegistrationSlot::Active(module_id),
8843        module_id,
8844        snapshot,
8845        child,
8846        wait,
8847    )
8848    .await
8849}
8850
8851/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8852///
8853/// Keyed on the slot rather than the bare module id because during a swap the
8854/// id's active slot is already held by the incumbent: an id-keyed wait would
8855/// report the incumbent's registration as the candidate's and a candidate that
8856/// never registers would look registered. A swap candidate waits on
8857/// `crate::registry::RegistrationSlot::Candidate`.
8858async fn wait_for_slot_registration(
8859    registry: &Registry,
8860    slot: crate::registry::RegistrationSlot<'_>,
8861    module_id: &str,
8862    snapshot: &SharedSnapshot,
8863    child: &mut SupervisedChild,
8864    wait: Duration,
8865) -> Result<RegistrationWaitOutcome, SuperviseError> {
8866    let deadline = Instant::now() + wait;
8867    loop {
8868        if registry
8869            .registration(slot)
8870            .map_err(SuperviseError::Registry)?
8871            .is_some()
8872        {
8873            return Ok(RegistrationWaitOutcome::Registered);
8874        }
8875
8876        let now = Instant::now();
8877        if now >= deadline {
8878            return Ok(RegistrationWaitOutcome::TimedOut);
8879        }
8880        let remaining = deadline.saturating_duration_since(now);
8881        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8882
8883        tokio::select! {
8884            wait_result = child.wait() => {
8885                let status = wait_result.map_err(|source| SuperviseError::Wait {
8886                    module_id: module_id.to_string(),
8887                    source,
8888                })?;
8889                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8890                    snapshot,
8891                    child,
8892                    &status,
8893                )));
8894            }
8895            _ = sleep(poll) => {}
8896        }
8897    }
8898}
8899
8900fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8901    // A replacement process that exits before HELLO did not provide service, even
8902    // if it used status 0. Count it against the restart cap as a new-binary failure.
8903    if exit_report.kind != ExitKind::DeliberateSeverance {
8904        exit_report.kind = ExitKind::Crash;
8905    }
8906    exit_report
8907}
8908
8909async fn handle_reload_child_registration_failure(
8910    spec: &ModuleSpec,
8911    runtime: &SupervisorRuntimeConfig,
8912    registry: &Registry,
8913    process_liveness: &SupervisorProcessLiveness,
8914    snapshot: &SharedSnapshot,
8915    _child: &mut Option<SupervisedChild>,
8916    failure: ReloadRegistrationFailure,
8917) -> Result<(), SuperviseError> {
8918    let ReloadRegistrationFailure {
8919        exit_report,
8920        reason,
8921    } = failure;
8922    match on_child_exit(
8923        spec,
8924        runtime.restart_policy,
8925        registry,
8926        snapshot,
8927        &runtime.terminal_ring,
8928        &runtime.spawn_events,
8929        &runtime.child_roster,
8930        exit_report,
8931    )
8932    .await
8933    {
8934        NextAction::Stop {
8935            registration_released,
8936        } => {
8937            if registration_released {
8938                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8939            }
8940        }
8941        NextAction::Restart { schedule } => {
8942            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8943                schedule.delay
8944            });
8945            if let Some(schedule) = schedule {
8946                log_crash_respawn(&spec.module_id, schedule);
8947            }
8948            schedule_respawn(
8949                runtime,
8950                snapshot,
8951                &spec.module_id,
8952                delay,
8953                RespawnKind::Spawn,
8954            )?;
8955        }
8956    }
8957    Err(SuperviseError::ReloadFailed {
8958        module_id: spec.module_id.clone(),
8959        reason,
8960    })
8961}
8962
8963async fn handle_reload_spawn_failure(
8964    spec: &ModuleSpec,
8965    runtime: &SupervisorRuntimeConfig,
8966    process_liveness: &SupervisorProcessLiveness,
8967    snapshot: &SharedSnapshot,
8968    _child: &mut Option<SupervisedChild>,
8969    reason: String,
8970) -> Result<(), SuperviseError> {
8971    let now = Instant::now();
8972    let mut schedule = None;
8973    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8974        clear_current_process_facts(state);
8975        if state.enabled {
8976            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8977            state.state = if schedule.is_some() {
8978                ModuleState::Restarting
8979            } else {
8980                ModuleState::Failed
8981            };
8982        } else {
8983            state.state = ModuleState::Disabled;
8984        }
8985    })?;
8986    if let Some(schedule) = schedule {
8987        schedule_respawn(
8988            runtime,
8989            snapshot,
8990            &spec.module_id,
8991            schedule.delay,
8992            RespawnKind::Spawn,
8993        )?;
8994    } else {
8995        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8996    }
8997    Err(SuperviseError::ReloadFailed {
8998        module_id: spec.module_id.clone(),
8999        reason,
9000    })
9001}
9002
9003fn control_flags() -> Flags {
9004    Flags::new(false, Priority::Passive, false)
9005}
9006
9007#[allow(clippy::too_many_arguments)]
9008async fn drain_optional_child(
9009    module_id: &str,
9010    protocol: ModuleProtocol,
9011    stop_notice: StopNotice,
9012    registry: &Registry,
9013    forwarding: Option<&ForwardingTable>,
9014    snapshot: &SharedSnapshot,
9015    terminal_ring: &Arc<Mutex<TerminalRing>>,
9016    spawn_events: &SpawnEventFeed,
9017    child: &mut Option<SupervisedChild>,
9018    drain_timeout: Duration,
9019    final_state: ModuleState,
9020    enabled: Option<bool>,
9021) -> Result<(), SuperviseError> {
9022    if let Some(child) = child.take() {
9023        drain_child_to_state(
9024            module_id,
9025            protocol,
9026            stop_notice,
9027            registry,
9028            forwarding,
9029            snapshot,
9030            terminal_ring,
9031            spawn_events,
9032            child,
9033            drain_timeout,
9034            final_state,
9035            enabled,
9036        )
9037        .await
9038    } else {
9039        update_snapshot(snapshot, Some(module_id), |state| {
9040            state.state = final_state;
9041            if let Some(enabled) = enabled {
9042                state.enabled = enabled;
9043            }
9044            clear_current_process_facts(state);
9045        })?;
9046        release_dead_registration(registry, forwarding, snapshot, module_id).await
9047    }
9048}
9049
9050#[allow(clippy::too_many_arguments)]
9051async fn drain_child_to_state(
9052    module_id: &str,
9053    _protocol: ModuleProtocol,
9054    stop_notice: StopNotice,
9055    registry: &Registry,
9056    forwarding: Option<&ForwardingTable>,
9057    snapshot: &SharedSnapshot,
9058    terminal_ring: &Arc<Mutex<TerminalRing>>,
9059    spawn_events: &SpawnEventFeed,
9060    mut child: SupervisedChild,
9061    drain_timeout: Duration,
9062    final_state: ModuleState,
9063    enabled: Option<bool>,
9064) -> Result<(), SuperviseError> {
9065    let protocol = child.protocol;
9066    update_snapshot(snapshot, Some(module_id), |state| {
9067        state.state = ModuleState::Draining;
9068        state.draining_to_replace = final_state == ModuleState::Restarting;
9069        if let Some(enabled) = enabled {
9070            state.enabled = enabled;
9071        }
9072    })?;
9073
9074    // The wait below is the same budget in every case; what differs is
9075    // whether anything has ASKED the child to stop before it starts. Only a
9076    // forwarding drain that reached the module's registered connection has
9077    // (`module.draining`, then a module GOODBYE). Every other child was told
9078    // nothing: a `protocol: "none"` module, which never registers; a subc
9079    // module spawned moments ago that has not sent HELLO yet; or a stop that
9080    // runs no forwarding drain. Without a signal the budget is only a delay
9081    // in front of SIGKILL -- and the not-yet-registered child is the worst
9082    // case, because it registers into a module that is already draining,
9083    // is never told, and is killed while healthy.
9084    if stop_notice != StopNotice::SentOverConnection {
9085        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
9086            info!(
9087                module_id,
9088                pid = child.pid,
9089                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9090                "module has no connection yet; requesting stop by signal"
9091            );
9092        }
9093        request_graceful_stop(module_id, &child);
9094    }
9095
9096    let exit_report = match timeout(drain_timeout, child.wait()).await {
9097        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
9098        Ok(Err(source)) => {
9099            fail_snapshot(snapshot, Some(module_id), None);
9100            return Err(SuperviseError::Wait {
9101                module_id: module_id.to_string(),
9102                source,
9103            });
9104        }
9105        Err(_) => {
9106            // Mirror the sibling arm above: state is already `Draining`, and an
9107            // error propagated from here would strand it there -- a state
9108            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
9109            // `Failed | Stopped`), leaving an operator Restart as the only exit.
9110            // `Failed` before `?` keeps the module operator-visible and
9111            // revivable. Trigger is an ESRCH race (process exits between the
9112            // drain timeout firing and the kill) or a post-kill wait failure
9113            // (issue #34).
9114            //
9115            // Logged because the kill is otherwise visible only as signal 9 in
9116            // the terminal ring, and the budget it follows can be long enough
9117            // that consumers see a stretch of refusals with no stated cause.
9118            warn!(
9119                module_id,
9120                pid = child.pid,
9121                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9122                reason = ?final_state,
9123                ?stop_notice,
9124                "drain budget expired before the module exited; killing it"
9125            );
9126            child.start_kill().map_err(|source| {
9127                fail_snapshot(snapshot, Some(module_id), None);
9128                SuperviseError::Kill {
9129                    module_id: module_id.to_string(),
9130                    source,
9131                }
9132            })?;
9133            let status = child.wait().await.map_err(|source| {
9134                fail_snapshot(snapshot, Some(module_id), None);
9135                SuperviseError::Wait {
9136                    module_id: module_id.to_string(),
9137                    source,
9138                }
9139            })?;
9140            classify_reaped_child_exit(snapshot, &child, &status)
9141        }
9142    };
9143
9144    update_snapshot(snapshot, Some(module_id), |state| {
9145        state.state = final_state;
9146        if let Some(enabled) = enabled {
9147            state.enabled = enabled;
9148        }
9149        clear_current_process_facts(state);
9150        state.last_exit = Some(exit_report.clone());
9151        if exit_report.kind == ExitKind::DeliberateSeverance {
9152            state.lifetime_restarts += 1;
9153        }
9154    })?;
9155    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9156    record_terminal_with_detail(
9157        module_id,
9158        terminal_ring,
9159        spawn_events,
9160        &exit_report,
9161        terminal_disposition(final_state),
9162        detail,
9163    );
9164    child.drain_stderr(module_id).await;
9165
9166    release_dead_registration(registry, forwarding, snapshot, module_id).await
9167}
9168
9169/// Ask a child that nothing else has asked to stop, by signal.
9170///
9171/// A registered subc module is asked over its own connection: the drain sends
9172/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
9173/// module GOODBYE, and the module stops itself. A module that speaks no subc
9174/// wire receives none of that, and neither does a subc module that has not
9175/// registered yet, so for them the drain budget would be pure delay in front of
9176/// a SIGKILL -- and for a process with a store to flush (JetStream is the
9177/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
9178/// into a recovery on the next start.
9179///
9180/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
9181/// rule rather than an optimisation: that module's graceful stop is already
9182/// running by the time its child is drained, and a signal would race it.
9183///
9184/// Best-effort by construction. A child that has already exited is the ordinary
9185/// case rather than an error (the kill lands on a reaped or exiting pid), so a
9186/// failure is logged at debug and the wait-then-kill below still decides the
9187/// outcome.
9188#[cfg(unix)]
9189fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9190    let Some(pid) = child
9191        .id()
9192        .and_then(|pid| i32::try_from(pid).ok())
9193        .and_then(rustix::process::Pid::from_raw)
9194    else {
9195        debug!(
9196            module_id,
9197            "no pid to signal for teardown; falling through to the drain wait"
9198        );
9199        return;
9200    };
9201    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9202        Ok(()) => debug!(
9203            module_id,
9204            "sent SIGTERM to a module nothing else asked to stop"
9205        ),
9206        Err(err) => debug!(
9207            module_id,
9208            error = %err,
9209            "SIGTERM to module failed; the drain wait and kill still apply"
9210        ),
9211    }
9212}
9213
9214/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9215/// Windows does offer need cooperation this supervisor cannot assume: a console
9216/// control event requires sharing a console with the child, and `WM_CLOSE`
9217/// requires the child to pump a message loop. A supervised server process does
9218/// neither, so there is nothing to send and teardown is the wait followed by the
9219/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9220/// the thing `protocol: "none"` exists to avoid.
9221#[cfg(not(unix))]
9222fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9223    debug!(
9224        module_id,
9225        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9226    );
9227}
9228
9229fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9230    match final_state {
9231        ModuleState::Stopped => TerminalDisposition::Stopped,
9232        ModuleState::Disabled => TerminalDisposition::Disabled,
9233        ModuleState::Restarting => TerminalDisposition::Restarting,
9234        ModuleState::Failed => TerminalDisposition::Failed,
9235        ModuleState::Starting
9236        | ModuleState::Running
9237        | ModuleState::Unresponsive
9238        | ModuleState::Draining => {
9239            unreachable!("terminal exits only finish in terminal or restarting states")
9240        }
9241    }
9242}
9243
9244/// Release a reaped child's registration before allowing another spawn.
9245///
9246/// EOF is not a process-lifetime signal: an inherited socket can stay open
9247/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9248/// reading EOF. After the normal release grace, request connection close (which
9249/// cancels both reads and dispatch), then allow one more release grace for the
9250/// connection guard's forwarding cleanup. Never evict a different connection.
9251async fn release_dead_registration(
9252    registry: &Registry,
9253    forwarding: Option<&ForwardingTable>,
9254    snapshot: &SharedSnapshot,
9255    module_id: &str,
9256) -> Result<(), SuperviseError> {
9257    let result = async {
9258        let registration = registry
9259            .get_module(module_id)
9260            .map_err(SuperviseError::Registry)?;
9261        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9262            Ok(()) => return Ok(()),
9263            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9264            Err(err) => return Err(err),
9265        }
9266        let pid = lock_snapshot(snapshot)?.reaped_pid;
9267        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9268            warn!(
9269                module_id,
9270                pid,
9271                connection_id = registration.connection_id.get(),
9272                "reaped module registration outlived release grace; closing dead connection"
9273            );
9274            forwarding.request_connection_close(
9275                registration.connection_id,
9276                CloseReason::new(
9277                    "supervised_process_reaped",
9278                    format!("module '{module_id}' pid {pid} exited"),
9279                ),
9280            );
9281            wait_for_slot_registration_release(
9282                registry,
9283                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9284                REGISTRY_RELEASE_TIMEOUT,
9285            )
9286            .await?;
9287        }
9288        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9289    }
9290    .await;
9291    if let Err(err) = &result {
9292        fail_snapshot(snapshot, Some(module_id), None);
9293        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9294    }
9295    result
9296}
9297
9298/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9299/// plain stop or restart waits for before it spawns a replacement.
9300async fn wait_for_registration_release(
9301    registry: &Registry,
9302    module_id: &str,
9303    wait: Duration,
9304) -> Result<(), SuperviseError> {
9305    wait_for_slot_registration_release(
9306        registry,
9307        crate::registry::RegistrationSlot::Active(module_id),
9308        wait,
9309    )
9310    .await
9311}
9312
9313/// Wait for the registration in `slot` to go away.
9314///
9315/// Keyed on the slot rather than the bare module id because a successful swap
9316/// never empties the id's active slot (the promoted candidate is in it), so an
9317/// id-keyed wait for the incumbent's release would always time out. Draining a
9318/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9319/// incumbent's connection instead.
9320async fn wait_for_slot_registration_release(
9321    registry: &Registry,
9322    slot: crate::registry::RegistrationSlot<'_>,
9323    wait: Duration,
9324) -> Result<(), SuperviseError> {
9325    let deadline = Instant::now() + wait;
9326    let mut release_events = registration_release_events().subscribe();
9327    let still_active = |registration: &crate::registry::ModuleRegistration| {
9328        SuperviseError::RegistrationStillActive {
9329            module_id: registration.manifest.module_id.clone(),
9330            waited: wait,
9331        }
9332    };
9333    loop {
9334        let _observed_generation = *release_events.borrow_and_update();
9335        let Some(registration) = registry
9336            .registration(slot)
9337            .map_err(SuperviseError::Registry)?
9338        else {
9339            return Ok(());
9340        };
9341
9342        let now = Instant::now();
9343        if now >= deadline {
9344            return Err(still_active(&registration));
9345        }
9346
9347        let remaining = deadline.saturating_duration_since(now);
9348        match timeout(remaining, release_events.changed()).await {
9349            Ok(Ok(())) | Ok(Err(_)) => {}
9350            Err(_) => return Err(still_active(&registration)),
9351        }
9352    }
9353}
9354
9355#[cfg(test)]
9356mod slot_registration_wait_tests {
9357    use super::*;
9358    use crate::registry::{ConnectionId, RegistrationSlot};
9359    use subc_protocol::manifest::ModuleManifest;
9360
9361    #[tokio::test]
9362    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9363        let registry = Arc::new(Registry::default());
9364        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9365        let runtime = supervisor.runtime_config();
9366        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9367        let spec = ModuleSpec {
9368            module_id: "enable-stale-registration".to_string(),
9369            program: PathBuf::from("/missing/enable-retry-test"),
9370            args: Vec::new(),
9371            env: Vec::new(),
9372            reserved: false,
9373            reserved_prefixes: Vec::new(),
9374            protocol: ModuleProtocol::Subc,
9375            overlap: Default::default(),
9376        };
9377        let connection = ConnectionId::new(90);
9378        registry
9379            .register_with_control_ops(
9380                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9381                1,
9382                connection,
9383                Vec::new(),
9384            )
9385            .unwrap();
9386        let mut child = None;
9387        let err = set_child_enabled(
9388            &spec,
9389            &runtime,
9390            &registry,
9391            &supervisor.process_liveness,
9392            &snapshot,
9393            &mut child,
9394            true,
9395        )
9396        .await
9397        .unwrap_err();
9398        assert!(matches!(
9399            err,
9400            SuperviseError::RegistrationStillActive { .. }
9401        ));
9402        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9403        assert!(child.is_none());
9404        registry.deregister_connection(connection).unwrap();
9405        let err = set_child_enabled(
9406            &spec,
9407            &runtime,
9408            &registry,
9409            &supervisor.process_liveness,
9410            &snapshot,
9411            &mut child,
9412            true,
9413        )
9414        .await
9415        .unwrap_err();
9416        assert!(
9417            matches!(err, SuperviseError::Spawn { .. }),
9418            "second enable must attempt a spawn: {err}"
9419        );
9420        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9421    }
9422
9423    const INCUMBENT: u64 = 1;
9424    const CANDIDATE: u64 = 2;
9425
9426    fn swapped_registry() -> Arc<Registry> {
9427        let registry = Arc::new(Registry::default());
9428        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9429        registry
9430            .register_with_control_ops(
9431                manifest.clone(),
9432                1,
9433                ConnectionId::new(INCUMBENT),
9434                Vec::new(),
9435            )
9436            .unwrap();
9437        registry
9438            .register_candidate_with_control_ops(
9439                manifest,
9440                1,
9441                ConnectionId::new(CANDIDATE),
9442                Vec::new(),
9443            )
9444            .unwrap();
9445        registry
9446    }
9447
9448    /// After a promotion the id's active slot is held by the new process, so an
9449    /// id-keyed wait for the incumbent's release can never succeed; the
9450    /// connection-keyed wait completes as soon as the incumbent deregisters.
9451    #[tokio::test]
9452    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9453        let registry = swapped_registry();
9454        registry.promote_candidate("m").unwrap().unwrap();
9455
9456        assert!(matches!(
9457            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9458            Err(SuperviseError::RegistrationStillActive { .. })
9459        ));
9460
9461        // Still held while the incumbent's connection has not deregistered.
9462        assert!(matches!(
9463            wait_for_slot_registration_release(
9464                &registry,
9465                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9466                Duration::from_millis(50),
9467            )
9468            .await,
9469            Err(SuperviseError::RegistrationStillActive { .. })
9470        ));
9471
9472        let releaser = Arc::clone(&registry);
9473        let release = tokio::spawn(async move {
9474            sleep(Duration::from_millis(20)).await;
9475            releaser
9476                .deregister_connection(ConnectionId::new(INCUMBENT))
9477                .unwrap();
9478            notify_registration_release();
9479        });
9480        wait_for_slot_registration_release(
9481            &registry,
9482            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9483            Duration::from_secs(5),
9484        )
9485        .await
9486        .expect("the incumbent's own registration is released");
9487        release.await.unwrap();
9488        assert!(registry.get_module("m").unwrap().is_some());
9489    }
9490
9491    /// The candidate slot is waited on separately from the active slot: the
9492    /// incumbent's registration neither holds up nor stands in for it.
9493    #[tokio::test]
9494    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9495        let registry = swapped_registry();
9496        assert!(matches!(
9497            wait_for_slot_registration_release(
9498                &registry,
9499                RegistrationSlot::Candidate("m"),
9500                Duration::from_millis(50),
9501            )
9502            .await,
9503            Err(SuperviseError::RegistrationStillActive { .. })
9504        ));
9505        registry
9506            .deregister_connection(ConnectionId::new(CANDIDATE))
9507            .unwrap();
9508        wait_for_slot_registration_release(
9509            &registry,
9510            RegistrationSlot::Candidate("m"),
9511            Duration::from_millis(50),
9512        )
9513        .await
9514        .expect("a candidate slot with no candidate is released");
9515        assert!(registry
9516            .registration(RegistrationSlot::Active("m"))
9517            .unwrap()
9518            .is_some());
9519    }
9520}
9521
9522fn classify_exit(status: &ExitStatus) -> ExitReport {
9523    ExitReport {
9524        kind: if status.success() {
9525            ExitKind::Clean
9526        } else {
9527            ExitKind::Crash
9528        },
9529        code: status.code(),
9530        signal: exit_signal(status),
9531        at_ms: unix_ms_now(),
9532    }
9533}
9534
9535/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9536/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9537/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9538/// disposition still must be `Failed` so the terminal ring is not silently missing
9539/// an entry, matching what `fail_snapshot` records for this same arm.
9540fn wait_error_exit_report() -> ExitReport {
9541    ExitReport {
9542        kind: ExitKind::Crash,
9543        code: None,
9544        signal: None,
9545        at_ms: unix_ms_now(),
9546    }
9547}
9548
9549#[cfg(unix)]
9550fn exit_signal(status: &ExitStatus) -> Option<i32> {
9551    use std::os::unix::process::ExitStatusExt;
9552
9553    status.signal()
9554}
9555
9556#[cfg(not(unix))]
9557fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9558    None
9559}
9560
9561/// Give an operator-touched module its full crash budget back.
9562///
9563/// Named for the counter it used to zero; it now empties the in-window ring,
9564/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9565/// ledger of what happened survives every operator action.
9566fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9567    update_snapshot(snapshot, Some(module_id), |state| {
9568        state.clear_crash_restarts();
9569    })
9570}
9571
9572fn set_running(
9573    snapshot: &SharedSnapshot,
9574    child: &SupervisedChild,
9575    module_id: &str,
9576    spawn_events: &SpawnEventFeed,
9577) -> Result<(), SuperviseError> {
9578    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9579        module_id: Some(module_id.to_string()),
9580    })?;
9581    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9582    if std::mem::take(&mut state.coalesced_restart_pending) {
9583        let generation = state.spawn_generation;
9584        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9585    }
9586    state.drain_disposition_detail = None;
9587    state.spawn_failure = None;
9588    // Every caller of this is a plain spawn, which always uses the primary key;
9589    // a promoted swap candidate sets the flag itself after this returns.
9590    state.in_alternate_slot = false;
9591    state.configuration_updated_since_spawn = false;
9592    state.spawned_protocol = Some(child.protocol);
9593    state.state = ModuleState::Running;
9594    state.enabled = true;
9595    state.process_alive = true;
9596    state.pid = child.id();
9597    #[cfg(target_os = "macos")]
9598    {
9599        state.report_ready = Some(Arc::clone(&child.report_ready));
9600    }
9601    state.spawned_at_ms = Some(child.spawned_at_ms);
9602    state.spawned_from = Some(child.spawned_from.clone());
9603    state.spawned_file_identity = child.spawned_file_identity;
9604    state.process_start_time = child.process_start_time;
9605    Ok(())
9606}
9607
9608fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9609    state.process_alive = false;
9610    state.spawned_protocol = None;
9611    state.pid = None;
9612    #[cfg(target_os = "macos")]
9613    {
9614        state.report_ready = None;
9615    }
9616    state.spawned_at_ms = None;
9617    state.spawned_from = None;
9618    state.spawned_file_identity = None;
9619    state.process_start_time = None;
9620    state.deliberate_severance = None;
9621}
9622
9623#[cfg(test)]
9624fn record_deliberate_severance(
9625    snapshot: &SharedSnapshot,
9626    identity: ProcessIdentity,
9627) -> Result<(), SuperviseError> {
9628    update_snapshot(snapshot, None, |state| {
9629        state.deliberate_severance = Some(identity);
9630    })
9631}
9632
9633fn apply_deliberate_severance_marker(
9634    snapshot: &SharedSnapshot,
9635    exited_identity: Option<ProcessIdentity>,
9636    mut exit_report: ExitReport,
9637) -> ExitReport {
9638    let marker = lock_snapshot(snapshot)
9639        .ok()
9640        .and_then(|mut state| state.deliberate_severance.take());
9641    if marker.is_some() && marker == exited_identity {
9642        exit_report.kind = ExitKind::DeliberateSeverance;
9643    }
9644    exit_report
9645}
9646
9647fn classify_reaped_child_exit(
9648    snapshot: &SharedSnapshot,
9649    child: &SupervisedChild,
9650    status: &ExitStatus,
9651) -> ExitReport {
9652    let _ = update_snapshot(snapshot, None, |state| {
9653        state.reaped_pid = Some(child.pid);
9654        state.spawn_failure = child.spawn_failure.clone();
9655    });
9656    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9657}
9658
9659fn fail_snapshot(
9660    snapshot: &SharedSnapshot,
9661    module_id: Option<&str>,
9662    last_exit: Option<ExitReport>,
9663) {
9664    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9665        state.state = ModuleState::Failed;
9666        clear_current_process_facts(state);
9667        if let Some(last_exit) = last_exit {
9668            state.last_exit = Some(last_exit);
9669        }
9670    }) {
9671        error!(error = %err, "failed to mark supervisor state failed");
9672    }
9673}
9674
9675fn update_snapshot(
9676    snapshot: &SharedSnapshot,
9677    module_id: Option<&str>,
9678    update: impl FnOnce(&mut SupervisorSnapshot),
9679) -> Result<(), SuperviseError> {
9680    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9681        module_id: module_id.map(ToOwned::to_owned),
9682    })?;
9683    update(&mut state);
9684    Ok(())
9685}
9686
9687const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9688
9689fn lock_snapshot_for_control<'a>(
9690    snapshot: &'a SharedSnapshot,
9691    module_id: &str,
9692    caller: &'static str,
9693) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9694    let started_at = Instant::now();
9695    let guard = lock_snapshot(snapshot)?;
9696    let waited = started_at.elapsed();
9697    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9698        warn!(
9699            module_id = %module_id,
9700            waited_ms = waited.as_millis() as u64,
9701            caller = %caller,
9702            "slow snapshot lock"
9703        );
9704    }
9705    Ok(guard)
9706}
9707
9708fn lock_snapshot(
9709    snapshot: &SharedSnapshot,
9710) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9711    snapshot
9712        .lock()
9713        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9714}
9715
9716#[cfg(test)]
9717mod terminal_history_tests {
9718    use std::{
9719        path::PathBuf,
9720        sync::Arc,
9721        time::{Duration, Instant},
9722    };
9723
9724    use tokio::{sync::mpsc, time::sleep};
9725
9726    use super::{
9727        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9728        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9729        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9730        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9731        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9732        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedConfiguration,
9733        SupervisedModule, SupervisedModuleInner, Supervisor, SupervisorHandle,
9734        SupervisorHealthStatus, SupervisorSnapshot,
9735    };
9736    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9737    // use for their own wall-clock deadlines: crash-restart instants must be on
9738    // the same clock the production code stamps them with, which is tokio's (and
9739    // is what `start_paused` tests can move).
9740    use super::Instant as ClockInstant;
9741    use crate::{
9742        registry::Registry,
9743        terminal_ring::{TerminalRing, TerminalRingConfig},
9744    };
9745    use std::sync::Mutex;
9746    use subc_control::TerminalDisposition;
9747
9748    /// See the twin in `control.rs` for why this derives the path from
9749    /// `current_exe()` and why the existence check is here: `--lib` alone does
9750    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9751    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9752    pub(super) fn fake_aft_stub_path() -> PathBuf {
9753        let mut path = std::env::current_exe().expect("current_exe available in tests");
9754        path.pop();
9755        path.pop();
9756        path.push(if cfg!(windows) {
9757            "fake-aft-stub.exe"
9758        } else {
9759            "fake-aft-stub"
9760        });
9761        assert!(
9762            path.exists(),
9763            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9764             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9765            path.display()
9766        );
9767        path
9768    }
9769
9770    #[test]
9771    fn reserved_never_spawned_refuses_every_hello() {
9772        // The canary hole: a reserved id whose module has never spawned had NO
9773        // gate entry and admitted anyone -- the reservation protected the nonce
9774        // holder, not the NAME. Now the entry is present with no legitimate
9775        // holder and refuses all comers.
9776        let supervisor = SupervisorHandle::default();
9777        supervisor.apply_identity_configuration(&ModuleSpec {
9778            module_id: "never-spawned".to_string(),
9779            program: PathBuf::from("/usr/bin/false"),
9780            args: Vec::new(),
9781            env: Vec::new(),
9782            reserved: true,
9783            reserved_prefixes: Vec::new(),
9784            protocol: ModuleProtocol::Subc,
9785            overlap: Default::default(),
9786        });
9787        assert!(
9788            supervisor
9789                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9790                .is_some(),
9791            "forged nonce must refuse on a reserved never-spawned id"
9792        );
9793        assert!(
9794            supervisor
9795                .reserved_hello_rejection("never-spawned", None)
9796                .is_some(),
9797            "absent nonce must refuse on a reserved never-spawned id"
9798        );
9799        // And a real spawn nonce minted later admits exactly that nonce.
9800        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9801        supervisor.apply_identity_configuration(&ModuleSpec {
9802            module_id: "never-spawned".to_string(),
9803            program: PathBuf::from("/usr/bin/false"),
9804            args: Vec::new(),
9805            env: Vec::new(),
9806            reserved: true,
9807            reserved_prefixes: Vec::new(),
9808            protocol: ModuleProtocol::Subc,
9809            overlap: Default::default(),
9810        });
9811        assert!(supervisor
9812            .reserved_hello_rejection("never-spawned", Some("minted"))
9813            .is_none());
9814        assert!(supervisor
9815            .reserved_hello_rejection("never-spawned", Some("forged"))
9816            .is_some());
9817    }
9818
9819    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9820    /// happened, which is what "spent budget" looks like to every reader.
9821    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9822        let now = ClockInstant::now();
9823        for _ in 0..count {
9824            state.crash_restarts.push_back(now);
9825        }
9826    }
9827
9828    /// Age the oldest recorded restart out of `window`, standing in for the hours
9829    /// that would otherwise have to pass. Injecting the instant is the point: a
9830    /// test that slept a real window would take ten minutes and still prove less.
9831    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9832        let aged = state
9833            .crash_restarts
9834            .front()
9835            .expect("a crash restart must be recorded before it can be aged")
9836            .checked_sub(window + Duration::from_secs(1))
9837            .expect("the test clock is far enough from its origin to age an instant");
9838        state.crash_restarts[0] = aged;
9839    }
9840
9841    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9842        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9843        seed_crash_restarts(&mut state, count);
9844        state
9845    }
9846
9847    #[test]
9848    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9849        let policy = RestartPolicy::new(3, Duration::ZERO);
9850        let now = ClockInstant::now();
9851        assert!(daemon_will_restart(
9852            &mut snapshot_with_restarts(true, 2),
9853            &policy,
9854            now
9855        ));
9856        assert!(!daemon_will_restart(
9857            &mut snapshot_with_restarts(true, 3),
9858            &policy,
9859            now
9860        ));
9861        assert!(!daemon_will_restart(
9862            &mut snapshot_with_restarts(false, 0),
9863            &policy,
9864            now
9865        ));
9866    }
9867
9868    #[test]
9869    fn crash_restart_backoff_escalates_with_in_window_count() {
9870        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9871            .with_max_backoff(Duration::from_secs(30));
9872        let now = ClockInstant::now();
9873        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9874        let schedules = (0..4)
9875            .map(|_| {
9876                state
9877                    .next_crash_restart(&policy, now)
9878                    .expect("the test policy allows four crash restarts")
9879            })
9880            .collect::<Vec<_>>();
9881
9882        assert_eq!(
9883            schedules
9884                .iter()
9885                .map(|schedule| schedule.restart_in_window)
9886                .collect::<Vec<_>>(),
9887            vec![0, 1, 2, 3]
9888        );
9889        assert_eq!(
9890            schedules
9891                .iter()
9892                .map(|schedule| schedule.delay)
9893                .collect::<Vec<_>>(),
9894            vec![
9895                Duration::from_millis(100),
9896                Duration::from_secs(1),
9897                Duration::from_secs(10),
9898                Duration::from_secs(30),
9899            ]
9900        );
9901    }
9902
9903    #[test]
9904    fn crash_restart_backoff_resets_after_ring_clear() {
9905        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9906        let now = ClockInstant::now();
9907        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9908        assert_eq!(
9909            state.next_crash_restart(&policy, now).unwrap().delay,
9910            Duration::from_millis(100)
9911        );
9912        assert_eq!(
9913            state.next_crash_restart(&policy, now).unwrap().delay,
9914            Duration::from_secs(1)
9915        );
9916
9917        state.clear_crash_restarts();
9918        let schedule = state
9919            .next_crash_restart(&policy, now)
9920            .expect("a cleared ring must allow another restart");
9921        assert_eq!(schedule.restart_in_window, 0);
9922        assert_eq!(schedule.delay, Duration::from_millis(100));
9923    }
9924
9925    #[test]
9926    fn crash_restart_backoff_ignores_aged_restarts() {
9927        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9928        let now = ClockInstant::now();
9929        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9930        state
9931            .next_crash_restart(&policy, now)
9932            .expect("the first restart is allowed");
9933        state
9934            .next_crash_restart(&policy, now)
9935            .expect("the second restart is allowed");
9936        state.crash_restarts[0] = now
9937            .checked_sub(policy.window + Duration::from_secs(1))
9938            .expect("the fake clock can age a restart past the window");
9939
9940        let schedule = state
9941            .next_crash_restart(&policy, now)
9942            .expect("an aged restart must release its slot");
9943        assert_eq!(schedule.restart_in_window, 1);
9944        assert_eq!(schedule.delay, Duration::from_secs(1));
9945        assert_eq!(state.crash_restarts.len(), 2);
9946    }
9947
9948    /// The budget is a rate: the same three spent restarts refuse a respawn
9949    /// while they are recent and allow one once they have aged past the window.
9950    /// Nothing about the module changed in between, which is the whole point.
9951    #[test]
9952    fn a_budget_spent_before_the_window_no_longer_refuses() {
9953        let policy = RestartPolicy::new(3, Duration::ZERO);
9954        let mut state = snapshot_with_restarts(true, 3);
9955        let now = ClockInstant::now();
9956        assert!(!daemon_will_restart(&mut state, &policy, now));
9957
9958        assert!(daemon_will_restart(
9959            &mut state,
9960            &policy,
9961            now + policy.window + Duration::from_secs(1)
9962        ));
9963        assert!(
9964            state.crash_restarts.is_empty(),
9965            "reading the budget must drop the instants that left the window"
9966        );
9967    }
9968
9969    fn module_with_recovery_snapshot(
9970        state: ModuleState,
9971        enabled: bool,
9972        restart_count: u32,
9973    ) -> SupervisedModule {
9974        let registry = Arc::new(Registry::default());
9975        let supervisor =
9976            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9977        let runtime = supervisor.runtime_config();
9978        let spec = ModuleSpec {
9979            module_id: "recovery-snapshot".to_string(),
9980            program: fake_aft_stub_path(),
9981            args: Vec::new(),
9982            env: Vec::new(),
9983            reserved: false,
9984            reserved_prefixes: Vec::new(),
9985            protocol: ModuleProtocol::Subc,
9986            overlap: Default::default(),
9987        };
9988        let mut snapshot = SupervisorSnapshot::new(state, enabled);
9989        seed_crash_restarts(&mut snapshot, restart_count);
9990        // These tests read synthetic snapshots. A real child and monitor would
9991        // race those reads by replacing the requested state during startup.
9992        let (commands, _rx) = mpsc::channel(4);
9993        SupervisedModule {
9994            inner: Arc::new(SupervisedModuleInner {
9995                module_id: spec.module_id.clone(),
9996                registry,
9997                snapshot: Arc::new(Mutex::new(snapshot)),
9998                configuration: Arc::new(Mutex::new(SupervisedConfiguration {
9999                    spec,
10000                    health: runtime.health,
10001                })),
10002                stderr_ring: runtime.stderr_ring,
10003                terminal_ring: runtime.terminal_ring,
10004                commands,
10005                monitor: Mutex::new(None),
10006                restart_policy: runtime.restart_policy,
10007                effective_drain_timeout: runtime.effective_drain_timeout,
10008                provenance_probe: supervisor.provenance_probe.clone(),
10009            }),
10010        }
10011    }
10012
10013    #[cfg(target_os = "linux")]
10014    #[tokio::test]
10015    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
10016        let supervisor =
10017            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
10018                .with_cgroup_placement(None);
10019        let result = supervisor.spawn(ModuleSpec {
10020            module_id: "no-cgroup-placement".to_string(),
10021            program: fake_aft_stub_path(),
10022            args: Vec::new(),
10023            env: Vec::new(),
10024            reserved: false,
10025            reserved_prefixes: Vec::new(),
10026            protocol: ModuleProtocol::Subc,
10027            overlap: Default::default(),
10028        });
10029
10030        assert!(
10031            result.is_ok(),
10032            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
10033        );
10034    }
10035
10036    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10037    async fn undecided_snapshot_uses_shared_restart_predicate() {
10038        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
10039            .will_recover_after_connection_loss()
10040            .unwrap());
10041        assert!(
10042            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
10043                .will_recover_after_connection_loss()
10044                .unwrap()
10045        );
10046    }
10047
10048    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10049    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
10050        assert!(
10051            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
10052                .will_recover_after_connection_loss()
10053                .unwrap()
10054        );
10055    }
10056
10057    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10058    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
10059        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
10060            .will_recover_after_connection_loss()
10061            .unwrap());
10062        assert!(
10063            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
10064                .will_recover_after_connection_loss()
10065                .unwrap()
10066        );
10067    }
10068
10069    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10070    async fn warming_snapshot_is_limited_to_startup_phases() {
10071        for state in [
10072            ModuleState::Starting,
10073            ModuleState::Running,
10074            ModuleState::Restarting,
10075        ] {
10076            assert!(
10077                module_with_recovery_snapshot(state, true, 0)
10078                    .is_warming()
10079                    .unwrap(),
10080                "{state:?} should be warming"
10081            );
10082        }
10083        for state in [
10084            ModuleState::Unresponsive,
10085            ModuleState::Draining,
10086            ModuleState::Stopped,
10087            ModuleState::Failed,
10088            ModuleState::Disabled,
10089        ] {
10090            assert!(
10091                !module_with_recovery_snapshot(state, true, 0)
10092                    .is_warming()
10093                    .unwrap(),
10094                "{state:?} should not be warming"
10095            );
10096        }
10097    }
10098
10099    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10100    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
10101        let registry = Arc::new(Registry::default());
10102        let supervisor =
10103            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
10104        let module = supervisor
10105            .spawn(ModuleSpec {
10106                module_id: "terminal-history".to_string(),
10107                program: fake_aft_stub_path(),
10108                args: Vec::new(),
10109                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10110                reserved: false,
10111                reserved_prefixes: Vec::new(),
10112                protocol: ModuleProtocol::Subc,
10113                overlap: Default::default(),
10114            })
10115            .unwrap();
10116
10117        let deadline = Instant::now() + Duration::from_secs(5);
10118        loop {
10119            let history = module.terminal_history();
10120            if history.entries.len() == 2 {
10121                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
10122                assert_eq!(history.dropped, 0);
10123                assert_eq!(
10124                    history
10125                        .entries
10126                        .iter()
10127                        .map(|entry| entry.exit_code)
10128                        .collect::<Vec<_>>(),
10129                    vec![Some(23), Some(23)]
10130                );
10131                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
10132                return;
10133            }
10134            assert!(
10135                Instant::now() < deadline,
10136                "module did not retain two terminal exits: {history:?}"
10137            );
10138            sleep(Duration::from_millis(10)).await;
10139        }
10140    }
10141
10142    /// A disable issued while a crash respawn is still backing off must preempt
10143    /// that respawn: the operator's stop wins, the disable must not queue behind
10144    /// the backoff, and the module must never come back up afterwards.
10145    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10146    async fn disable_during_crash_backoff_cancels_pending_respawn() {
10147        let backoff = Duration::from_secs(2);
10148        let supervisor = Supervisor::new_for_test(
10149            Arc::new(Registry::default()),
10150            RestartPolicy::new(10, backoff),
10151        );
10152        let module = supervisor
10153            .spawn(ModuleSpec {
10154                module_id: "disable-during-backoff".to_string(),
10155                program: fake_aft_stub_path(),
10156                args: Vec::new(),
10157                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10158                reserved: false,
10159                reserved_prefixes: Vec::new(),
10160                protocol: ModuleProtocol::Subc,
10161                overlap: Default::default(),
10162            })
10163            .unwrap();
10164
10165        // Wait for the first crash to put the module into its backoff window.
10166        let deadline = Instant::now() + Duration::from_secs(5);
10167        loop {
10168            if module.status().unwrap().state == ModuleState::Restarting {
10169                break;
10170            }
10171            assert!(
10172                Instant::now() < deadline,
10173                "module never entered the crash backoff"
10174            );
10175            sleep(Duration::from_millis(10)).await;
10176        }
10177
10178        let started = Instant::now();
10179        module.set_enabled(false).await.unwrap();
10180        let waited = started.elapsed();
10181
10182        assert!(
10183            waited < backoff / 2,
10184            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10185        );
10186        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10187
10188        // Outlast the backoff: the respawn it was counting down to must never run.
10189        sleep(backoff + Duration::from_millis(500)).await;
10190        let status = module.status().unwrap();
10191        assert_eq!(status.state, ModuleState::Disabled);
10192        assert_eq!(
10193            status.spawn_generation, 1,
10194            "module respawned after the operator disabled it"
10195        );
10196    }
10197
10198    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
10199    /// the shape of nats-server, the program this rule exists for.
10200    #[cfg(unix)]
10201    fn protocol_none_sigterm_exits_clean_spec(
10202        module_id: &str,
10203        dir: &std::path::Path,
10204    ) -> (ModuleSpec, PathBuf, PathBuf) {
10205        let ready = dir.join("ready");
10206        let marker = dir.join("sigterm");
10207        let spec = ModuleSpec {
10208            module_id: module_id.to_string(),
10209            program: fake_aft_stub_path(),
10210            args: Vec::new(),
10211            env: vec![
10212                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10213                (
10214                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10215                    marker.display().to_string(),
10216                ),
10217                (
10218                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10219                    ready.display().to_string(),
10220                ),
10221            ],
10222            reserved: false,
10223            reserved_prefixes: Vec::new(),
10224            protocol: ModuleProtocol::None,
10225            overlap: Default::default(),
10226        };
10227        (spec, ready, marker)
10228    }
10229
10230    /// Wait for a file the child writes, so a signal is never sent before the
10231    /// child's SIGTERM handler is installed (the default disposition would
10232    /// kill it by signal and the exit would not be clean).
10233    #[cfg(unix)]
10234    async fn wait_for_file(path: &std::path::Path) {
10235        let deadline = Instant::now() + Duration::from_secs(10);
10236        while !path.exists() {
10237            assert!(
10238                Instant::now() < deadline,
10239                "{} never appeared",
10240                path.display()
10241            );
10242            sleep(Duration::from_millis(10)).await;
10243        }
10244    }
10245
10246    /// A protocol-none module that exits 0 because something OUTSIDE the
10247    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10248    /// the crash-path disposition rather than `stopped`.
10249    #[cfg(unix)]
10250    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10251    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10252        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10253        let (spec, ready, marker) =
10254            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10255        let supervisor = Supervisor::new_for_test(
10256            Arc::new(Registry::default()),
10257            RestartPolicy::new(3, Duration::ZERO),
10258        );
10259        let module = supervisor.spawn(spec).unwrap();
10260        wait_for_file(&ready).await;
10261        // The ready file proves the child installed its SIGTERM handler, not
10262        // that the supervisor has processed the privacy trampoline's exec
10263        // acknowledgement. On macOS status withholds the pid until then.
10264        let deadline = Instant::now() + Duration::from_secs(10);
10265        let first_pid = loop {
10266            let status = module.status().unwrap();
10267            if status.state == ModuleState::Running {
10268                if let Some(pid) = status.pid {
10269                    break pid;
10270                }
10271            }
10272            assert!(
10273                Instant::now() < deadline,
10274                "a running module must report its pid after exec confirmation: {status:?}"
10275            );
10276            sleep(Duration::from_millis(10)).await;
10277        };
10278
10279        rustix::process::kill_process(
10280            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10281            rustix::process::Signal::TERM,
10282        )
10283        .unwrap();
10284
10285        let deadline = Instant::now() + Duration::from_secs(10);
10286        let respawned = loop {
10287            let status = module.status().unwrap();
10288            if status.state == ModuleState::Running
10289                && status.pid.is_some_and(|pid| pid != first_pid)
10290            {
10291                break status;
10292            }
10293            assert!(
10294                Instant::now() < deadline,
10295                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10296            );
10297            sleep(Duration::from_millis(10)).await;
10298        };
10299        assert_eq!(respawned.spawn_generation, 2);
10300        assert!(
10301            marker.exists(),
10302            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10303        );
10304
10305        let history = module.terminal_history();
10306        assert_eq!(history.entries.len(), 1, "{history:?}");
10307        let entry = &history.entries[0];
10308        assert_eq!(entry.exit_code, Some(0));
10309        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10310        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10311
10312        module.stop().await.unwrap();
10313    }
10314
10315    /// Repeated unrequested clean exits of a protocol-none module spend the
10316    /// restart budget exactly as crashes do, and the module ends `failed` with
10317    /// the budget named.
10318    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10319    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10320        let supervisor = Supervisor::new_for_test(
10321            Arc::new(Registry::default()),
10322            RestartPolicy::new(1, Duration::ZERO),
10323        );
10324        let module = supervisor
10325            .spawn(ModuleSpec {
10326                module_id: "none-clean-exit-budget".to_string(),
10327                program: fake_aft_stub_path(),
10328                args: Vec::new(),
10329                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10330                reserved: false,
10331                reserved_prefixes: Vec::new(),
10332                protocol: ModuleProtocol::None,
10333                overlap: Default::default(),
10334            })
10335            .unwrap();
10336
10337        // Failed follows the terminal write; this deadline only bounds a hang,
10338        // not an assumed duration for the two launches or their exit recording.
10339        let deadline = Instant::now() + Duration::from_secs(10);
10340        loop {
10341            let status = module.status().unwrap();
10342            if status.state == ModuleState::Failed {
10343                break;
10344            }
10345            assert!(
10346                Instant::now() < deadline,
10347                "module never exhausted its budget: {status:?} {:?}",
10348                module.terminal_history()
10349            );
10350            sleep(Duration::from_millis(10)).await;
10351        }
10352        let history = module.terminal_history();
10353        assert_eq!(
10354            history
10355                .entries
10356                .iter()
10357                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10358                .collect::<Vec<_>>(),
10359            vec![
10360                (Some(0), TerminalDisposition::Restarting),
10361                (Some(0), TerminalDisposition::Failed),
10362            ]
10363        );
10364        let detail = history.entries[1]
10365            .disposition_detail
10366            .as_deref()
10367            .expect("a budget failure names the budget");
10368        assert!(detail.contains("max_restarts=1"), "{detail}");
10369        assert_eq!(module.status().unwrap().spawn_generation, 2);
10370    }
10371
10372    #[test]
10373    fn restart_budget_failure_is_published_after_its_terminal_record() {
10374        let supervisor = Supervisor::new_for_test(
10375            Arc::new(Registry::default()),
10376            RestartPolicy::new(0, Duration::ZERO),
10377        );
10378        let runtime = supervisor.runtime_config();
10379        let spec = ModuleSpec {
10380            protocol: ModuleProtocol::None,
10381            ..windowed_crash_spec("budget-publication-order")
10382        };
10383        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
10384            ModuleState::Running,
10385            true,
10386        )));
10387        let reaped_snapshot = snapshot.clone();
10388        let events = runtime.spawn_events.clone();
10389        let history = runtime.terminal_ring.clone();
10390
10391        // Exit recording takes the event-feed lock before the history lock.
10392        // Holding it pauses the writer after choosing a disposition but before
10393        // recording history, without assuming anything about scheduler timing.
10394        let before_record = events.0.lock().unwrap();
10395        let reap = std::thread::spawn(move || {
10396            tokio::runtime::Builder::new_current_thread()
10397                .enable_all()
10398                .build()
10399                .unwrap()
10400                .block_on(on_child_exit(
10401                    &spec,
10402                    runtime.restart_policy,
10403                    &supervisor.registry,
10404                    &reaped_snapshot,
10405                    &runtime.terminal_ring,
10406                    &runtime.spawn_events,
10407                    &runtime.child_roster,
10408                    ExitReport {
10409                        kind: ExitKind::Clean,
10410                        code: Some(0),
10411                        signal: None,
10412                        at_ms: 1,
10413                    },
10414                ))
10415        });
10416        // The deadline bounds a hung writer only; last_exit is the handshake.
10417        let deadline = Instant::now() + Duration::from_secs(10);
10418        let before_state = loop {
10419            let state = lock_snapshot(&snapshot).unwrap();
10420            if state.last_exit.is_some() {
10421                break state.state;
10422            }
10423            drop(state);
10424            assert!(Instant::now() < deadline, "exit decision was not reached");
10425            std::thread::yield_now();
10426        };
10427        let before_history = history.lock().unwrap().snapshot();
10428        drop(before_record);
10429        assert!(matches!(reap.join().unwrap(), NextAction::Stop { .. }));
10430        assert!(before_history.entries.is_empty());
10431        assert_ne!(
10432            before_state,
10433            ModuleState::Failed,
10434            "Failed was visible before its terminal record could be written"
10435        );
10436        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10437        let history = history.lock().unwrap().snapshot();
10438        assert_eq!(history.entries.len(), 1);
10439        assert_eq!(history.entries[0].exit_code, Some(0));
10440        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10441    }
10442
10443    #[tokio::test]
10444    async fn restart_budget_failure_remains_failed_when_journal_append_fails() {
10445        let dir = subc_test_support::TestTempDir::new("budget-journal-failure");
10446        let path = dir.join("terminals.jsonl");
10447        std::fs::create_dir(&path).unwrap();
10448        let supervisor = Supervisor::new_for_test(
10449            Arc::new(Registry::default()),
10450            RestartPolicy::new(0, Duration::ZERO),
10451        )
10452        .with_terminal_journal(path, "budget-journal-failure".into());
10453        let runtime = supervisor.runtime_config();
10454        let spec = windowed_crash_spec("budget-journal-failure");
10455        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10456        assert!(matches!(
10457            on_child_exit(
10458                &spec,
10459                runtime.restart_policy,
10460                &supervisor.registry,
10461                &snapshot,
10462                &runtime.terminal_ring,
10463                &runtime.spawn_events,
10464                &runtime.child_roster,
10465                crash_exit_report(1),
10466            )
10467            .await,
10468            NextAction::Stop { .. }
10469        ));
10470        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10471        let history = runtime
10472            .terminal_ring
10473            .lock()
10474            .unwrap()
10475            .durable_history(&spec.module_id);
10476        assert!(history.journal_write_failures > 0);
10477        assert_eq!(history.entries.len(), 1);
10478        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10479    }
10480
10481    #[test]
10482    fn restart_budget_failure_remains_failed_when_exit_recording_panics() {
10483        let supervisor = Supervisor::new_for_test(
10484            Arc::new(Registry::default()),
10485            RestartPolicy::new(0, Duration::ZERO),
10486        );
10487        let runtime = supervisor.runtime_config();
10488        let spec = windowed_crash_spec("budget-recording-panic");
10489        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10490        runtime.spawn_events.emit_spawned(&spec.module_id, 1, 1);
10491        // Exhausting the event sequence makes emit_exited panic before the
10492        // terminal write, exercising publication on the recording unwind.
10493        runtime.spawn_events.0.lock().unwrap().seq = u64::MAX;
10494        let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
10495            tokio::runtime::Builder::new_current_thread()
10496                .enable_all()
10497                .build()
10498                .unwrap()
10499                .block_on(on_child_exit(
10500                    &spec,
10501                    runtime.restart_policy,
10502                    &supervisor.registry,
10503                    &snapshot,
10504                    &runtime.terminal_ring,
10505                    &runtime.spawn_events,
10506                    &runtime.child_roster,
10507                    crash_exit_report(1),
10508                ))
10509        }));
10510        let panic = result.err().expect("recording must still unwind");
10511        assert_eq!(
10512            panic.downcast_ref::<String>().map(String::as_str),
10513            Some("spawn event sequence exhausted")
10514        );
10515        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10516    }
10517
10518    /// A stop the supervisor itself requests still stops a protocol-none
10519    /// module, even though the child answers the SIGTERM with exit 0.
10520    #[cfg(unix)]
10521    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10522    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10523        for disable in [false, true] {
10524            let label = if disable {
10525                "none-requested-disable"
10526            } else {
10527                "none-requested-stop"
10528            };
10529            let dir = subc_test_support::TestTempDir::new(label);
10530            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10531            let supervisor = Supervisor::new_for_test(
10532                Arc::new(Registry::default()),
10533                RestartPolicy::new(3, Duration::ZERO),
10534            );
10535            let module = supervisor.spawn(spec).unwrap();
10536            wait_for_file(&ready).await;
10537
10538            if disable {
10539                module.set_enabled(false).await.unwrap();
10540            } else {
10541                module.stop().await.unwrap();
10542            }
10543            assert!(
10544                marker.exists(),
10545                "{label}: the child must have left through its SIGTERM handler with exit 0"
10546            );
10547
10548            // Long enough for a zero-backoff respawn to have happened if the
10549            // exit had been treated as a crash.
10550            sleep(Duration::from_millis(500)).await;
10551            let status = module.status().unwrap();
10552            let expected = if disable {
10553                ModuleState::Disabled
10554            } else {
10555                ModuleState::Stopped
10556            };
10557            assert_eq!(status.state, expected, "{label}");
10558            assert_eq!(
10559                status.spawn_generation, 1,
10560                "{label}: respawned after a requested stop"
10561            );
10562            let history = module.terminal_history();
10563            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10564            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10565            assert_ne!(
10566                history.entries[0].disposition,
10567                TerminalDisposition::Restarting,
10568                "{label}"
10569            );
10570        }
10571    }
10572
10573    /// A subc-wire module that exits 0 on its own is still a stop: the
10574    /// protocol-none rule must not reach it.
10575    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10576    async fn subc_wire_clean_exit_is_still_a_stop() {
10577        let supervisor = Supervisor::new_for_test(
10578            Arc::new(Registry::default()),
10579            RestartPolicy::new(3, Duration::ZERO),
10580        );
10581        let module = supervisor
10582            .spawn(ModuleSpec {
10583                module_id: "wire-clean-exit".to_string(),
10584                program: fake_aft_stub_path(),
10585                args: Vec::new(),
10586                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10587                reserved: false,
10588                reserved_prefixes: Vec::new(),
10589                protocol: ModuleProtocol::Subc,
10590                overlap: Default::default(),
10591            })
10592            .unwrap();
10593
10594        let deadline = Instant::now() + Duration::from_secs(10);
10595        while module.terminal_history().entries.is_empty() {
10596            assert!(Instant::now() < deadline, "module never exited");
10597            sleep(Duration::from_millis(10)).await;
10598        }
10599        // Long enough for a zero-backoff respawn to have happened.
10600        sleep(Duration::from_millis(500)).await;
10601        let status = module.status().unwrap();
10602        assert_eq!(status.state, ModuleState::Stopped);
10603        assert_eq!(status.spawn_generation, 1);
10604        let history = module.terminal_history();
10605        assert_eq!(history.entries.len(), 1, "{history:?}");
10606        assert_eq!(history.entries[0].exit_code, Some(0));
10607        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10608    }
10609
10610    #[cfg(unix)]
10611    #[tokio::test]
10612    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10613        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10614        let record = dir.join("live-children.json");
10615        let supervisor = Supervisor::new_for_test(
10616            Arc::new(Registry::default()),
10617            RestartPolicy::new(0, Duration::ZERO),
10618        );
10619        let mut runtime = supervisor.runtime_config();
10620        runtime.child_roster.record_to(record.clone());
10621        let gate = Arc::new(super::ReloadExitRecordGate::default());
10622        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10623        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10624        let spec = ModuleSpec {
10625            module_id: "reload-exit-roster".into(),
10626            program: fake_aft_stub_path(),
10627            args: Vec::new(),
10628            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10629            reserved: false,
10630            reserved_prefixes: Vec::new(),
10631            protocol: ModuleProtocol::Subc,
10632            overlap: Default::default(),
10633        };
10634        let mut child = None;
10635        let reload = super::finish_reload_child(
10636            &spec,
10637            &runtime,
10638            &supervisor.registry,
10639            &supervisor.process_liveness,
10640            &snapshot,
10641            &mut child,
10642        );
10643        tokio::pin!(reload);
10644        tokio::select! {
10645            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10646            _ = gate.reached.notified() => {}
10647        }
10648        assert!(runtime
10649            .terminal_ring
10650            .lock()
10651            .unwrap()
10652            .snapshot()
10653            .entries
10654            .is_empty());
10655        assert_eq!(
10656            crate::live_children::read_record(&record).unwrap().len(),
10657            1,
10658            "shutdown must still wait for the reaped child until its terminal record exists"
10659        );
10660        runtime.child_roster.close();
10661        gate.resume.notify_one();
10662        assert!(reload.await.is_err());
10663        assert!(crate::live_children::read_record(&record)
10664            .unwrap()
10665            .is_empty());
10666        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10667        assert_eq!(history.entries.len(), 1);
10668        assert_eq!(
10669            history.entries[0].disposition,
10670            TerminalDisposition::DaemonShutdown
10671        );
10672    }
10673
10674    /// Each restart-producing arm has its own state transition. Keeping their
10675    /// lifetime count assertions adjacent prevents a later new arm from silently
10676    /// spending budget without recording the historical restart.
10677    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10678    async fn every_restart_increment_path_advances_lifetime_count() {
10679        let supervisor = Supervisor::new_for_test(
10680            Arc::new(Registry::default()),
10681            RestartPolicy::new(1, Duration::ZERO),
10682        );
10683        let runtime = supervisor.runtime_config();
10684        let spec = ModuleSpec {
10685            module_id: "lifetime-increment-path".to_string(),
10686            program: PathBuf::from("/unused/lifetime-increment-path"),
10687            args: Vec::new(),
10688            env: Vec::new(),
10689            reserved: false,
10690            reserved_prefixes: Vec::new(),
10691            protocol: ModuleProtocol::Subc,
10692            overlap: Default::default(),
10693        };
10694
10695        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10696        assert!(matches!(
10697            on_child_exit(
10698                &spec,
10699                runtime.restart_policy,
10700                &supervisor.registry,
10701                &crash_snapshot,
10702                &runtime.terminal_ring,
10703                &runtime.spawn_events,
10704                &runtime.child_roster,
10705                ExitReport {
10706                    kind: ExitKind::Crash,
10707                    code: Some(1),
10708                    signal: None,
10709                    at_ms: 1,
10710                },
10711            )
10712            .await,
10713            NextAction::Restart { schedule: _ }
10714        ));
10715        let (crash_restarts, crash_lifetime) = {
10716            let state = lock_snapshot(&crash_snapshot).unwrap();
10717            (state.crash_restarts.len(), state.lifetime_restarts)
10718        };
10719        assert_eq!(crash_restarts, 1);
10720        assert_eq!(crash_lifetime, 1);
10721
10722        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10723        let mut health_child = None;
10724        assert!(matches!(
10725            health_restart_child(
10726                &spec,
10727                &runtime,
10728                &supervisor.registry,
10729                &supervisor.process_liveness,
10730                &health_snapshot,
10731                &mut health_child,
10732                SupervisorHealthStatus::Failing,
10733                None,
10734                2,
10735            )
10736            .await,
10737            Ok(())
10738        ));
10739        assert!(health_child.is_none());
10740        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10741        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10742        let (health_restarts, health_lifetime) = {
10743            let state = lock_snapshot(&health_snapshot).unwrap();
10744            (state.crash_restarts.len(), state.lifetime_restarts)
10745        };
10746        assert_eq!(health_restarts, 1);
10747        assert_eq!(health_lifetime, 1);
10748
10749        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10750        let mut reload_child = None;
10751        assert!(matches!(
10752            handle_reload_spawn_failure(
10753                &spec,
10754                &runtime,
10755                &supervisor.process_liveness,
10756                &reload_snapshot,
10757                &mut reload_child,
10758                "forced reload spawn failure".to_string(),
10759            )
10760            .await,
10761            Err(SuperviseError::ReloadFailed { .. })
10762        ));
10763        let (reload_restarts, reload_lifetime) = {
10764            let state = lock_snapshot(&reload_snapshot).unwrap();
10765            (state.crash_restarts.len(), state.lifetime_restarts)
10766        };
10767        assert_eq!(reload_restarts, 1);
10768        assert_eq!(reload_lifetime, 1);
10769    }
10770
10771    #[tokio::test]
10772    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10773        let supervisor = Supervisor::new_for_test(
10774            Arc::new(Registry::default()),
10775            RestartPolicy::new(3, Duration::ZERO),
10776        );
10777        let runtime = supervisor.runtime_config();
10778        let spec = ModuleSpec {
10779            module_id: "deliberately-severed".to_string(),
10780            program: PathBuf::from("/unused/deliberately-severed"),
10781            args: Vec::new(),
10782            env: Vec::new(),
10783            reserved: false,
10784            reserved_prefixes: Vec::new(),
10785            protocol: ModuleProtocol::Subc,
10786            overlap: Default::default(),
10787        };
10788        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10789        let process = ProcessIdentity {
10790            pid: 41,
10791            start_time: 101,
10792        };
10793        record_deliberate_severance(&snapshot, process).unwrap();
10794        let exit_report = apply_deliberate_severance_marker(
10795            &snapshot,
10796            Some(process),
10797            ExitReport {
10798                kind: ExitKind::Crash,
10799                code: Some(1),
10800                signal: None,
10801                at_ms: 1,
10802            },
10803        );
10804        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10805
10806        assert!(matches!(
10807            on_child_exit(
10808                &spec,
10809                runtime.restart_policy,
10810                &supervisor.registry,
10811                &snapshot,
10812                &runtime.terminal_ring,
10813                &runtime.spawn_events,
10814                &runtime.child_roster,
10815                exit_report,
10816            )
10817            .await,
10818            NextAction::Restart { schedule: _ }
10819        ));
10820        let state = lock_snapshot(&snapshot).unwrap();
10821        assert_eq!(state.lifetime_restarts, 1);
10822        assert_eq!(state.crash_restarts.len(), 0);
10823    }
10824
10825    #[tokio::test]
10826    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10827        let supervisor = Supervisor::new_for_test(
10828            Arc::new(Registry::default()),
10829            RestartPolicy::new(3, Duration::ZERO),
10830        );
10831        let runtime = supervisor.runtime_config();
10832        let spec = ModuleSpec {
10833            module_id: "genuine-crash".to_string(),
10834            program: PathBuf::from("/unused/genuine-crash"),
10835            args: Vec::new(),
10836            env: Vec::new(),
10837            reserved: false,
10838            reserved_prefixes: Vec::new(),
10839            protocol: ModuleProtocol::Subc,
10840            overlap: Default::default(),
10841        };
10842        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10843
10844        assert!(matches!(
10845            on_child_exit(
10846                &spec,
10847                runtime.restart_policy,
10848                &supervisor.registry,
10849                &snapshot,
10850                &runtime.terminal_ring,
10851                &runtime.spawn_events,
10852                &runtime.child_roster,
10853                ExitReport {
10854                    kind: ExitKind::Crash,
10855                    code: Some(1),
10856                    signal: None,
10857                    at_ms: 1,
10858                },
10859            )
10860            .await,
10861            NextAction::Restart { schedule: _ }
10862        ));
10863        let state = lock_snapshot(&snapshot).unwrap();
10864        assert_eq!(state.lifetime_restarts, 1);
10865        assert_eq!(state.crash_restarts.len(), 1);
10866    }
10867
10868    fn crash_exit_report(at_ms: u64) -> ExitReport {
10869        ExitReport {
10870            kind: ExitKind::Crash,
10871            code: Some(1),
10872            signal: None,
10873            at_ms,
10874        }
10875    }
10876
10877    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10878        ModuleSpec {
10879            module_id: module_id.to_string(),
10880            program: PathBuf::from("/unused").join(module_id),
10881            args: Vec::new(),
10882            env: Vec::new(),
10883            reserved: false,
10884            reserved_prefixes: Vec::new(),
10885            protocol: ModuleProtocol::Subc,
10886            overlap: Default::default(),
10887        }
10888    }
10889
10890    /// A real crash loop still stops. Three crashes with nothing aging out spend
10891    /// a budget of two and the third respawn is refused, and both surfaces an
10892    /// operator has -- the log line and the retained terminal record -- name the
10893    /// window rather than only the cap, because `max_restarts=2` alone is what
10894    /// this budget used to mean.
10895    #[tokio::test]
10896    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10897        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10898        let supervisor = Supervisor::new_for_test(
10899            Arc::new(Registry::default()),
10900            RestartPolicy::new(2, Duration::ZERO),
10901        );
10902        let runtime = supervisor.runtime_config();
10903        let spec = windowed_crash_spec("crash-loop-in-window");
10904        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10905
10906        for attempt in 1..=2 {
10907            assert!(
10908                matches!(
10909                    on_child_exit(
10910                        &spec,
10911                        runtime.restart_policy,
10912                        &supervisor.registry,
10913                        &snapshot,
10914                        &runtime.terminal_ring,
10915                        &runtime.spawn_events,
10916                        &runtime.child_roster,
10917                        crash_exit_report(attempt),
10918                    )
10919                    .await,
10920                    NextAction::Restart { schedule: _ }
10921                ),
10922                "crash {attempt} is inside the budget and must respawn"
10923            );
10924        }
10925
10926        assert!(matches!(
10927            on_child_exit(
10928                &spec,
10929                runtime.restart_policy,
10930                &supervisor.registry,
10931                &snapshot,
10932                &runtime.terminal_ring,
10933                &runtime.spawn_events,
10934                &runtime.child_roster,
10935                crash_exit_report(3),
10936            )
10937            .await,
10938            NextAction::Stop { .. }
10939        ));
10940
10941        {
10942            let state = lock_snapshot(&snapshot).unwrap();
10943            assert_eq!(state.state, ModuleState::Failed);
10944            assert_eq!(state.crash_restarts.len(), 2);
10945            assert_eq!(state.lifetime_restarts, 2);
10946        }
10947
10948        let history = runtime
10949            .terminal_ring
10950            .lock()
10951            .expect("terminal ring is not poisoned")
10952            .snapshot();
10953        let last = history
10954            .entries
10955            .last()
10956            .expect("the refused crash is retained");
10957        assert_eq!(last.disposition, TerminalDisposition::Failed);
10958        assert_eq!(
10959            last.disposition_detail.as_deref(),
10960            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10961        );
10962
10963        let captured = crate::router::test_log::captured_logs(&logs);
10964        assert!(
10965            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10966            "the stop must be logged with its window: {captured}"
10967        );
10968    }
10969
10970    /// The rate, stated as a test: three crashes where the first has aged past
10971    /// the window are two crashes as far as the budget is concerned, so the
10972    /// third respawn is allowed and the ring holds only the two recent ones.
10973    ///
10974    /// This is the case a lifetime counter got wrong -- and the case the daemon
10975    /// now hits routinely, since a module exits non-zero every time its
10976    /// connection to the daemon drops.
10977    #[tokio::test]
10978    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10979        let supervisor = Supervisor::new_for_test(
10980            Arc::new(Registry::default()),
10981            RestartPolicy::new(2, Duration::ZERO),
10982        );
10983        let runtime = supervisor.runtime_config();
10984        let spec = windowed_crash_spec("crash-across-windows");
10985        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10986
10987        for attempt in 1..=2 {
10988            assert!(matches!(
10989                on_child_exit(
10990                    &spec,
10991                    runtime.restart_policy,
10992                    &supervisor.registry,
10993                    &snapshot,
10994                    &runtime.terminal_ring,
10995                    &runtime.spawn_events,
10996                    &runtime.child_roster,
10997                    crash_exit_report(attempt),
10998                )
10999                .await,
11000                NextAction::Restart { schedule: _ }
11001            ));
11002        }
11003
11004        // The oldest crash moves out of the window; nothing else about the
11005        // module changes.
11006        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
11007            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
11008        })
11009        .unwrap();
11010
11011        assert!(
11012            matches!(
11013                on_child_exit(
11014                    &spec,
11015                    runtime.restart_policy,
11016                    &supervisor.registry,
11017                    &snapshot,
11018                    &runtime.terminal_ring,
11019                    &runtime.spawn_events,
11020                    &runtime.child_roster,
11021                    crash_exit_report(3),
11022                )
11023                .await,
11024                NextAction::Restart { schedule: _ }
11025            ),
11026            "a crash older than the window must not hold a budget slot"
11027        );
11028
11029        let state = lock_snapshot(&snapshot).unwrap();
11030        assert_eq!(state.state, ModuleState::Restarting);
11031        assert_eq!(
11032            state.crash_restarts.len(),
11033            2,
11034            "the aged instant is dropped and the new one takes its place"
11035        );
11036        assert_eq!(
11037            state.lifetime_restarts, 3,
11038            "the ledger counts every restart, including the ones the window forgot"
11039        );
11040    }
11041
11042    /// An operator restart hands the budget back whole, and the ledger keeps
11043    /// counting. Those are different questions -- "how close is this module to
11044    /// being stopped" and "how many times has it been replaced" -- and the
11045    /// operator action answers only the first.
11046    #[tokio::test]
11047    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
11048        let supervisor = Supervisor::new_for_test(
11049            Arc::new(Registry::default()),
11050            RestartPolicy::new(2, Duration::ZERO),
11051        );
11052        let runtime = supervisor.runtime_config();
11053        let spec = windowed_crash_spec("operator-cleared-budget");
11054        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11055
11056        for attempt in 1..=2 {
11057            assert!(matches!(
11058                on_child_exit(
11059                    &spec,
11060                    runtime.restart_policy,
11061                    &supervisor.registry,
11062                    &snapshot,
11063                    &runtime.terminal_ring,
11064                    &runtime.spawn_events,
11065                    &runtime.child_roster,
11066                    crash_exit_report(attempt),
11067                )
11068                .await,
11069                NextAction::Restart { schedule: _ }
11070            ));
11071        }
11072
11073        reset_restart_count(&snapshot, &spec.module_id).unwrap();
11074        {
11075            let state = lock_snapshot(&snapshot).unwrap();
11076            assert!(
11077                state.crash_restarts.is_empty(),
11078                "an operator restart returns the full budget"
11079            );
11080            assert_eq!(
11081                state.lifetime_restarts, 2,
11082                "clearing the budget must not unmake the crashes"
11083            );
11084        }
11085
11086        assert!(
11087            matches!(
11088                on_child_exit(
11089                    &spec,
11090                    runtime.restart_policy,
11091                    &supervisor.registry,
11092                    &snapshot,
11093                    &runtime.terminal_ring,
11094                    &runtime.spawn_events,
11095                    &runtime.child_roster,
11096                    crash_exit_report(3),
11097                )
11098                .await,
11099                NextAction::Restart { schedule: _ }
11100            ),
11101            "the cleared budget must be spendable again"
11102        );
11103        let state = lock_snapshot(&snapshot).unwrap();
11104        assert_eq!(state.crash_restarts.len(), 1);
11105        assert_eq!(state.lifetime_restarts, 3);
11106    }
11107
11108    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
11109    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
11110        let severed = ProcessIdentity {
11111            pid: 41,
11112            start_time: 101,
11113        };
11114        let successor = ProcessIdentity {
11115            pid: 41,
11116            start_time: 202,
11117        };
11118        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
11119        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
11120            state.pid = Some(successor.pid);
11121            state.process_start_time = Some(successor.start_time);
11122        })
11123        .unwrap();
11124        assert!(!module.record_deliberate_severance(severed).unwrap());
11125
11126        let exit_report = apply_deliberate_severance_marker(
11127            &module.inner.snapshot,
11128            Some(successor),
11129            ExitReport {
11130                kind: ExitKind::Crash,
11131                code: Some(1),
11132                signal: None,
11133                at_ms: 1,
11134            },
11135        );
11136
11137        assert_eq!(exit_report.kind, ExitKind::Crash);
11138    }
11139
11140    #[tokio::test]
11141    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
11142        let registry = Registry::default();
11143        let supervisor = Supervisor::new_for_test(
11144            Arc::new(Registry::default()),
11145            RestartPolicy::new(3, Duration::ZERO),
11146        );
11147        let runtime = supervisor.runtime_config();
11148        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11149        let spec = ModuleSpec {
11150            module_id: "drain-deliberate-severance".to_string(),
11151            program: fake_aft_stub_path(),
11152            args: Vec::new(),
11153            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11154            reserved: false,
11155            reserved_prefixes: Vec::new(),
11156            protocol: ModuleProtocol::Subc,
11157            overlap: Default::default(),
11158        };
11159        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11160        let process = ProcessIdentity {
11161            pid: 41,
11162            start_time: 101,
11163        };
11164        child.process_identity = Some(process);
11165        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
11166            state.pid = Some(process.pid);
11167            state.process_start_time = Some(process.start_time);
11168        })
11169        .unwrap();
11170        record_deliberate_severance(&snapshot, process).unwrap();
11171
11172        drain_child_to_state(
11173            &spec.module_id,
11174            spec.protocol,
11175            // The child exits on its own; no signal may change the exit this
11176            // test classifies.
11177            StopNotice::SentOverConnection,
11178            &registry,
11179            None,
11180            &snapshot,
11181            &runtime.terminal_ring,
11182            &runtime.spawn_events,
11183            child,
11184            Duration::from_secs(1),
11185            ModuleState::Stopped,
11186            Some(false),
11187        )
11188        .await
11189        .unwrap();
11190
11191        let state = lock_snapshot(&snapshot).unwrap();
11192        assert_eq!(
11193            state.last_exit.as_ref().map(|exit| exit.kind),
11194            Some(ExitKind::DeliberateSeverance)
11195        );
11196        assert_eq!(state.lifetime_restarts, 1);
11197        assert_eq!(state.crash_restarts.len(), 0);
11198        drop(state);
11199        let history = runtime.terminal_ring.lock().unwrap().snapshot();
11200        assert_eq!(
11201            history.entries[0].exit_kind,
11202            subc_control::TerminalExitKind::DeliberateSeverance
11203        );
11204    }
11205
11206    #[tokio::test]
11207    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
11208        let registry = Registry::default();
11209        let supervisor = Supervisor::new_for_test(
11210            Arc::new(Registry::default()),
11211            RestartPolicy::new(3, Duration::ZERO),
11212        );
11213        let runtime = supervisor.runtime_config();
11214        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11215        let spec = ModuleSpec {
11216            module_id: "ordinary-drain".to_string(),
11217            program: fake_aft_stub_path(),
11218            args: Vec::new(),
11219            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11220            reserved: false,
11221            reserved_prefixes: Vec::new(),
11222            protocol: ModuleProtocol::Subc,
11223            overlap: Default::default(),
11224        };
11225        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11226
11227        drain_child_to_state(
11228            &spec.module_id,
11229            spec.protocol,
11230            // The child exits on its own; no signal may change the exit this
11231            // test classifies.
11232            StopNotice::SentOverConnection,
11233            &registry,
11234            None,
11235            &snapshot,
11236            &runtime.terminal_ring,
11237            &runtime.spawn_events,
11238            child,
11239            Duration::from_secs(1),
11240            ModuleState::Stopped,
11241            Some(false),
11242        )
11243        .await
11244        .unwrap();
11245
11246        let state = lock_snapshot(&snapshot).unwrap();
11247        assert_eq!(
11248            state.last_exit.as_ref().map(|exit| exit.kind),
11249            Some(ExitKind::Crash)
11250        );
11251        assert_eq!(state.lifetime_restarts, 0);
11252        assert_eq!(state.crash_restarts.len(), 0);
11253    }
11254
11255    #[test]
11256    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
11257        // The server's generic fatal-routing branch only knows that the
11258        // connection failed; it does not know that the daemon deliberately
11259        // initiated a process-killing severance. Keep this seam explicit so a
11260        // future connection error path cannot silently reintroduce the stale
11261        // exemption that mislabels a later genuine crash.
11262        assert!(!include_str!("server.rs")
11263            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
11264    }
11265
11266    /// The `route.closed` `drained` value must be the quiescence wait's own
11267    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
11268    /// measurement at all and `false` is the one honest constant. This is the exact
11269    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
11270    /// on every return path, including the one that used to return early via `?`
11271    /// with `route.closing` already sent and no `route.closed` ever following.
11272    #[test]
11273    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
11274        assert!(drained_after_quiescence_wait(&Ok(true)));
11275        assert!(!drained_after_quiescence_wait(&Ok(false)));
11276        assert!(!drained_after_quiescence_wait(&Err(
11277            SuperviseError::StatePoisoned { module_id: None }
11278        )));
11279    }
11280
11281    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
11282    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
11283    /// already reaped out-of-band) still leaves a terminal record rather than none
11284    /// at all. Triggering the real `wait()` I/O error from an integration test would
11285    /// need a genuine already-reaped-child race, which is OS-specific and not
11286    /// something this suite attempts elsewhere; this test instead verifies the
11287    /// record produced for that arm end-to-end through the real `TerminalRing`, and
11288    /// the call site itself is verified by inspection to sit in that exact arm.
11289    #[test]
11290    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
11291        let ring = Arc::new(Mutex::new(TerminalRing::new(
11292            TerminalRingConfig::default(),
11293            0,
11294        )));
11295        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
11296
11297        let snapshot = ring.lock().unwrap().snapshot();
11298        assert_eq!(snapshot.entries.len(), 1);
11299        let entry = &snapshot.entries[0];
11300        assert_eq!(entry.exit_code, None);
11301        assert_eq!(entry.exit_signal, None);
11302        assert_eq!(entry.disposition, TerminalDisposition::Failed);
11303    }
11304
11305    #[test]
11306    fn wait_error_exit_path_preserves_spawn_event_density() {
11307        let feed = super::SpawnEventFeed::default();
11308        feed.configure_incarnation("wait-error-density".to_string());
11309        feed.emit_spawned("wait-error", 41, 1);
11310        let ring = Arc::new(Mutex::new(TerminalRing::new(
11311            TerminalRingConfig::default(),
11312            0,
11313        )));
11314
11315        record_wait_error_terminal("wait-error", &ring, &feed);
11316        feed.emit_spawned("after-wait-error", 42, 2);
11317
11318        let state = feed.0.lock().unwrap();
11319        let sequences = state
11320            .events
11321            .iter()
11322            .map(|event| event.cursor.seq)
11323            .collect::<Vec<_>>();
11324        assert_eq!(sequences, vec![1, 2, 3]);
11325        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11326        assert_eq!(state.events[1].exit_code, None);
11327        assert_eq!(state.events[1].exit_signal, None);
11328    }
11329
11330    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
11331    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
11332    /// not a clean exit it never actually observed.
11333    #[test]
11334    fn wait_error_exit_report_is_classified_as_a_crash() {
11335        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11336    }
11337}
11338
11339#[cfg(all(test, unix))]
11340mod health_event_tests {
11341    use super::*;
11342    use subc_protocol::{manifest::Concurrency, session::ModuleControlResponse};
11343    use subc_test_support::TestTempDir;
11344
11345    struct Actor {
11346        module: SupervisedModule,
11347        registry: Arc<Registry>,
11348        forwarding: Arc<ForwardingTable>,
11349        spec: ModuleSpec,
11350        health: HealthConfig,
11351        rx: mpsc::Receiver<crate::router::OutboundFrame>,
11352        _home: TestTempDir,
11353    }
11354
11355    async fn wait_for<T>(reason: &str, mut observe: impl FnMut() -> Option<T>) -> T {
11356        // OS exec and socket readiness are real I/O. Stay runnable while waiting
11357        // for an observed condition, rather than letting paused time auto-advance
11358        // deadlines or assuming a fixed number of yields finishes the work.
11359        let deadline = std::time::Instant::now() + Duration::from_secs(10);
11360        let tick = Instant::now();
11361        loop {
11362            if let Some(value) = observe() {
11363                assert_eq!(Instant::now(), tick, "{reason} must not wait for a timer");
11364                return value;
11365            }
11366            assert!(
11367                std::time::Instant::now() < deadline,
11368                "timed out waiting for {reason}"
11369            );
11370            tokio::task::yield_now().await;
11371        }
11372    }
11373
11374    impl Actor {
11375        async fn start(protocol: ModuleProtocol, cadence: Duration) -> Self {
11376            let home = TestTempDir::new("health-events");
11377            let registry = Arc::new(Registry::default());
11378            let forwarding = Arc::new(ForwardingTable::default());
11379            let health = HealthConfig {
11380                cadence,
11381                deadline: Duration::from_secs(3600),
11382                ..HealthConfig::default()
11383            };
11384            let supervisor =
11385                Supervisor::new_for_test(registry.clone(), RestartPolicy::new(3, Duration::ZERO))
11386                    .with_forwarding(forwarding.clone())
11387                    .with_health_config(health.clone());
11388            let spec = ModuleSpec {
11389                module_id: "event-health-child".into(),
11390                program: PathBuf::from("/bin/sh"),
11391                // The process stays alive but has no real bus peer. Registrations
11392                // below exercise the same registry writes as accepted HELLOs.
11393                args: vec!["-c".into(), "exec sleep 600".into()],
11394                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11395                    .into_iter()
11396                    .map(|key| (key.into(), home.path().display().to_string()))
11397                    .collect(),
11398                reserved: false,
11399                reserved_prefixes: vec![],
11400                protocol,
11401                overlap: Default::default(),
11402            };
11403            let module = supervisor.spawn(spec.clone()).unwrap();
11404            let (_, rx) = mpsc::channel(8);
11405            let actor = Self {
11406                module,
11407                registry,
11408                forwarding,
11409                spec,
11410                health,
11411                rx,
11412                _home: home,
11413            };
11414            actor
11415                .wait_for_select(|checkpoint| checkpoint.generation > 0)
11416                .await;
11417            actor
11418        }
11419
11420        fn turns(&self) -> u64 {
11421            lock_snapshot(&self.module.inner.snapshot)
11422                .unwrap()
11423                .actor_turns
11424        }
11425
11426        async fn wait_for_select(
11427            &self,
11428            expected: impl Fn(&ActorSelectCheckpoint) -> bool,
11429        ) -> ActorSelectCheckpoint {
11430            wait_for("actor select checkpoint", || {
11431                let state = lock_snapshot(&self.module.inner.snapshot).unwrap();
11432                state.actor_select.filter(|checkpoint| {
11433                    checkpoint.generation == state.spawn_generation
11434                        && state.state == ModuleState::Running
11435                        && state.process_alive
11436                        && state.reported_pid().is_some()
11437                        && expected(checkpoint)
11438                })
11439            })
11440            .await
11441        }
11442
11443        fn hello(&mut self, connection: u64, health: bool) {
11444            let manifest =
11445                subc_protocol::manifest::ModuleManifest::builder(&self.spec.module_id, "1.0.0")
11446                    .protocol_ver(subc_protocol::PROTOCOL_VERSION)
11447                    .build();
11448            let connection = ConnectionId::new(connection);
11449            let (tx, rx) = mpsc::channel(8);
11450            self.rx = rx;
11451            self.forwarding
11452                .register_module_connection(
11453                    connection,
11454                    self.spec.module_id.clone(),
11455                    subc_protocol::PROTOCOL_VERSION,
11456                    Concurrency::ModuleManaged,
11457                    FrameSink::new(tx),
11458                )
11459                .unwrap();
11460            self.registry
11461                .register_with_control_ops(
11462                    manifest,
11463                    subc_protocol::PROTOCOL_VERSION,
11464                    connection,
11465                    if health {
11466                        vec![MODULE_CONTROL_OP_HEALTH_CHECK.into()]
11467                    } else {
11468                        vec![]
11469                    },
11470                )
11471                .unwrap();
11472        }
11473
11474        async fn probe(&mut self) -> crate::router::OutboundFrame {
11475            let frame = wait_for("outbound health probe", || match self.rx.try_recv() {
11476                Ok(frame) => Some(frame),
11477                Err(mpsc::error::TryRecvError::Empty) => None,
11478                Err(mpsc::error::TryRecvError::Disconnected) => panic!("health peer disconnected"),
11479            })
11480            .await;
11481            let body: Value = serde_json::from_slice(&frame.body).unwrap();
11482            assert_eq!(body["op"], MODULE_CONTROL_OP_HEALTH_CHECK);
11483            frame
11484        }
11485
11486        fn no_probe(&mut self) {
11487            assert!(matches!(
11488                self.rx.try_recv(),
11489                Err(mpsc::error::TryRecvError::Empty)
11490            ));
11491        }
11492
11493        fn answer(&self, connection: u64, frame: crate::router::OutboundFrame) {
11494            self.forwarding
11495                .complete_module_control_rpc(
11496                    ConnectionId::new(connection),
11497                    frame.header.corr,
11498                    Some(MODULE_CONTROL_OP_HEALTH_CHECK),
11499                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11500                        status: HealthStatus::Ok,
11501                        detail: None,
11502                        metrics: None,
11503                    }),
11504                )
11505                .unwrap();
11506        }
11507    }
11508
11509    #[tokio::test(start_paused = true)]
11510    async fn no_health_children_park_for_sixty_seconds() {
11511        let none = Actor::start(ModuleProtocol::None, Duration::from_secs(30)).await;
11512        let unregistered = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11513        let mut unadvertised = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11514        unadvertised.hello(1, false);
11515        // Include any one-off registration wake in the budget; receiving a
11516        // no-health HELLO must not turn a parked child into a polling child.
11517        // Advance in 10 ms steps so the old poll really executes ~6,000 turns;
11518        // a single 60 s jump would only observe one expired timer.
11519        for _ in 0..6000 {
11520            tokio::time::advance(Duration::from_millis(10)).await;
11521            tokio::task::yield_now().await;
11522        }
11523        for actor in [&none, &unregistered, &unadvertised] {
11524            assert!(
11525                actor.turns() <= 5,
11526                "no-health actor woke {} times",
11527                actor.turns()
11528            );
11529            assert_eq!(actor.module.state().unwrap(), ModuleState::Running);
11530        }
11531    }
11532
11533    #[tokio::test(start_paused = true)]
11534    async fn late_hello_starts_a_due_probe_in_the_notification_tick() {
11535        // With a zero cadence the first probe is due at once, so this measures
11536        // only how fast the registration notification starts health probing,
11537        // not the normal 30-33 s wait before a first probe (checked below).
11538        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::ZERO).await;
11539        let tick = Instant::now();
11540        actor.hello(1, true);
11541        actor.probe().await;
11542        assert_eq!(Instant::now(), tick, "HELLO must not wait for a poll");
11543    }
11544
11545    #[tokio::test(start_paused = true)]
11546    async fn hello_keeps_the_jittered_cadence_and_catalog_updates_do_not_reset_it() {
11547        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11548        let hello_tick = Instant::now();
11549        actor.hello(1, true);
11550        let armed = actor
11551            .wait_for_select(|checkpoint| {
11552                checkpoint.registered_connection == Some(ConnectionId::new(1))
11553                    && checkpoint.next_probe_at.is_some()
11554            })
11555            .await;
11556        let delay = jittered_health_delay(&actor.spec.module_id, 0, actor.health.cadence);
11557        assert!((Duration::from_secs(30)..Duration::from_secs(33)).contains(&delay));
11558        let first_deadline = hello_tick + delay;
11559        assert_eq!(armed.next_probe_at, Some(first_deadline));
11560        tokio::time::advance(Duration::from_secs(29)).await;
11561        let turns = actor.turns();
11562        actor
11563            .registry
11564            .replace_catalog_for_connection(ConnectionId::new(1), vec![], None, Some(true))
11565            .unwrap();
11566        let updated = actor
11567            .wait_for_select(|checkpoint| checkpoint.turn > turns)
11568            .await;
11569        assert_eq!(
11570            updated.next_probe_at,
11571            Some(first_deadline),
11572            "catalog update must not reset cadence"
11573        );
11574        actor.no_probe();
11575        tokio::time::advance(first_deadline - Instant::now()).await;
11576        let frame = actor.probe().await;
11577        actor.answer(1, frame);
11578        let next = jittered_health_delay(&actor.spec.module_id, 1, actor.health.cadence);
11579        let next_deadline = Instant::now() + next;
11580        let rearmed = actor
11581            .wait_for_select(|checkpoint| {
11582                checkpoint
11583                    .next_probe_at
11584                    .is_some_and(|deadline| deadline > first_deadline)
11585            })
11586            .await;
11587        assert_eq!(rearmed.next_probe_at, Some(next_deadline));
11588        tokio::time::advance(next - Duration::from_nanos(1)).await;
11589        actor.no_probe();
11590        tokio::time::advance(Duration::from_nanos(1)).await;
11591        actor.probe().await;
11592    }
11593
11594    #[tokio::test(start_paused = true)]
11595    async fn disconnect_disarms_and_reconnect_rearms_health() {
11596        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11597        actor.hello(1, true);
11598        actor
11599            .wait_for_select(|checkpoint| {
11600                checkpoint.registered_connection == Some(ConnectionId::new(1))
11601                    && checkpoint.next_probe_at.is_some()
11602            })
11603            .await;
11604        let turns = actor.turns();
11605        let tick = Instant::now();
11606        actor
11607            .registry
11608            .deregister_connection(ConnectionId::new(1))
11609            .unwrap();
11610        let disarmed = actor
11611            .wait_for_select(|checkpoint| {
11612                checkpoint.turn > turns && checkpoint.registered_connection.is_none()
11613            })
11614            .await;
11615        assert_eq!(disarmed.next_probe_at, None);
11616        assert_eq!(disarmed.wake_after, None);
11617        assert_eq!(
11618            Instant::now(),
11619            tick,
11620            "disconnect must disarm in the notification tick"
11621        );
11622        tokio::time::advance(Duration::from_secs(60)).await;
11623        actor.no_probe();
11624        let reconnect_tick = Instant::now();
11625        actor.hello(2, true);
11626        let rearmed = actor
11627            .wait_for_select(|checkpoint| {
11628                checkpoint.registered_connection == Some(ConnectionId::new(2))
11629                    && checkpoint.next_probe_at.is_some()
11630            })
11631            .await;
11632        let delay = jittered_health_delay(&actor.spec.module_id, 0, actor.health.cadence);
11633        assert_eq!(rearmed.next_probe_at, Some(reconnect_tick + delay));
11634        tokio::time::advance(delay).await;
11635        actor.probe().await;
11636    }
11637
11638    #[tokio::test(start_paused = true)]
11639    async fn reregistration_without_health_disarms_the_old_schedule() {
11640        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11641        actor.hello(1, true);
11642        actor
11643            .wait_for_select(|checkpoint| {
11644                checkpoint.registered_connection == Some(ConnectionId::new(1))
11645                    && checkpoint.next_probe_at.is_some()
11646            })
11647            .await;
11648        actor
11649            .registry
11650            .deregister_connection(ConnectionId::new(1))
11651            .unwrap();
11652        actor.hello(2, false);
11653        let disarmed = actor
11654            .wait_for_select(|checkpoint| {
11655                checkpoint.registered_connection == Some(ConnectionId::new(2))
11656            })
11657            .await;
11658        assert_eq!(disarmed.next_probe_at, None);
11659        assert_eq!(disarmed.wake_after, None);
11660        tokio::time::advance(Duration::from_secs(60)).await;
11661        actor.no_probe();
11662    }
11663
11664    #[tokio::test(start_paused = true)]
11665    async fn restart_waits_for_its_own_hello_before_rearming_health() {
11666        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11667        actor.hello(1, true);
11668        let armed = actor
11669            .wait_for_select(|checkpoint| {
11670                checkpoint.registered_connection == Some(ConnectionId::new(1))
11671                    && checkpoint.next_probe_at.is_some()
11672            })
11673            .await;
11674        // Simulate the old process's connection closing before the restart
11675        // tears the process down. The replacement process has started but has
11676        // not sent its HELLO, so it is not registered.
11677        actor
11678            .registry
11679            .deregister_connection(ConnectionId::new(1))
11680            .unwrap();
11681        actor.module.restart(Some(0)).await.unwrap();
11682        let parked = actor
11683            .wait_for_select(|checkpoint| {
11684                checkpoint.generation > armed.generation
11685                    && checkpoint.registered_connection.is_none()
11686                    && checkpoint.next_probe_at.is_none()
11687            })
11688            .await;
11689        assert_eq!(parked.wake_after, None);
11690        actor.rx.close();
11691        // Drain the stop notice the restart queued for the old process's connection,
11692        // so it isn't mistaken for traffic to the replacement.
11693        while actor.rx.try_recv().is_ok() {}
11694        let turns = actor.turns();
11695        tokio::time::advance(Duration::from_secs(60)).await;
11696        let after = actor.turns();
11697        eprintln!(
11698            "restart park turns: before={turns} after={after} delta={}",
11699            after - turns
11700        );
11701        assert_eq!(after, turns, "replacement must park before HELLO");
11702        let hello_tick = Instant::now();
11703        actor.hello(2, true);
11704        let rearmed = actor
11705            .wait_for_select(|checkpoint| {
11706                checkpoint.registered_connection == Some(ConnectionId::new(2))
11707                    && checkpoint.next_probe_at.is_some()
11708            })
11709            .await;
11710        let delay = jittered_health_delay(&actor.spec.module_id, 0, actor.health.cadence);
11711        assert_eq!(rearmed.next_probe_at, Some(hello_tick + delay));
11712        tokio::time::advance(delay).await;
11713        actor.probe().await;
11714    }
11715
11716    #[tokio::test(start_paused = true)]
11717    async fn rescan_adds_and_removes_http_health_without_polling() {
11718        use tokio::{
11719            io::{AsyncReadExt, AsyncWriteExt},
11720            net::TcpListener,
11721        };
11722        let mut actor = Actor::start(ModuleProtocol::None, Duration::from_secs(30)).await;
11723        let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
11724        actor.health.http = Some(format!("http://{}/health", listener.local_addr().unwrap()));
11725        // One millisecond separates successive probes without a polling delay.
11726        actor.health.cadence = Duration::from_millis(1);
11727        let (started_tx, mut started_rx) = mpsc::channel(8);
11728        let peer = tokio::spawn(async move {
11729            loop {
11730                let (mut socket, _) = listener.accept().await.unwrap();
11731                let mut request = [0; 1024];
11732                assert!(socket.read(&mut request).await.unwrap() > 0);
11733                started_tx.send(()).await.unwrap();
11734                socket
11735                    .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n")
11736                    .await
11737                    .unwrap();
11738            }
11739        });
11740        let tick = Instant::now();
11741        let turns = actor.turns();
11742        actor
11743            .module
11744            .update_configuration(actor.spec.clone(), actor.health.clone(), None)
11745            .await
11746            .unwrap();
11747        let armed = actor
11748            .wait_for_select(|checkpoint| {
11749                checkpoint.turn > turns && checkpoint.next_probe_at.is_some()
11750            })
11751            .await;
11752        assert_eq!(armed.next_probe_at, Some(tick + Duration::from_millis(1)));
11753        assert_eq!(Instant::now(), tick);
11754        tokio::time::advance(Duration::from_millis(1)).await;
11755        let mut started = false;
11756        wait_for("successful HTTP probe", || {
11757            started |= started_rx.try_recv().is_ok();
11758            (started && actor.module.status().unwrap().health.status == SupervisorHealthStatus::Ok)
11759                .then_some(())
11760        })
11761        .await;
11762        assert!(started, "added HTTP check must start at its first deadline");
11763        assert_eq!(
11764            actor.module.status().unwrap().health.status,
11765            SupervisorHealthStatus::Ok
11766        );
11767        assert_eq!(Instant::now(), tick + Duration::from_millis(1));
11768        actor.health.http = None;
11769        let turns = actor.turns();
11770        actor
11771            .module
11772            .update_configuration(actor.spec.clone(), actor.health.clone(), None)
11773            .await
11774            .unwrap();
11775        let parked = actor
11776            .wait_for_select(|checkpoint| {
11777                checkpoint.turn > turns && checkpoint.next_probe_at.is_none()
11778            })
11779            .await;
11780        assert_eq!(parked.wake_after, None);
11781        let turns = actor.turns();
11782        tokio::time::advance(Duration::from_secs(60)).await;
11783        assert_eq!(
11784            actor.turns(),
11785            turns,
11786            "removed HTTP check must park the actor"
11787        );
11788        assert!(
11789            started_rx.try_recv().is_err(),
11790            "removed HTTP check must stay disarmed"
11791        );
11792        assert_eq!(
11793            actor.module.status().unwrap().health,
11794            ModuleHealthStatus::default()
11795        );
11796        peer.abort();
11797    }
11798}
11799
11800#[cfg(test)]
11801mod health_evidence_tests {
11802    use super::{HealthProbeError, HealthProbeEvidence};
11803    use std::collections::HashSet;
11804
11805    /// The evidential asymmetry, asserted rather than described.
11806    ///
11807    /// Exactly ONE observation is proof a module cannot serve, and the one that
11808    /// fires under CPU starvation is not it. Before the split, all fifteen
11809    /// construction sites collapsed into a single String, so a timeout carried the
11810    /// same weight as a dead lane -- which is how a healthy module was restarted
11811    /// three times in one day.
11812    #[test]
11813    fn only_a_dead_lane_is_proof_of_death() {
11814        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11815        // Three non-proof classes, each for a different reason: silence is
11816        // consistent with health, a bad answer proves the module ALIVE, and a
11817        // daemon-side fault never reached the module at all.
11818        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11819        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11820        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11821    }
11822
11823    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
11824    ///
11825    /// A shared label renders two different observations identically in the line an
11826    /// operator reads after an unexplained restart -- the exact confusion this
11827    /// change removes.
11828    #[test]
11829    fn every_evidence_class_has_a_distinct_label() {
11830        let labels = [
11831            HealthProbeError::lane_dead("").label(),
11832            HealthProbeError::no_answer("").label(),
11833            HealthProbeError::bad_answer("").label(),
11834            HealthProbeError::misconfigured("").label(),
11835        ];
11836        let unique: HashSet<_> = labels.iter().collect();
11837        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11838    }
11839
11840    /// The class is additional information, not a replacement.
11841    ///
11842    /// An operator needs both "this was silence" and the specific text saying how
11843    /// long we waited; a classification that swallowed the message would trade one
11844    /// missing distinction for another.
11845    #[test]
11846    fn classification_preserves_the_original_message() {
11847        let err = HealthProbeError::no_answer("module did not answer within 5s");
11848        assert_eq!(err.to_string(), "module did not answer within 5s");
11849        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11850    }
11851}
11852
11853#[cfg(test)]
11854mod health_tombstone_tests {
11855    use std::{path::PathBuf, sync::Arc, time::Duration};
11856
11857    use subc_protocol::{
11858        manifest::Concurrency,
11859        session::{HealthStatus, ModuleControlResponse},
11860    };
11861    use tokio::sync::mpsc;
11862
11863    use super::{
11864        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11865        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11866    };
11867    use crate::{
11868        control::ControlHandler,
11869        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11870        registry::{ConnectionId, Registry},
11871        router::FrameSink,
11872    };
11873
11874    struct ProbeHarness {
11875        spec: ModuleSpec,
11876        runtime: SupervisorRuntimeConfig,
11877        forwarding: Arc<ForwardingTable>,
11878        module_connection: ConnectionId,
11879        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11880        handler: ControlHandler,
11881        module: super::SupervisedModule,
11882    }
11883
11884    fn probe_harness() -> ProbeHarness {
11885        let registry = Arc::new(Registry::default());
11886        let forwarding = Arc::new(ForwardingTable::default());
11887        let supervisor_handle = super::SupervisorHandle::new();
11888        let health = HealthConfig {
11889            http: None,
11890            cadence: Duration::from_secs(30),
11891            deadline: Duration::from_secs(5),
11892            failure_threshold: 3,
11893            on_degraded: HealthAction::Report,
11894            on_failing: HealthAction::Report,
11895            critical: false,
11896        };
11897        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11898            .with_forwarding(Arc::clone(&forwarding))
11899            .with_handle(supervisor_handle.clone())
11900            .with_health_config(health);
11901        let spec = ModuleSpec {
11902            module_id: "late-health-module".to_string(),
11903            program: PathBuf::from("disabled-module"),
11904            args: Vec::new(),
11905            env: Vec::new(),
11906            reserved: false,
11907            reserved_prefixes: Vec::new(),
11908            protocol: ModuleProtocol::Subc,
11909            overlap: Default::default(),
11910        };
11911        let module = supervisor
11912            .supervise_configured(spec.clone(), false)
11913            .unwrap();
11914        let runtime = supervisor.runtime_config();
11915        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11916            .with_supervisor(supervisor_handle);
11917        let module_connection = ConnectionId::new(700);
11918        let (module_tx, module_rx) = mpsc::channel(8);
11919        forwarding
11920            .register_module_connection(
11921                module_connection,
11922                spec.module_id.clone(),
11923                subc_protocol::PROTOCOL_VERSION,
11924                Concurrency::ModuleManaged,
11925                FrameSink::new(module_tx),
11926            )
11927            .unwrap();
11928
11929        ProbeHarness {
11930            spec,
11931            runtime,
11932            forwarding,
11933            module_connection,
11934            module_rx,
11935            handler,
11936            module,
11937        }
11938    }
11939
11940    async fn finish_after(
11941        harness: &mut ProbeHarness,
11942        stall: Duration,
11943    ) -> ModuleControlRpcCompletion {
11944        assert!(stall > harness.runtime.health.deadline);
11945        let deadline = harness.runtime.health.deadline;
11946        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11947        let answer = async {
11948            let frame = harness.module_rx.recv().await.expect("health.check frame");
11949            tokio::time::advance(deadline).await;
11950            tokio::task::yield_now().await;
11951            tokio::time::advance(stall - deadline).await;
11952            harness
11953                .forwarding
11954                .complete_module_control_rpc(
11955                    harness.module_connection,
11956                    frame.header.corr,
11957                    Some("health.check"),
11958                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11959                        status: HealthStatus::Ok,
11960                        detail: None,
11961                        metrics: None,
11962                    }),
11963                )
11964                .unwrap()
11965        };
11966        let (probe_result, completion) = tokio::join!(probe, answer);
11967        let err = probe_result.expect_err("probe must miss its deadline");
11968        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11969        completion
11970    }
11971
11972    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11973        let deadline = harness.runtime.health.deadline;
11974        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11975        let exhaust_deadline = async {
11976            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11977            tokio::time::advance(deadline).await;
11978            tokio::task::yield_now().await;
11979        };
11980        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11981        let err = probe_result.expect_err("probe must miss its deadline");
11982        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11983    }
11984
11985    async fn run_probe_cycle(harness: &mut ProbeHarness, answer: bool) {
11986        let registry = Arc::clone(&harness.module.inner.registry);
11987        let snapshot = Arc::clone(&harness.module.inner.snapshot);
11988        let process_liveness = super::SupervisorProcessLiveness::default();
11989        let mut child = None;
11990        let cycle = super::run_health_probe_cycle(
11991            &harness.spec,
11992            &harness.runtime,
11993            &registry,
11994            &process_liveness,
11995            &snapshot,
11996            &mut child,
11997        );
11998        let peer = async {
11999            let frame = harness.module_rx.recv().await.expect("health.check frame");
12000            if answer {
12001                harness
12002                    .forwarding
12003                    .complete_module_control_rpc(
12004                        harness.module_connection,
12005                        frame.header.corr,
12006                        Some("health.check"),
12007                        ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
12008                            status: HealthStatus::Ok,
12009                            detail: None,
12010                            metrics: Some(serde_json::json!({"ready": true})),
12011                        }),
12012                    )
12013                    .unwrap();
12014            } else {
12015                tokio::time::advance(harness.runtime.health.deadline).await;
12016                tokio::task::yield_now().await;
12017            }
12018        };
12019        tokio::join!(cycle, peer);
12020    }
12021
12022    #[tokio::test(start_paused = true)]
12023    async fn unanswered_probe_is_unknown_until_threshold_and_ok_report_recovers() {
12024        let mut harness = probe_harness();
12025        // Drive the probe cycle directly with an in-memory wire peer. Stop the
12026        // disabled module's monitor so only this test owns lifecycle transitions;
12027        // no OS process is launched, and a restart is observed at scheduling.
12028        let monitor = harness.module.inner.monitor.lock().unwrap().take().unwrap();
12029        monitor.abort();
12030        let _ = monitor.await;
12031        super::update_snapshot(&harness.module.inner.snapshot, None, |state| {
12032            state.enabled = true;
12033            state.state = super::ModuleState::Running;
12034            state.process_alive = true;
12035        })
12036        .unwrap();
12037
12038        run_probe_cycle(&mut harness, true).await;
12039        assert_eq!(
12040            harness.module.status().unwrap().health.status,
12041            super::SupervisorHealthStatus::Ok
12042        );
12043
12044        for failures in 1..harness.runtime.health.failure_threshold {
12045            run_probe_cycle(&mut harness, false).await;
12046            let status = harness.module.status().unwrap();
12047            assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
12048            assert_eq!(status.health.consecutive_failures, failures);
12049            assert!(status.health.last_probe_ms.is_some());
12050            assert!(status.health.detail.unwrap().starts_with("[no-answer]"));
12051            assert!(status.health.metrics.is_none());
12052            assert_eq!(status.state, super::ModuleState::Running);
12053            assert!(status.process_alive);
12054            assert_eq!(status.restart_count, 0);
12055            assert_eq!(status.lifetime_restarts, 0);
12056            assert!(status.health.last_action.is_none());
12057            assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
12058        }
12059
12060        run_probe_cycle(&mut harness, true).await;
12061        let recovered = harness.module.status().unwrap();
12062        assert_eq!(recovered.health.status, super::SupervisorHealthStatus::Ok);
12063        assert_eq!(recovered.health.consecutive_failures, 0);
12064        assert!(recovered.health.detail.is_none());
12065        assert_eq!(
12066            recovered.health.metrics,
12067            Some(serde_json::json!({"ready": true}))
12068        );
12069        assert_eq!(recovered.lifetime_restarts, 0);
12070
12071        for failures in 1..=harness.runtime.health.failure_threshold {
12072            run_probe_cycle(&mut harness, false).await;
12073            let status = harness.module.status().unwrap();
12074            assert_eq!(status.health.consecutive_failures, failures);
12075            if failures < harness.runtime.health.failure_threshold {
12076                assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
12077                assert_eq!(status.state, super::ModuleState::Running);
12078                assert_eq!(status.lifetime_restarts, 0);
12079                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
12080            } else {
12081                assert_eq!(
12082                    status.health.status,
12083                    super::SupervisorHealthStatus::Unresponsive
12084                );
12085                assert_eq!(status.state, super::ModuleState::Restarting);
12086                assert_eq!(status.restart_count, 1);
12087                assert_eq!(status.lifetime_restarts, 1);
12088                assert_eq!(status.health.last_action.as_deref(), Some("restart"));
12089                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_some());
12090            }
12091        }
12092    }
12093
12094    #[tokio::test(start_paused = true)]
12095    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
12096        let mut harness = probe_harness();
12097
12098        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
12099        let first_latency = match &first {
12100            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
12101            other => panic!("late answer was not retained: {other:?}"),
12102        };
12103        assert!(harness.handler.observe_module_control_completion(first));
12104
12105        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
12106        let second_latency = match &second {
12107            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
12108            other => panic!("late answer was not retained: {other:?}"),
12109        };
12110        assert!(harness.handler.observe_module_control_completion(second));
12111
12112        assert_eq!(first_latency, Duration::from_secs(8));
12113        assert_eq!(
12114            second_latency - first_latency,
12115            Duration::from_secs(3),
12116            "latency must grow linearly with the additional stall"
12117        );
12118        let health = harness.module.status().unwrap().health;
12119        assert_eq!(health.late_answer_count, 2);
12120        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
12121    }
12122
12123    /// A module that answers every probe late must never march to the kill
12124    /// threshold: the late answer proves it is alive, so it must clear the miss
12125    /// streak the timeout recorded. Without the reset, a CPU-starved module
12126    /// that serves every probe seconds past the deadline accumulates
12127    /// `consecutive_failures` to the threshold and is killed — the exact
12128    /// sequence from the 2026-08-14 aft disable, where the daemon logged
12129    /// "proves the module is alive" five times while counting five misses.
12130    #[tokio::test(start_paused = true)]
12131    async fn late_answer_clears_the_consecutive_failure_streak() {
12132        let mut harness = probe_harness();
12133
12134        // Timeout recorded first: the probe path saw no answer in time.
12135        time_out_without_answer(&mut harness).await;
12136        harness
12137            .module
12138            .record_health_probe_failure_for_test("[no-answer] test miss")
12139            .unwrap();
12140        assert_eq!(
12141            harness.module.status().unwrap().health.consecutive_failures,
12142            1,
12143            "precondition: the miss must be on the streak before the late answer"
12144        );
12145
12146        // The stalled reply then lands: proof of life.
12147        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
12148        assert!(matches!(
12149            late,
12150            ModuleControlRpcCompletion::LateHealthAnswer { .. }
12151        ));
12152        assert!(harness.handler.observe_module_control_completion(late));
12153
12154        let health = harness.module.status().unwrap().health;
12155        assert_eq!(
12156            health.consecutive_failures, 0,
12157            "a late answer is an answer: the streak must reset"
12158        );
12159        assert_eq!(health.late_answer_count, 1);
12160    }
12161
12162    #[tokio::test(start_paused = true)]
12163    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
12164        let mut harness = probe_harness();
12165
12166        for _ in 0..20 {
12167            time_out_without_answer(&mut harness).await;
12168            assert_eq!(
12169                harness.forwarding.health_probe_tombstone_count().unwrap(),
12170                1
12171            );
12172        }
12173    }
12174}
12175
12176#[cfg(test)]
12177mod child_env_tests {
12178    use super::{
12179        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
12180        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
12181        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
12182    };
12183    use std::{ffi::OsStr, path::PathBuf};
12184    use tokio::process::Command;
12185
12186    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
12187        ModuleSpec {
12188            module_id: "env-plan".to_string(),
12189            program: PathBuf::from("/nonexistent"),
12190            args: Vec::new(),
12191            env,
12192            reserved: false,
12193            reserved_prefixes: Vec::new(),
12194            protocol: ModuleProtocol::Subc,
12195            overlap: Default::default(),
12196        }
12197    }
12198
12199    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
12200    /// one still gets its own.
12201    ///
12202    /// This is the narrow goal `env_clear()` was reached for, and the reason the
12203    /// fix is `env_remove` rather than deleting the line: an operator's ambient
12204    /// filter silently becoming an unconfigured module's log level is a real
12205    /// defect, just a much smaller one than clearing the environment.
12206    ///
12207    /// Asserted on the command plan rather than a spawned child because proving
12208    /// the ABSENCE of an inherited variable needs the parent's environment
12209    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
12210    /// removal as `(key, None)`, which is exactly the distinction wanted: not
12211    /// "absent because nobody set it" but "explicitly unset for the child".
12212    #[test]
12213    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
12214        let mut command = Command::new("/nonexistent");
12215        apply_child_env(&mut command, &spec(Vec::new()));
12216        let removed = command
12217            .as_std()
12218            .get_envs()
12219            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
12220        assert!(
12221            removed,
12222            "ambient CK_LOG must be explicitly removed for an unconfigured module"
12223        );
12224
12225        let mut configured = Command::new("/nonexistent");
12226        apply_child_env(
12227            &mut configured,
12228            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
12229        );
12230        let effective = configured
12231            .as_std()
12232            .get_envs()
12233            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
12234            .last()
12235            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
12236        assert_eq!(
12237            effective,
12238            Some(Some("debug".to_string())),
12239            "a module's configured CK_LOG must survive the ambient removal"
12240        );
12241    }
12242
12243    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
12244    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
12245    /// the same reason as the CK_LOG test above.
12246    ///
12247    /// The argument is the load-bearing half: a stock binary exits on an
12248    /// unknown flag before it listens, so with `--subc` appended the mode
12249    /// could not supervise the one process it exists for. Found by the first
12250    /// conformance run (nats-server: `flag provided but not defined: -subc`).
12251    #[test]
12252    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
12253        let connection_file = std::path::Path::new("/run/subc-connection.json");
12254        let handle = SupervisorHandle::new();
12255
12256        let mut none_spec = spec(Vec::new());
12257        none_spec.protocol = ModuleProtocol::None;
12258        let mut none = Command::new("/nonexistent");
12259        let none_handoff =
12260            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
12261                .expect("protocol-none spawn args apply");
12262        assert!(
12263            none_handoff.is_none(),
12264            "protocol:none spawn must not receive a nonce descriptor"
12265        );
12266        assert!(
12267            !none.as_std().get_envs().any(|(key, value)| key
12268                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
12269                && value.is_some()),
12270            "protocol:none spawn must not name a nonce descriptor"
12271        );
12272        let none_args: Vec<String> = none
12273            .as_std()
12274            .get_args()
12275            .map(|a| a.to_string_lossy().into_owned())
12276            .collect();
12277        assert!(
12278            !none_args.iter().any(|a| a == SUBC_ARG),
12279            "protocol:none argv must not carry --subc; got {none_args:?}"
12280        );
12281        let none_has_nonce = none
12282            .as_std()
12283            .get_envs()
12284            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
12285        assert!(
12286            !none_has_nonce,
12287            "protocol:none spawn must not receive a launch nonce"
12288        );
12289        let none_has_module_id = none
12290            .as_std()
12291            .get_envs()
12292            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
12293        assert!(
12294            none_has_module_id,
12295            "SUBC_MODULE_ID is inert and stays on every path"
12296        );
12297        assert!(
12298            handle.spawn_nonce(&none_spec.module_id).is_none(),
12299            "no nonce record for a process that will never present one"
12300        );
12301
12302        // Control: the subc-wire path is unchanged by the branch above.
12303        let wire_spec = spec(Vec::new());
12304        let mut wire = Command::new("/nonexistent");
12305        let wire_handoff =
12306            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
12307                .expect("subc-wire spawn args apply");
12308        let wire_fd_env = wire
12309            .as_std()
12310            .get_envs()
12311            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
12312            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
12313        #[cfg(unix)]
12314        assert_eq!(
12315            wire_fd_env,
12316            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
12317            "a subc-wire spawn names the pipe it will receive at descriptor 3"
12318        );
12319        #[cfg(not(unix))]
12320        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
12321        let wire_args: Vec<String> = wire
12322            .as_std()
12323            .get_args()
12324            .map(|a| a.to_string_lossy().into_owned())
12325            .collect();
12326        assert_eq!(
12327            wire_args,
12328            vec![
12329                SUBC_ARG.to_string(),
12330                connection_file.to_string_lossy().into_owned()
12331            ],
12332            "a subc-wire spawn still carries --subc <path>"
12333        );
12334        assert_eq!(
12335            wire.as_std()
12336                .get_envs()
12337                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
12338            !cfg!(unix),
12339            "only Windows supplies the environment nonce"
12340        );
12341        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
12342    }
12343
12344    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
12345    /// spec tries to set it; only a swap candidate carries it.
12346    ///
12347    /// "Set it only on candidates" is not enough, because spawn applies the
12348    /// spec's env verbatim and the daemon's own environment is inherited: either
12349    /// could hand a plain restart the swap role, and a module reading it would
12350    /// warm on its long swap budget while callers wait. Asserted as an explicit
12351    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
12352    /// test above gives.
12353    #[test]
12354    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
12355        let role = |command: &Command| {
12356            command
12357                .as_std()
12358                .get_envs()
12359                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
12360                .last()
12361                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
12362        };
12363        let forged = spec(vec![(
12364            SUBC_SPAWN_ROLE_ENV.to_string(),
12365            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
12366        )]);
12367
12368        let mut plain = Command::new("/nonexistent");
12369        apply_child_env(&mut plain, &forged);
12370        apply_spawn_role(&mut plain, SpawnRole::Plain);
12371        assert_eq!(
12372            role(&plain),
12373            Some(None),
12374            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
12375        );
12376
12377        let mut candidate = Command::new("/nonexistent");
12378        apply_child_env(&mut candidate, &spec(Vec::new()));
12379        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
12380        assert_eq!(
12381            role(&candidate),
12382            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
12383        );
12384    }
12385
12386    /// Daemon-private capture retention keys never reach the child.
12387    ///
12388    /// cortexkit-log exposes retention as a Rust struct with no environment
12389    /// names, so these entries are supervisor metadata. Passing them through
12390    /// would invent a public child-process contract by accident.
12391    #[test]
12392    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
12393        let mut command = Command::new("/nonexistent");
12394        apply_child_env(
12395            &mut command,
12396            &spec(vec![
12397                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
12398                ("KEPT".to_string(), "yes".to_string()),
12399            ]),
12400        );
12401        let keys: Vec<String> = command
12402            .as_std()
12403            .get_envs()
12404            .filter(|(_, value)| value.is_some())
12405            .map(|(key, _)| key.to_string_lossy().into_owned())
12406            .collect();
12407        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
12408        assert!(
12409            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
12410            "daemon-private capture key leaked to the child: {keys:?}"
12411        );
12412    }
12413}
12414
12415#[cfg(test)]
12416mod jitter_tests {
12417    use super::jittered_health_delay;
12418    use std::{collections::HashSet, time::Duration};
12419
12420    /// Module ids drawn from a real fleet, so the dispersal claim is about names
12421    /// that actually occur rather than invented ones.
12422    ///
12423    /// This is a SAMPLE, not a registry: the property under test is that distinct
12424    /// ids disperse, which holds for any set of distinct strings. Several entries
12425    /// are already historical (modules get renamed), and that costs nothing here --
12426    /// but it means a reader must not mistake this for the live module set, and a
12427    /// rename sweep will match it without there being anything to change.
12428    const FLEET: [&str; 14] = [
12429        "aft",
12430        "alfonso-core",
12431        "magic-context",
12432        "broca",
12433        "thalamus",
12434        "quota",
12435        "engram",
12436        "plexus",
12437        "cerebellum",
12438        "astrocyte",
12439        "synapse",
12440        "subc-mcp",
12441        "cortexkit-credentials",
12442        "subc-federation",
12443    ];
12444
12445    /// Probes must not converge after a fleet-wide restart.
12446    ///
12447    /// This is the property the jitter exists for: every module reconnects at
12448    /// once, and without dispersal all fourteen would then probe on the same
12449    /// tick forever. Nothing failed visibly when this went untested -- a
12450    /// convergent fleet still probes correctly, just in a burst, so the symptom
12451    /// is a periodic load spike that looks like whatever else is running.
12452    #[test]
12453    fn probe_delays_disperse_across_the_fleet() {
12454        let cadence = Duration::from_secs(30);
12455        let delays: HashSet<Duration> = FLEET
12456            .iter()
12457            .map(|id| jittered_health_delay(id, 0, cadence))
12458            .collect();
12459        assert_eq!(
12460            delays.len(),
12461            FLEET.len(),
12462            "every supervised module must land on its own probe offset"
12463        );
12464    }
12465
12466    /// The offset may only ever DELAY a probe, never bring it forward.
12467    ///
12468    /// A delay below the cadence would probe a module more often than
12469    /// configured, which is the opposite of what an operator asked for and
12470    /// would tighten the failure budget without anyone changing it.
12471    #[test]
12472    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
12473        let cadence = Duration::from_secs(30);
12474        let span = cadence / 10;
12475        for id in FLEET {
12476            for probe_index in 0..8 {
12477                let delay = jittered_health_delay(id, probe_index, cadence);
12478                assert!(
12479                    delay >= cadence,
12480                    "{id}#{probe_index}: jitter must not shorten the cadence"
12481                );
12482                assert!(
12483                    delay < cadence + span,
12484                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
12485                );
12486            }
12487        }
12488    }
12489
12490    /// A module keeps its offset across daemon restarts.
12491    ///
12492    /// The delay is derived rather than randomised precisely so a restart does
12493    /// not re-roll every module into a fresh chance of collision. A random
12494    /// source would satisfy the dispersal test above and quietly lose this.
12495    #[test]
12496    fn a_module_offset_is_stable_across_restarts() {
12497        let cadence = Duration::from_secs(30);
12498        for id in FLEET {
12499            assert_eq!(
12500                jittered_health_delay(id, 0, cadence),
12501                jittered_health_delay(id, 0, cadence),
12502                "{id}: the same module and probe index must produce the same offset"
12503            );
12504        }
12505    }
12506
12507    /// A zero cadence disables probing rather than producing a busy loop.
12508    #[test]
12509    fn zero_cadence_yields_zero_delay() {
12510        assert_eq!(
12511            jittered_health_delay("aft", 0, Duration::ZERO),
12512            Duration::ZERO
12513        );
12514    }
12515}
12516
12517#[cfg(all(test, target_os = "linux"))]
12518mod cgroup_placement_tests {
12519    use super::{
12520        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
12521        SupervisedChild,
12522    };
12523    use crate::stderr_tail::{StderrRing, StderrTailConfig};
12524    use std::{
12525        fs, io,
12526        path::{Path, PathBuf},
12527        sync::{Arc, Mutex},
12528    };
12529    use subc_test_support::TestTempDir;
12530    use tokio::process::Command;
12531
12532    #[tokio::test]
12533    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
12534        use super::*;
12535        let dir = TestTempDir::new("unique-spawn-cgroups");
12536        let root = PathBuf::from(format!(
12537            "/sys/fs/cgroup/subc-unique-test-{}-{}",
12538            std::process::id(),
12539            unix_ms_now()
12540        ));
12541        if let Err(error) = fs::create_dir(&root) {
12542            assert!(
12543                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12544                "required cgroup test cannot execute: {error}"
12545            );
12546            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
12547            return;
12548        }
12549        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
12550        let group_count = || {
12551            fs::read_dir(root.join("subc-modules"))
12552                .unwrap()
12553                .map(|entry| entry.unwrap().file_type().unwrap())
12554                .filter(|kind| kind.is_dir())
12555                .count()
12556        };
12557        let supervisor =
12558            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
12559                .with_cgroup_placement(Some(placement.clone()));
12560        let runtime = supervisor.runtime_config();
12561        let mut spec = ModuleSpec {
12562            module_id: "unique-spawn".into(),
12563            program: PathBuf::from("/bin/sleep"),
12564            args: vec!["60".into()],
12565            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
12566                .into_iter()
12567                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
12568                .collect(),
12569            reserved: false,
12570            reserved_prefixes: vec![],
12571            protocol: ModuleProtocol::None,
12572            overlap: Default::default(),
12573        };
12574        let spawn = |spec: &ModuleSpec| {
12575            spawn_child(
12576                spec,
12577                None,
12578                None,
12579                &runtime.stderr_ring,
12580                None,
12581                &runtime.child_roster,
12582                Some(&placement),
12583            )
12584            .unwrap()
12585        };
12586        let mut live = spawn(&spec);
12587        for _ in 0..3 {
12588            // A new process can enter the old slot while retirement is pending.
12589            let next = spawn(&spec);
12590            assert_ne!(live.module_id, next.module_id);
12591            live.start_kill().unwrap();
12592            live.wait().await.unwrap();
12593            live = next;
12594            assert!(
12595                live.child.try_wait().unwrap().is_none(),
12596                "retiring the old slot must not kill the replacement"
12597            );
12598            assert_eq!(
12599                group_count(),
12600                1,
12601                "only the live spawn's cgroup should remain"
12602            );
12603        }
12604        supervisor.begin_daemon_shutdown();
12605        let reap = tokio::spawn(async move {
12606            live.wait().await.unwrap();
12607        });
12608        supervisor
12609            .end_children_for_daemon_shutdown(false, std::future::pending())
12610            .await;
12611        reap.await.unwrap();
12612        assert_eq!(group_count(), 0);
12613        // A normal exit uses the same tree-cleanup path as a killed spawn.
12614        spec.program = PathBuf::from("/bin/true");
12615        spec.args.clear();
12616        let fresh_roster = ChildRoster::default();
12617        let mut short = spawn_child(
12618            &spec,
12619            None,
12620            None,
12621            &runtime.stderr_ring,
12622            None,
12623            &fresh_roster,
12624            Some(&placement),
12625        )
12626        .unwrap();
12627        short.wait().await.unwrap();
12628        assert_eq!(group_count(), 0);
12629        spec.module_id = "_".repeat(255);
12630        let mut long_id = spawn_child(
12631            &spec,
12632            None,
12633            None,
12634            &runtime.stderr_ring,
12635            None,
12636            &fresh_roster,
12637            Some(&placement),
12638        )
12639        .unwrap();
12640        long_id.wait().await.unwrap();
12641        assert_eq!(
12642            group_count(),
12643            0,
12644            "valid long module IDs must not exceed cgroup NAME_MAX"
12645        );
12646        fs::remove_dir(root.join("subc-modules")).unwrap();
12647        fs::remove_dir(root).unwrap();
12648        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
12649    }
12650
12651    #[test]
12652    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
12653        let path = Path::new("/definitely-missing-subc-cgroup");
12654        let mut command = Command::new("true");
12655        let error = apply_cgroup_placement(
12656            &mut command,
12657            &ModuleSpec {
12658                module_id: "broken-cgroup".to_string(),
12659                program: PathBuf::from("true"),
12660                args: Vec::new(),
12661                env: Vec::new(),
12662                reserved: false,
12663                reserved_prefixes: Vec::new(),
12664                protocol: ModuleProtocol::Subc,
12665                overlap: Default::default(),
12666            },
12667            path,
12668        )
12669        .expect_err("a parent cgroup open failure must reject the supervised spawn");
12670        let reason = error.to_string();
12671
12672        assert!(
12673            matches!(error, SuperviseError::Cgroup { .. }),
12674            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
12675        );
12676        assert!(
12677            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
12678            "parent cgroup open failure must name cgroup.procs: {reason}"
12679        );
12680    }
12681
12682    #[tokio::test]
12683    async fn reaping_a_child_removes_its_empty_module_cgroup() {
12684        let root = TestTempDir::new("supervisor-reap-cgroup");
12685        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12686        let placement = subc_cgroup::prepare_at(&root)
12687            .expect("prepare scratch cgroup root")
12688            .expect("scratch root has a cgroup.procs marker");
12689        let module_id = "reaped-module";
12690        let module = placement
12691            .module_path(module_id)
12692            .expect("create scratch module cgroup");
12693        let child = Command::new("true")
12694            .env("XDG_DATA_HOME", root.path())
12695            .env("XDG_RUNTIME_DIR", root.path())
12696            .env("XDG_CONFIG_HOME", root.path())
12697            .spawn()
12698            .expect("spawn short-lived child");
12699        let pid = child.id().expect("spawned child has pid");
12700        let mut child = SupervisedChild {
12701            child,
12702            protocol: ModuleProtocol::Subc,
12703            module_id: module_id.to_string(),
12704            cgroup_placement: Some(placement),
12705            stdout_pump: None,
12706            stderr_pump: None,
12707            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
12708            spawned_at_ms: 0,
12709            spawned_from: PathBuf::from("true"),
12710            spawned_file_identity: None,
12711            process_start_time: None,
12712            process_identity: None,
12713            pid,
12714            roster_guard: None,
12715            #[cfg(target_os = "macos")]
12716            privacy_exec: None,
12717            spawn_failure: None,
12718        };
12719
12720        child.wait().await.expect("reap short-lived child");
12721
12722        assert!(
12723            !module.exists(),
12724            "reaping the supervised child must remove its empty cgroup"
12725        );
12726    }
12727
12728    #[test]
12729    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
12730        let root = TestTempDir::new("supervisor-non-empty-cgroup");
12731        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12732        let placement = subc_cgroup::prepare_at(&root)
12733            .expect("prepare scratch cgroup root")
12734            .expect("scratch root has a cgroup.procs marker");
12735        let module = placement
12736            .module_path("surviving-module")
12737            .expect("create scratch module cgroup");
12738        fs::write(module.join("surviving-process"), b"still present")
12739            .expect("make scratch cgroup non-empty");
12740        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
12741
12742        remove_module_cgroup(&placement, "surviving-module");
12743
12744        let logs = crate::router::test_log::captured_logs(&logs);
12745        assert!(
12746            module.exists(),
12747            "failed removal must leave the cgroup intact"
12748        );
12749        assert!(
12750            logs.contains("could not remove module cgroup after process exit; continuing teardown")
12751                && logs.contains("surviving-module"),
12752            "best-effort removal must report the failure without returning it: {logs}"
12753        );
12754    }
12755
12756    #[test]
12757    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
12758        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
12759        let reason = SuperviseError::Spawn {
12760            program: PathBuf::from("/bin/true"),
12761            source: io::Error::from_raw_os_error(13),
12762            cgroup_path: Some(cgroup_path.clone()),
12763        }
12764        .to_string();
12765
12766        assert!(
12767            reason.contains(&cgroup_path.display().to_string()),
12768            "a pre_exec spawn failure must name the cgroup path: {reason}"
12769        );
12770    }
12771}
12772
12773#[cfg(test)]
12774mod spawn_subscriber_lag_tests {
12775    use super::*;
12776
12777    /// A subscriber whose connection stops draining is dropped once its frame
12778    /// channel fills. The client must learn that from a terminal Error frame
12779    /// after the frames already queued for it, not from a stream that simply
12780    /// goes quiet.
12781    #[tokio::test]
12782    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
12783        let feed = SpawnEventFeed::default();
12784        feed.configure_incarnation("lag-incarnation".to_string());
12785        // A one-slot connection queue that nobody reads until the emits are
12786        // done: the forwarder parks on it and the subscriber channel fills.
12787        let (tx, mut rx) = mpsc::channel(1);
12788        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
12789            .expect("subscribe");
12790        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
12791        for index in 0..emitted {
12792            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
12793            // Let the forwarder take what it can so the fill point is the
12794            // subscriber channel, not a scheduling accident.
12795            tokio::task::yield_now().await;
12796        }
12797        assert_eq!(
12798            feed.subscriber_count(),
12799            0,
12800            "the lagged subscriber must be removed"
12801        );
12802
12803        let mut data = Vec::new();
12804        let mut last = None;
12805        loop {
12806            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
12807                .await
12808                .expect("the forwarder must finish once the subscriber is dropped");
12809            let Some(outbound) = next else { break };
12810            let frame = outbound.frame;
12811            if frame.header.ty == FrameType::StreamData {
12812                assert!(last.is_none(), "no data may follow the terminal frame");
12813                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
12814                data.push(event.cursor.seq);
12815            } else {
12816                assert!(last.is_none(), "exactly one terminal frame");
12817                last = Some(frame);
12818            }
12819        }
12820        assert!(!data.is_empty(), "queued frames drain before the terminal");
12821        for pair in data.windows(2) {
12822            assert_eq!(
12823                pair[1],
12824                pair[0] + 1,
12825                "queued frames arrive dense and in order"
12826            );
12827        }
12828        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
12829        assert_eq!(terminal.header.ty, FrameType::Error);
12830        assert_eq!(terminal.header.corr, 7);
12831        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
12832        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
12833        let detail = body.detail.expect("lagged error carries detail");
12834        assert_eq!(
12835            detail["first_undelivered_cursor"]["seq"],
12836            data.last().unwrap() + 1,
12837            "the named cursor is the first event the subscriber did not receive"
12838        );
12839        assert_eq!(
12840            detail["first_undelivered_cursor"]["daemon_incarnation"],
12841            "lag-incarnation"
12842        );
12843    }
12844}
12845
12846#[cfg(test)]
12847mod terminal_history_read_concurrency_tests {
12848    use super::*;
12849    use crate::terminal_journal::read_pause;
12850    use std::sync::mpsc as std_mpsc;
12851    use subc_test_support::TestTempDir;
12852
12853    fn journaled_ring(
12854        journal: &Arc<crate::terminal_journal::TerminalJournal>,
12855    ) -> Arc<Mutex<TerminalRing>> {
12856        Arc::new(Mutex::new(
12857            TerminalRing::new(TerminalRingConfig::default(), 1)
12858                .with_journal(Some(Arc::clone(journal))),
12859        ))
12860    }
12861
12862    fn crash(at_ms: u64) -> ExitReport {
12863        ExitReport {
12864            kind: ExitKind::Crash,
12865            code: Some(1),
12866            signal: None,
12867            at_ms,
12868        }
12869    }
12870
12871    /// Record an exit on another thread and report whether it finished within
12872    /// `bound`. The recorder thread is left running if it did not.
12873    fn record_within(
12874        module_id: &'static str,
12875        ring: &Arc<Mutex<TerminalRing>>,
12876        at_ms: u64,
12877        bound: Duration,
12878    ) -> bool {
12879        let ring = Arc::clone(ring);
12880        let (done, done_rx) = std_mpsc::channel();
12881        std::thread::spawn(move || {
12882            record_terminal(
12883                module_id,
12884                &ring,
12885                &SpawnEventFeed::default(),
12886                &crash(at_ms),
12887                TerminalDisposition::Restarting,
12888            );
12889            let _ = done.send(());
12890        });
12891        done_rx.recv_timeout(bound).is_ok()
12892    }
12893
12894    /// A history read in progress must not hold the journal writer (which every
12895    /// module's exit recording needs) or the module's own ring. Exits recorded
12896    /// while the read is paused complete promptly; the paused read answers as of
12897    /// the moment it started, and the next read has each exit exactly once.
12898    #[test]
12899    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12900        let dir = TestTempDir::new("terminal-history-concurrent-read");
12901        let path = dir.join("terminals.jsonl");
12902        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12903            path.clone(),
12904            "daemon".into(),
12905        ));
12906        let reader_ring = journaled_ring(&journal);
12907        let other_ring = journaled_ring(&journal);
12908        assert!(record_within(
12909            "reader-module",
12910            &reader_ring,
12911            10,
12912            Duration::from_secs(5)
12913        ));
12914
12915        let (started, release) = read_pause::install(&path);
12916        let reading = {
12917            let ring = Arc::clone(&reader_ring);
12918            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12919        };
12920        started
12921            .recv_timeout(Duration::from_secs(5))
12922            .expect("the history read reached its pause");
12923
12924        let bound = Duration::from_secs(1);
12925        assert!(
12926            record_within("other-module", &other_ring, 20, bound),
12927            "another module's exit waited on a history read (journal writer held)"
12928        );
12929        assert!(
12930            record_within("reader-module", &reader_ring, 30, bound),
12931            "the read module's own exit waited on its history read (ring held)"
12932        );
12933
12934        drop(release);
12935        let paused = reading.join().unwrap();
12936        assert_eq!(
12937            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12938            vec![10],
12939            "an exit recorded after the read began lands in neither half of it"
12940        );
12941        assert_eq!(paused.journal_skipped_lines, 0);
12942        assert_eq!(paused.journal_read_errors, 0);
12943
12944        let after = durable_terminal_history_of(&reader_ring, "reader-module");
12945        assert_eq!(
12946            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12947            vec![10, 30],
12948            "the next read merges ring and journal with no duplicate"
12949        );
12950        assert_eq!(after.journal_skipped_lines, 0);
12951    }
12952}
12953
12954/// What a restart does with the exited process's stderr reader. These drive
12955/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12956/// holds, so a reader that has not been scheduled by the bound is a controlled
12957/// input rather than something only a loaded machine produces.
12958#[cfg(test)]
12959mod stderr_settle_tests {
12960    use std::{
12961        future::Future,
12962        io,
12963        pin::Pin,
12964        sync::{Arc, Mutex},
12965        task::{Context, Poll},
12966        time::Duration,
12967    };
12968
12969    use tokio::{
12970        io::{AsyncRead, ReadBuf},
12971        sync::oneshot,
12972        time::Instant,
12973    };
12974
12975    use super::{settle_stderr_pump, StderrPump};
12976    use crate::stderr_tail::{
12977        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12978    };
12979
12980    const BOUND: Duration = Duration::from_millis(250);
12981
12982    /// Yields `before`, then stays pending until the gate is released, then
12983    /// yields `after` and reaches EOF. The bytes after the gate were written
12984    /// by a process that has already exited; only the reader is behind.
12985    struct HeldReader {
12986        before: Option<Vec<u8>>,
12987        gate: Option<oneshot::Receiver<()>>,
12988        after: io::Cursor<Vec<u8>>,
12989    }
12990
12991    impl AsyncRead for HeldReader {
12992        fn poll_read(
12993            mut self: Pin<&mut Self>,
12994            cx: &mut Context<'_>,
12995            buf: &mut ReadBuf<'_>,
12996        ) -> Poll<io::Result<()>> {
12997            if let Some(bytes) = self.before.take() {
12998                buf.put_slice(&bytes);
12999                return Poll::Ready(Ok(()));
13000            }
13001            if let Some(gate) = self.gate.as_mut() {
13002                match Pin::new(gate).poll(cx) {
13003                    Poll::Pending => return Poll::Pending,
13004                    Poll::Ready(_) => self.gate = None,
13005                }
13006            }
13007            Pin::new(&mut self.after).poll_read(cx, buf)
13008        }
13009    }
13010
13011    struct DiscardSink;
13012
13013    impl OutputSink for DiscardSink {
13014        fn write_line(&mut self, _line: &[u8]) {}
13015    }
13016
13017    fn line(text: &str) -> TailEntry {
13018        TailEntry::Line {
13019            text: text.to_string(),
13020            truncated: false,
13021            at_ms: None,
13022        }
13023    }
13024
13025    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
13026        ring.lock().unwrap()
13027    }
13028
13029    /// Start a reader for a new process generation that delivers `before`
13030    /// immediately and `after` only once the returned sender fires (or is
13031    /// dropped).
13032    fn held_pump(
13033        ring: &Arc<Mutex<StderrRing>>,
13034        before: &str,
13035        after: &str,
13036    ) -> (StderrPump, oneshot::Sender<()>) {
13037        let generation = lock(ring).begin_process();
13038        let (release, gate) = oneshot::channel();
13039        let reader = HeldReader {
13040            before: Some(before.as_bytes().to_vec()),
13041            gate: Some(gate),
13042            after: io::Cursor::new(after.as_bytes().to_vec()),
13043        };
13044        let task = tokio::spawn(pump_stderr_to(
13045            reader,
13046            Arc::clone(ring),
13047            generation,
13048            DiscardSink,
13049        ));
13050        (StderrPump { task, generation }, release)
13051    }
13052
13053    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
13054        for _ in 0..1000 {
13055            if done(&lock(ring)) {
13056                return;
13057            }
13058            tokio::time::sleep(Duration::from_millis(1)).await;
13059        }
13060        panic!(
13061            "ring never reached the expected state: {:?}",
13062            lock(ring).snapshot(None, None)
13063        );
13064    }
13065
13066    #[tokio::test(start_paused = true)]
13067    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
13068        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13069        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
13070
13071        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
13072        let before_release = lock(&ring).snapshot(None, None);
13073        assert!(
13074            matches!(before_release.capture, CaptureState::Incomplete { .. }),
13075            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
13076        );
13077
13078        // The restart: the next process starts and writes before the old
13079        // reader catches up.
13080        let next = lock(&ring).begin_process();
13081        lock(&ring).push_line_from(next, "next process booting");
13082        release.send(()).unwrap();
13083        wait_until(&ring, |ring| {
13084            ring.snapshot(None, None).capture == CaptureState::Captured
13085        })
13086        .await;
13087
13088        assert_eq!(
13089            untimed(lock(&ring).snapshot(None, None).entries),
13090            vec![
13091                line("booting"),
13092                line("config error: missing storage"),
13093                TailEntry::ProcessStart,
13094                line("next process booting"),
13095            ],
13096            "the crash's last line must survive a slow reader and stay in the crashed process's section"
13097        );
13098    }
13099
13100    #[tokio::test(start_paused = true)]
13101    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
13102    ) {
13103        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13104        // `_held` is never fired: a descendant keeps the pipe open for the
13105        // whole test.
13106        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
13107
13108        let started = Instant::now();
13109        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
13110        assert_eq!(
13111            started.elapsed(),
13112            BOUND,
13113            "the restart must wait exactly the bound for a pipe that stays open, no longer"
13114        );
13115
13116        let next = lock(&ring).begin_process();
13117        lock(&ring).push_line_from(next, "next process booting");
13118        tokio::time::sleep(Duration::from_secs(60)).await;
13119
13120        let snapshot = lock(&ring).snapshot(None, None);
13121        match &snapshot.capture {
13122            CaptureState::Incomplete { reason } => assert!(
13123                reason.contains("had not reached EOF") && reason.contains("250ms"),
13124                "the reason must say what is missing and after how long: {reason}"
13125            ),
13126            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
13127        }
13128        assert_eq!(
13129            untimed(snapshot.entries),
13130            vec![
13131                line("parent exiting"),
13132                TailEntry::ProcessStart,
13133                line("next process booting"),
13134            ]
13135        );
13136    }
13137
13138    #[tokio::test(start_paused = true)]
13139    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
13140        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13141        let (pump, release) = held_pump(&ring, "one\n", "two\n");
13142        release.send(()).unwrap();
13143
13144        settle_stderr_pump("clean", &ring, pump, BOUND).await;
13145
13146        let snapshot = lock(&ring).snapshot(None, None);
13147        assert_eq!(snapshot.capture, CaptureState::Captured);
13148        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
13149    }
13150}
13151
13152/// Containment of a module's process tree (issue #109).
13153///
13154/// The behaviour these defend against is a module helper surviving its module:
13155/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
13156/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
13157/// compounds it.
13158///
13159/// They run against the SUPERVISOR rather than the job-object crate because the
13160/// claim is about teardown: a crate-level test proves a job can reap a tree, not
13161/// that the daemon's drain path reaches it.
13162///
13163/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
13164/// lane there is a separate containment path with its own tests.
13165#[cfg(all(test, windows))]
13166mod job_containment_tests {
13167    use super::*;
13168    use std::{
13169        path::{Path, PathBuf},
13170        sync::{Arc, Mutex},
13171        time::{Duration, Instant},
13172    };
13173    use subc_test_support::TestTempDir;
13174
13175    /// The stub, expected beside this test executable.
13176    ///
13177    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
13178    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
13179    /// failure then reads as a broken test rather than an unbuilt dependency.
13180    fn stub_path() -> PathBuf {
13181        let mut path = std::env::current_exe().expect("current_exe available in tests");
13182        path.pop();
13183        path.pop();
13184        path.push("fake-aft-stub.exe");
13185        assert!(
13186            path.exists(),
13187            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
13188             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
13189            path.display()
13190        );
13191        path
13192    }
13193
13194    /// Poll for the grandchild pid the stub records, and parse it.
13195    fn read_grandchild_pid(path: &Path) -> u32 {
13196        let deadline = Instant::now() + Duration::from_secs(10);
13197        loop {
13198            if let Ok(contents) = std::fs::read_to_string(path) {
13199                if let Ok(pid) = contents.trim().parse() {
13200                    return pid;
13201                }
13202            }
13203            assert!(
13204                Instant::now() < deadline,
13205                "the stub never recorded a grandchild pid at {}",
13206                path.display()
13207            );
13208            std::thread::sleep(Duration::from_millis(10));
13209        }
13210    }
13211
13212    /// Everything one fixture run needs, so the two tests below differ in exactly
13213    /// one place: whether the child is contained.
13214    struct Fixture {
13215        _dir: TestTempDir,
13216        module_id: String,
13217        grandchild: u32,
13218        child: Option<SupervisedChild>,
13219        registry: Arc<Registry>,
13220        snapshot: Arc<Mutex<SupervisorSnapshot>>,
13221        terminal_ring: Arc<Mutex<TerminalRing>>,
13222        spawn_events: SpawnEventFeed,
13223    }
13224
13225    fn fixture(label: &str, module_id: &str) -> Fixture {
13226        let dir = TestTempDir::new(label);
13227        let pid_file = dir.join("grandchild.pid");
13228        let supervisor = Supervisor::new_for_test(
13229            Arc::new(Registry::default()),
13230            RestartPolicy::new(3, Duration::ZERO),
13231        );
13232        let runtime = supervisor.runtime_config();
13233        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13234        let spec = ModuleSpec {
13235            module_id: module_id.to_string(),
13236            program: stub_path(),
13237            // Zero args deliberately: a `--subc` argument would make the stub dial
13238            // a daemon that is not there, and the failure would land in the same
13239            // stderr ring this fixture exists to keep quiet.
13240            args: Vec::new(),
13241            env: vec![
13242                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
13243                (
13244                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
13245                    pid_file.display().to_string(),
13246                ),
13247            ],
13248            reserved: false,
13249            reserved_prefixes: Vec::new(),
13250            protocol: ModuleProtocol::Subc,
13251            overlap: Default::default(),
13252        };
13253        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
13254            .expect("spawn the supervised fixture");
13255        let grandchild = read_grandchild_pid(&pid_file);
13256        Fixture {
13257            _dir: dir,
13258            module_id: module_id.to_string(),
13259            grandchild,
13260            child: Some(child),
13261            registry: Arc::new(Registry::default()),
13262            snapshot,
13263            terminal_ring: Arc::clone(&runtime.terminal_ring),
13264            spawn_events: SpawnEventFeed::default(),
13265        }
13266    }
13267
13268    impl Fixture {
13269        /// Drain through the supervisor's own teardown path.
13270        async fn drain(&mut self) {
13271            let child = self
13272                .child
13273                .take()
13274                .expect("the fixture child is still present");
13275            drain_child_to_state(
13276                &self.module_id,
13277                ModuleProtocol::Subc,
13278                // No forwarding table in this fixture, so nothing reaches the
13279                // child over a connection.
13280                StopNotice::NotSent,
13281                &self.registry,
13282                None,
13283                &self.snapshot,
13284                &self.terminal_ring,
13285                &self.spawn_events,
13286                child,
13287                Duration::from_millis(500),
13288                ModuleState::Stopped,
13289                Some(false),
13290            )
13291            .await
13292            .expect("drain the supervised fixture");
13293        }
13294    }
13295
13296    /// Teardown reaps the grandchild, not merely the direct child.
13297    ///
13298    /// This is the assertion the change exists for. Before containment the
13299    /// grandchild survived: it is a separate process, and `start_kill` is
13300    /// `TerminateProcess` scoped to one pid.
13301    #[tokio::test]
13302    async fn teardown_reaps_the_grandchild() {
13303        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
13304        let grandchild = fixture.grandchild;
13305
13306        assert!(
13307            subc_jobobject::process_exists(grandchild),
13308            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
13309        );
13310
13311        fixture.drain().await;
13312
13313        assert!(
13314            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
13315            "grandchild {grandchild} outlived module teardown: the tree was not contained"
13316        );
13317    }
13318
13319    /// The mutation control: with containment withheld, the grandchild survives
13320    /// the same kill.
13321    ///
13322    /// This is the defect reproduction from #109 — a direct-child kill reaches
13323    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
13324    /// supervisor because `spawn_and_mark_running` now always contains on
13325    /// Windows, which is the point: there is no longer a path that spawns
13326    /// uncontained, so the control has to construct one.
13327    ///
13328    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
13329    /// grandchild ever dies here, that test is passing for a reason unrelated to
13330    /// the job object and the containment claim is unproven.
13331    #[test]
13332    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
13333        let dir = TestTempDir::new("teardown-uncontained");
13334        let pid_file = dir.join("grandchild.pid");
13335        let mut child = std::process::Command::new(stub_path())
13336            .env("FAKE_AFT_NEVER_CONNECT", "1")
13337            .env(
13338                "FAKE_AFT_GRANDCHILD_PID_FILE",
13339                pid_file.display().to_string(),
13340            )
13341            .stdin(std::process::Stdio::null())
13342            .stdout(std::process::Stdio::null())
13343            .stderr(std::process::Stdio::null())
13344            .spawn()
13345            .expect("spawn the uncontained fixture");
13346        let grandchild = read_grandchild_pid(&pid_file);
13347
13348        // Exactly what the pre-fix teardown did: kill the direct child.
13349        child.kill().expect("kill the direct child");
13350        let _ = child.wait();
13351
13352        assert!(
13353            subc_jobobject::process_exists(grandchild),
13354            "grandchild {grandchild} died with the direct child, so this control no longer \
13355             distinguishes contained from uncontained teardown and the regression test is \
13356             passing vacuously"
13357        );
13358
13359        // The orphan this control demonstrates is the leak the fix prevents, so
13360        // the control must not leave one behind.
13361        kill_tree(grandchild);
13362    }
13363
13364    /// Crash durability: closing the containment handle reaps the tree with no
13365    /// teardown code running at all.
13366    ///
13367    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
13368    /// call anything — and it is why containment is a kernel property of the
13369    /// handle rather than a step in the drain. Discovered by getting the
13370    /// mutation control wrong: clearing `job` to "disable" containment instead
13371    /// killed the tree, which is the guarantee, not a mistake.
13372    #[tokio::test]
13373    async fn dropping_containment_reaps_the_grandchild() {
13374        let mut fixture = fixture("drop-containment", "tree-drop");
13375        let grandchild = fixture.grandchild;
13376
13377        assert!(subc_jobobject::process_exists(grandchild));
13378
13379        // No `drain` call, no kill: dropping the handle is the entire mechanism.
13380        fixture.child.as_mut().expect("child present").job = None;
13381
13382        assert!(
13383            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
13384            "grandchild {grandchild} survived the containment handle closing, so a daemon \
13385             crash would leave the tree behind"
13386        );
13387    }
13388
13389    /// Kill a pid and its tree, then confirm it is gone.
13390    fn kill_tree(pid: u32) {
13391        let _ = std::process::Command::new("taskkill.exe")
13392            .args(["/PID", &pid.to_string(), "/T", "/F"])
13393            .stdin(std::process::Stdio::null())
13394            .stdout(std::process::Stdio::null())
13395            .stderr(std::process::Stdio::null())
13396            .status();
13397        assert!(
13398            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
13399            "could not clean up grandchild {pid}"
13400        );
13401    }
13402}
13403
13404#[cfg(test)]
13405mod privacy_trampoline_configuration_tests {
13406    #[tokio::test]
13407    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13408    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
13409        #[cfg(target_os = "macos")]
13410        {
13411            let supervisor = super::Supervisor::new(
13412                std::sync::Arc::new(crate::Registry::default()),
13413                super::RestartPolicy::default(),
13414            );
13415            let error = supervisor.spawn(spec()).unwrap_err();
13416            assert!(
13417                error
13418                    .to_string()
13419                    .contains("no privacy trampoline configured"),
13420                "{error}"
13421            );
13422        }
13423    }
13424
13425    #[tokio::test]
13426    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13427    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
13428        #[cfg(target_os = "macos")]
13429        {
13430            let supervisor = super::Supervisor::new(
13431                std::sync::Arc::new(crate::Registry::default()),
13432                super::RestartPolicy::default(),
13433            )
13434            .with_privacy_trampoline(std::env::current_exe().unwrap());
13435            let error = supervisor.spawn(spec()).unwrap_err();
13436            assert!(
13437                error
13438                    .to_string()
13439                    .contains("binary does not implement the privacy trampoline protocol"),
13440                "{error}"
13441            );
13442        }
13443    }
13444
13445    #[cfg(target_os = "macos")]
13446    fn spec() -> super::ModuleSpec {
13447        super::ModuleSpec {
13448            module_id: "privacy-configuration".into(),
13449            program: "/bin/sleep".into(),
13450            args: vec!["30".into()],
13451            env: vec![],
13452            reserved: false,
13453            reserved_prefixes: vec![],
13454            protocol: subc_control::ModuleProtocol::None,
13455            overlap: super::ModuleOverlap::Exclusive,
13456        }
13457    }
13458}
13459
13460#[cfg(test)]
13461mod privacy_exec_boundary_tests {
13462    #[cfg(target_os = "macos")]
13463    use super::*;
13464    #[cfg(target_os = "macos")]
13465    use std::{
13466        io::{Read, Write},
13467        net::{TcpListener, TcpStream},
13468    };
13469
13470    /// Unit-test-only pause at the actual early image read, not at a later
13471    /// status read. Production supervisors never inspect this environment key.
13472    #[cfg(target_os = "macos")]
13473    pub(super) fn before_image_sample(spec: &ModuleSpec, pid: u32) {
13474        if let Some((_, path)) = spec
13475            .env
13476            .iter()
13477            .find(|(key, _)| key == "SUBC_TEST_PRIVACY_SAMPLE_BARRIER")
13478        {
13479            let mut barrier = TcpStream::connect(path).unwrap();
13480            barrier
13481                .set_read_timeout(Some(Duration::from_secs(30)))
13482                .unwrap();
13483            barrier.write_all(&pid.to_ne_bytes()).unwrap();
13484            let mut release = [0];
13485            barrier.read_exact(&mut release).unwrap();
13486            assert_eq!(&release, b"X");
13487        }
13488    }
13489
13490    #[cfg(target_os = "macos")]
13491    fn spec(program: &str, args: &[&str]) -> ModuleSpec {
13492        ModuleSpec {
13493            module_id: "privacy-boundary".into(),
13494            program: program.into(),
13495            args: args.iter().map(|arg| (*arg).into()).collect(),
13496            env: vec![],
13497            reserved: false,
13498            reserved_prefixes: vec![],
13499            protocol: ModuleProtocol::None,
13500            overlap: ModuleOverlap::Exclusive,
13501        }
13502    }
13503
13504    #[cfg(target_os = "macos")]
13505    async fn accept(listener: TcpListener) -> TcpStream {
13506        // Socket readiness, not elapsed time, establishes both pause points.
13507        let listener = tokio::net::TcpListener::from_std({
13508            listener.set_nonblocking(true).unwrap();
13509            listener
13510        })
13511        .unwrap();
13512        let (stream, _) = tokio::time::timeout(Duration::from_secs(30), listener.accept())
13513            .await
13514            .unwrap()
13515            .unwrap();
13516        let stream = stream.into_std().unwrap();
13517        stream.set_nonblocking(false).unwrap();
13518        stream
13519            .set_read_timeout(Some(Duration::from_secs(30)))
13520            .unwrap();
13521        stream
13522    }
13523
13524    #[tokio::test]
13525    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13526    async fn macos_roster_withholds_a_nonnull_trampoline_image_until_exec_confirmation() {
13527        #[cfg(target_os = "macos")]
13528        {
13529            let root = subc_test_support::TestTempDir::new("privacy-roster-barrier");
13530            // Loopback sockets also work when the replay adapter's TMPDIR is
13531            // longer than Darwin's Unix-domain socket path limit.
13532            let exec_listener = TcpListener::bind("127.0.0.1:0").unwrap();
13533            let sample_listener = TcpListener::bind("127.0.0.1:0").unwrap();
13534            let record = root.join("live-children.json");
13535            let supervisor =
13536                Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
13537                    .with_live_children_record(&record);
13538            let runtime = supervisor.runtime_config();
13539            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13540            let mut spec = spec("/bin/sleep", &["30"]);
13541            spec.env = vec![
13542                (
13543                    "SUBC_TEST_PRIVACY_EXEC_BARRIER".into(),
13544                    exec_listener.local_addr().unwrap().to_string(),
13545                ),
13546                (
13547                    "SUBC_TEST_PRIVACY_SAMPLE_BARRIER".into(),
13548                    sample_listener.local_addr().unwrap().to_string(),
13549                ),
13550            ];
13551            let spawn = tokio::task::spawn_blocking(move || {
13552                spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap()
13553            });
13554            let mut sample = accept(sample_listener).await;
13555            let mut pid = [0; 4];
13556            sample.read_exact(&mut pid).unwrap();
13557            let pid = u32::from_ne_bytes(pid);
13558            let mut exec = accept(exec_listener).await;
13559            let mut ready = [0];
13560            exec.read_exact(&mut ready).unwrap();
13561            assert_eq!(&ready, b"R");
13562            // The early read is guaranteed to see a real, non-null trampoline
13563            // image: the fixture has reached its barrier and cannot exec yet.
13564            let trampoline = subc_os::file_identity(&test_privacy_trampoline()).unwrap();
13565            assert_eq!(
13566                observe_spawned_image(pid).unwrap().executable,
13567                Some(trampoline)
13568            );
13569            sample.write_all(b"X").unwrap();
13570            let mut child = spawn.await.unwrap();
13571            let early = crate::live_children::read_record(&record).unwrap();
13572            assert_eq!(early.len(), 1);
13573            assert_eq!(early[0].pid, pid);
13574            assert_eq!(
13575                early[0].executable, None,
13576                "unconfirmed trampoline image entered the roster"
13577            );
13578            assert!(child.report_ready.get().is_none());
13579            // The barrier's duration is unrelated to the production five-second
13580            // exec budget. Start the test's confirmation budget upon release.
13581            child.privacy_exec.as_mut().unwrap().deadline =
13582                tokio::time::Instant::now() + Duration::from_secs(30);
13583            exec.write_all(b"X").unwrap();
13584            child.confirm_privacy_exec().await;
13585            assert_eq!(child.spawn_failure, None);
13586            assert!(child.report_ready.get().is_some());
13587            let confirmed = crate::live_children::read_record(&record).unwrap();
13588            let module = subc_os::file_identity(std::path::Path::new("/bin/sleep")).unwrap();
13589            assert_ne!(module, trampoline);
13590            assert_eq!(confirmed[0].executable, Some(module.into()));
13591            child.start_kill().unwrap();
13592            child.wait().await.unwrap();
13593            child.release_roster();
13594        }
13595    }
13596
13597    #[tokio::test]
13598    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13599    async fn macos_already_exited_121_in_try_wait_is_a_module_exit_not_a_trampoline_refusal() {
13600        #[cfg(target_os = "macos")]
13601        {
13602            let registry = Arc::new(Registry::default());
13603            let policy = RestartPolicy::new(0, Duration::ZERO);
13604            let supervisor = Supervisor::new_for_test(Arc::clone(&registry), policy);
13605            let runtime = supervisor.runtime_config();
13606            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13607            let spec = spec("/bin/sh", &["-c", "exit 121"]);
13608            // Drive spawn and confirmation separately instead of starting the
13609            // monitor. WNOWAIT observes a real exit without consuming its status,
13610            // so confirmation's first try_wait must take the already-exited arm.
13611            let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
13612            let pid = child.pid;
13613            tokio::task::spawn_blocking(move || {
13614                subc_os::fork_exec_test::wait_for_child_exit_without_reaping(pid)
13615            })
13616            .await
13617            .unwrap()
13618            .unwrap();
13619            child.privacy_exec.as_mut().unwrap().deadline =
13620                tokio::time::Instant::now() + Duration::from_secs(30);
13621            let status = child.wait().await.unwrap();
13622            assert_eq!(status.code(), Some(121));
13623            assert!(child.privacy_exec.is_none());
13624            assert!(
13625                child.report_ready.get().is_none(),
13626                "an exited module must not publish a live pid"
13627            );
13628            let report = classify_reaped_child_exit(&snapshot, &child, &status);
13629            on_child_exit(
13630                &spec,
13631                policy,
13632                &registry,
13633                &snapshot,
13634                &runtime.terminal_ring,
13635                &runtime.spawn_events,
13636                &runtime.child_roster,
13637                report,
13638            )
13639            .await;
13640            let state = lock_snapshot(&snapshot).unwrap();
13641            assert_eq!(state.state, ModuleState::Failed);
13642            assert_eq!(state.last_exit.as_ref().unwrap().code, Some(121));
13643            assert_eq!(state.reported_pid(), None);
13644            drop(state);
13645            let history = runtime.terminal_ring.lock().unwrap().snapshot();
13646            assert_eq!(history.entries.len(), 1);
13647            let terminal = &history.entries[0];
13648            assert_eq!(terminal.exit_code, Some(121));
13649            assert_eq!(terminal.exit_kind, subc_control::TerminalExitKind::Crash);
13650            assert_eq!(terminal.disposition, TerminalDisposition::Failed);
13651            assert_eq!(
13652                terminal.disposition_detail.as_deref(),
13653                Some(policy.budget_exhausted_detail().as_str()),
13654                "module exit 121 was classified as a trampoline refusal: {terminal:?}"
13655            );
13656            assert_eq!(child.spawn_failure, None);
13657            child.release_roster();
13658        }
13659    }
13660}
13661
13662/// The daemon's real spawn path hands a subc-wire child its launch nonce on
13663/// descriptor 3, without an environment copy. The shell records the nonce
13664/// and its environment after exec so these tests observe the real handover.
13665#[cfg(all(test, unix))]
13666mod launch_nonce_descriptor_tests {
13667    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
13668    use crate::stderr_tail::{StderrRing, StderrTailConfig};
13669    use std::{
13670        path::PathBuf,
13671        sync::{Arc, Mutex},
13672        time::{Duration, Instant},
13673    };
13674    use subc_test_support::TestTempDir;
13675
13676    async fn probe(role: super::SpawnRole) {
13677        let scratch = TestTempDir::new("launch-nonce-descriptor");
13678        let fd_copy = scratch.join("from-descriptor");
13679        let env_copy = scratch.join("environment");
13680        let script = format!(
13681            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
13682            fd = fd_copy.display(), env = env_copy.display(),
13683        );
13684        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
13685        let spec = ModuleSpec {
13686            module_id: "nonce-descriptor-probe".to_string(),
13687            program: PathBuf::from("/bin/sh"),
13688            args: vec!["-c".to_string(), script],
13689            env: vec![
13690                xdg("XDG_DATA_HOME"),
13691                xdg("XDG_RUNTIME_DIR"),
13692                xdg("XDG_CONFIG_HOME"),
13693                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
13694            ],
13695            reserved: true,
13696            reserved_prefixes: Vec::new(),
13697            protocol: ModuleProtocol::Subc,
13698            overlap: Default::default(),
13699        };
13700        let handle = SupervisorHandle::new();
13701        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13702        let roster = ChildRoster::default();
13703        #[cfg(target_os = "macos")]
13704        {
13705            let path = super::test_privacy_trampoline();
13706            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
13707        }
13708        let child = super::spawn_child_in_slot(
13709            &spec,
13710            None,
13711            Some(&handle),
13712            &ring,
13713            None,
13714            &roster,
13715            #[cfg(target_os = "linux")]
13716            None,
13717            role,
13718            matches!(role, super::SpawnRole::SwapCandidate),
13719        )
13720        .expect("spawn probe");
13721        let deadline = Instant::now() + Duration::from_secs(10);
13722        while !(fd_copy.exists() && env_copy.exists()) {
13723            assert!(Instant::now() < deadline, "probe never wrote its copies");
13724            tokio::time::sleep(Duration::from_millis(20)).await;
13725        }
13726        let nonce = std::fs::read_to_string(fd_copy).unwrap();
13727        assert!(!nonce.is_empty());
13728        let environment = std::fs::read_to_string(env_copy).unwrap();
13729        assert!(environment
13730            .lines()
13731            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
13732        let copy = environment
13733            .lines()
13734            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
13735        assert_eq!(
13736            copy, None,
13737            "Unix children must never receive the environment nonce"
13738        );
13739        if matches!(role, super::SpawnRole::Plain) {
13740            assert_eq!(
13741                handle.spawn_nonce(&spec.module_id).as_deref(),
13742                Some(nonce.as_str())
13743            );
13744        }
13745        drop(child);
13746    }
13747
13748    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13749    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
13750        probe(super::SpawnRole::Plain).await;
13751    }
13752
13753    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13754    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
13755        probe(super::SpawnRole::SwapCandidate).await;
13756    }
13757}
13758
13759#[cfg(all(test, target_os = "linux"))]
13760mod cgroup_containment_tests {
13761    use super::*;
13762    use subc_test_support::TestTempDir;
13763
13764    fn running(pid: u32) -> bool {
13765        // An orphan can remain a zombie until the container init reaps it.
13766        std::fs::read_to_string(format!("/proc/{pid}/stat"))
13767            .ok()
13768            .and_then(|stat| {
13769                stat.rsplit_once(") ")
13770                    .map(|(_, rest)| rest.starts_with('Z'))
13771            })
13772            .is_some_and(|zombie| !zombie)
13773    }
13774
13775    #[tokio::test]
13776    async fn linux_teardown_reaps_the_grandchild() {
13777        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
13778    }
13779
13780    #[tokio::test]
13781    async fn linux_shutdown_straggler_reaps_the_grandchild() {
13782        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
13783    }
13784
13785    async fn teardown_tree(test_name: &str, shutdown: bool) {
13786        let dir = TestTempDir::new(test_name);
13787        let root = PathBuf::from(format!(
13788            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
13789            std::process::id(),
13790            unix_ms_now()
13791        ));
13792        if let Err(error) = std::fs::create_dir(&root) {
13793            assert!(
13794                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13795                "required cgroup test cannot execute: {error}"
13796            );
13797            eprintln!(
13798                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
13799                root.display()
13800            );
13801            return;
13802        }
13803        let placement = subc_cgroup::prepare_at(&root)
13804            .expect("prepare isolated kernel cgroup")
13805            .expect("isolated cgroup is delegated");
13806        let module_id = "tree-teardown";
13807        let module = placement
13808            .module_path(module_id)
13809            .expect("create isolated module cgroup");
13810        if !module.join("cgroup.kill").exists() {
13811            std::fs::remove_dir(&module).unwrap();
13812            std::fs::remove_dir(root.join("subc-modules")).unwrap();
13813            std::fs::remove_dir(&root).unwrap();
13814            assert!(
13815                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13816                "required cgroup.kill interface unavailable"
13817            );
13818            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
13819            return;
13820        }
13821        let supervisor = Supervisor::new_for_test(
13822            Arc::new(Registry::default()),
13823            RestartPolicy::new(3, Duration::ZERO),
13824        )
13825        .with_cgroup_placement(Some(placement));
13826        let mut runtime = supervisor.runtime_config();
13827        runtime.child_roster = runtime
13828            .child_roster
13829            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
13830        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13831        let pid_file = dir.join("grandchild.pid");
13832        let spec = ModuleSpec {
13833            module_id: module_id.to_string(),
13834            program: PathBuf::from("/bin/sh"),
13835            args: vec![
13836                "-c".into(),
13837                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
13838                "fixture".into(),
13839                pid_file.display().to_string(),
13840            ],
13841            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
13842                .into_iter()
13843                .map(|key| (key.to_string(), dir.display().to_string()))
13844                .collect(),
13845            reserved: false,
13846            reserved_prefixes: Vec::new(),
13847            protocol: ModuleProtocol::None,
13848            overlap: Default::default(),
13849        };
13850        let child =
13851            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
13852        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
13853        let grandchild: u32 = loop {
13854            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
13855                if let Ok(pid) = contents.trim().parse() {
13856                    break pid;
13857                }
13858            }
13859            assert!(
13860                tokio::time::Instant::now() < deadline,
13861                "grandchild pid was not recorded"
13862            );
13863            tokio::time::sleep(Duration::from_millis(10)).await;
13864        };
13865        assert!(
13866            running(grandchild),
13867            "grandchild must be alive before teardown"
13868        );
13869        if shutdown {
13870            let mut child = child;
13871            crate::child_roster::end_children_for_daemon_shutdown(
13872                &runtime.child_roster,
13873                false,
13874                std::future::pending(),
13875            )
13876            .await;
13877            child.wait().await.expect("reap shutdown straggler");
13878        } else {
13879            drain_child_to_state(
13880                module_id,
13881                ModuleProtocol::None,
13882                StopNotice::NotSent,
13883                &Registry::default(),
13884                None,
13885                &snapshot,
13886                &runtime.terminal_ring,
13887                &SpawnEventFeed::default(),
13888                child,
13889                Duration::from_millis(100),
13890                ModuleState::Stopped,
13891                Some(false),
13892            )
13893            .await
13894            .expect("real supervisor teardown");
13895        }
13896        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
13897        while running(grandchild) && tokio::time::Instant::now() < deadline {
13898            tokio::time::sleep(Duration::from_millis(10)).await;
13899        }
13900        let survived = running(grandchild);
13901        // Kill a surviving grandchild so a failed test does not leave it behind.
13902        if survived {
13903            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
13904            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
13905            tokio::time::sleep(Duration::from_millis(100)).await;
13906        }
13907        if module.exists() {
13908            std::fs::remove_dir(&module).expect("remove empty module cgroup");
13909        }
13910        std::fs::remove_dir(root.join("subc-modules")).unwrap();
13911        std::fs::remove_dir(&root).unwrap();
13912        assert!(
13913            !survived,
13914            "grandchild {grandchild} outlived module teardown"
13915        );
13916        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
13917    }
13918}