Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture")
134}
135
136#[cfg(target_os = "macos")]
137fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
138    use std::io::Read;
139    let mut probe = std::process::Command::new(path)
140        .args(["__disclaim-exec", "--probe"])
141        .stdin(Stdio::null())
142        .stdout(Stdio::piped())
143        .stderr(Stdio::piped())
144        .spawn()
145        .map_err(|error| {
146            format!(
147                "privacy trampoline probe failed for {}: {error}",
148                path.display()
149            )
150        })?;
151    let deadline = std::time::Instant::now() + Duration::from_secs(5);
152    let status = loop {
153        match probe.try_wait() {
154            Ok(Some(status)) => break status,
155            Ok(None) if std::time::Instant::now() < deadline => {
156                std::thread::sleep(Duration::from_millis(5))
157            }
158            result => {
159                let _ = probe.kill();
160                let _ = probe.wait();
161                return Err(format!(
162                    "privacy trampoline probe failed or timed out for {}: {result:?}",
163                    path.display()
164                ));
165            }
166        }
167    };
168    let mut answer = String::new();
169    if let Some(stdout) = probe.stdout.take() {
170        let _ = stdout.take(256).read_to_string(&mut answer);
171    }
172    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
173        return Ok(());
174    }
175    let mut diagnostic = String::new();
176    if let Some(stderr) = probe.stderr.take() {
177        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
178    }
179    let cause = diagnostic
180        .trim()
181        .strip_prefix("ck-subc: own privacy identity refused: ")
182        .unwrap_or("binary does not implement the privacy trampoline protocol");
183    Err(format!(
184        "{cause}: probe of {} exited {status}",
185        path.display()
186    ))
187}
188
189#[cfg(target_os = "macos")]
190fn privacy_command(
191    spec: &ModuleSpec,
192    roster: &ChildRoster,
193) -> Result<
194    (
195        Command,
196        Option<PrivacyExec>,
197        subc_os::privacy_identity::ExecAcknowledgement,
198    ),
199    SuperviseError,
200> {
201    let failure = |cause: String| {
202        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
203        SuperviseError::Spawn {
204            program: spec.program.clone(),
205            source: io::Error::other(cause),
206            cgroup_path: None,
207        }
208    };
209    let trampoline = roster.privacy_trampoline().map_err(failure)?;
210    // Resolve PATH with the same environment the Command will receive. For
211    // scripts retain the existing orphan-identity rule: the kernel chooses
212    // the interpreter, and its observed image is the one recorded. Do not
213    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
214    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
215        let path = spec
216            .env
217            .iter()
218            .find(|(key, _)| key == "PATH")
219            .map(|(_, value)| std::ffi::OsString::from(value))
220            .or_else(|| std::env::var_os("PATH"))
221            .unwrap_or_else(|| "/usr/bin:/bin".into());
222        std::env::split_paths(&path)
223            .map(|dir| dir.join(&spec.program))
224            .find(|path| path.is_file())
225            .unwrap_or_else(|| spec.program.clone())
226    } else {
227        spec.program.clone()
228    };
229    let expected = subc_os::file_identity(&program);
230    let trampoline_image = subc_os::file_identity(&trampoline);
231    let script = {
232        use std::io::Read;
233        let mut prefix = [0u8; 2];
234        std::fs::File::open(&program)
235            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
236    };
237    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
238        return Err(failure(
239            "privacy identity module executable is missing or is the trampoline itself".to_string(),
240        ));
241    }
242    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
243        .map_err(|error| failure(error.to_string()))?;
244    let reader =
245        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
246    let mut command = Command::new(&trampoline);
247    command
248        .arg("__disclaim-exec")
249        .arg(ack.fd().to_string())
250        .arg(&program);
251    ack.install(command.as_std_mut());
252    Ok((
253        command,
254        Some(PrivacyExec {
255            reader,
256            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
257            expected,
258            trampoline: trampoline_image,
259            script,
260            module_id: spec.module_id.clone(),
261        }),
262        ack,
263    ))
264}
265
266struct SupervisedChild {
267    child: Child,
268    #[cfg(target_os = "macos")]
269    privacy_exec: Option<PrivacyExec>,
270    /// Refusal before the module image was accepted, retained for terminal records.
271    spawn_failure: Option<String>,
272    /// The protocol this process was launched with. A reload can store a new
273    /// launch spec with a different protocol, but that takes effect only at the
274    /// next spawn, so this process keeps being handled by the protocol it
275    /// actually speaks.
276    protocol: ModuleProtocol,
277    /// This process's cgroup name: a bounded module/slot label followed by a
278    /// spawn suffix unique to this process (when cgroup placement is on). A
279    /// retired process in a slot may still be draining when a later one is
280    /// spawned into that slot, so the suffix keeps the later process out of
281    /// the retired one's cgroup, which is the domain a kill applies to.
282    #[cfg(target_os = "linux")]
283    module_id: String,
284    #[cfg(target_os = "linux")]
285    cgroup_placement: Option<subc_cgroup::Placement>,
286    /// The job that contains this child and every process it spawns (issue #109).
287    ///
288    /// Dropping this handle is what reaps a surviving tree when no supervisor
289    /// code runs — a daemon crash — because the job carries
290    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
291    ///
292    /// That limit is not crash-only, and the difference is worth knowing: a
293    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
294    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
295    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
296    /// module at once. Before this change they survived that, saw EOF on the
297    /// control socket, and ran their own teardown; Unix keeps that path
298    /// deliberately, so a module can seal a WAL or close a capture rather than
299    /// be killed mid-write. So this trades graceful teardown on every Windows
300    /// daemon stop for containment on a crash, which is the right way round
301    /// today: orphaned GPU workers are a reported, recurring problem, and the
302    /// modules that write most heavily do not run on Windows.
303    ///
304    /// The fix is a real Windows stop path — the daemon draining before it
305    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
306    /// reaches only what the drain left behind, which is what it should reach.
307    #[cfg(windows)]
308    job: Option<subc_jobobject::JobObject>,
309    stdout_pump: Option<JoinHandle<()>>,
310    stderr_pump: Option<StderrPump>,
311    stderr_ring: Arc<Mutex<StderrRing>>,
312    spawned_at_ms: u64,
313    spawned_from: PathBuf,
314    spawned_file_identity: Option<SpawnedFileIdentity>,
315    process_start_time: Option<u64>,
316    process_identity: Option<ProcessIdentity>,
317    pid: u32,
318    /// This process's entry in the daemon's child roster, released when the
319    /// process is reaped or this handle is dropped.
320    roster_guard: Option<crate::child_roster::RosterGuard>,
321}
322
323impl SupervisedChild {
324    fn id(&self) -> Option<u32> {
325        Some(self.pid)
326    }
327
328    fn process_identity(&self) -> Option<ProcessIdentity> {
329        self.process_identity
330    }
331
332    async fn wait(&mut self) -> io::Result<ExitStatus> {
333        #[cfg(target_os = "macos")]
334        self.confirm_privacy_exec().await;
335        // The roster entry is NOT released here. A daemon shutdown waits for the
336        // roster to empty and then exits the process, so releasing at the reap
337        // let it exit before the exit handler wrote this child's terminal record
338        // (the stderr drain and snapshot update sit in between), and the
339        // shutdown's own `daemon_shutdown` record was intermittently lost. The
340        // caller releases it after recording the exit (`release_roster`), and
341        // dropping the handle releases it too.
342        let result = self.child.wait().await;
343        #[cfg(target_os = "linux")]
344        if result.is_ok() {
345            if let Some(placement) = self.cgroup_placement.as_ref() {
346                cleanup_reaped_cgroup(placement, &self.module_id).await;
347                // Keep ownership while awaiting kernel population changes: a
348                // drain timeout may cancel this wait and then escalate/reap.
349                self.cgroup_placement = None;
350            }
351        }
352        result
353    }
354
355    #[cfg(target_os = "macos")]
356    async fn confirm_privacy_exec(&mut self) {
357        let Some(pending) = &mut self.privacy_exec else {
358            return;
359        };
360        let result = tokio::time::timeout_at(pending.deadline, async {
361            let mut record = Vec::new();
362            loop {
363                let mut ready = pending.reader.readable().await?;
364                let read = ready.try_io(|reader| {
365                    use std::io::Read;
366                    let mut reader = reader.get_ref();
367                    let mut buffer = [0u8; 256];
368                    reader.read(&mut buffer).map(|count| (count, buffer))
369                });
370                match read {
371                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
372                    Ok(Ok((count, buffer))) => {
373                        if record.len() + count > 1024 {
374                            return Err(io::Error::other(
375                                "privacy exec refusal record is too long",
376                            ));
377                        }
378                        record.extend_from_slice(&buffer[..count]);
379                    }
380                    Ok(Err(error)) => return Err(error),
381                    Err(_) => continue,
382                }
383            }
384        })
385        .await;
386        // Keep the reader in self across await: select cancellation must not
387        // discard the handshake or reset its original five-second deadline.
388        let pending = self.privacy_exec.as_ref().expect("pending exec");
389        let cause = match result {
390            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
391            Ok(Err(error)) => Some(format!(
392                "privacy identity exec acknowledgement failed: {error}"
393            )),
394            Ok(Ok(record)) if !record.is_empty() => Some(
395                std::str::from_utf8(&record)
396                    .ok()
397                    .and_then(|record| {
398                        record
399                            .trim()
400                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
401                    })
402                    .filter(|cause| !cause.is_empty())
403                    .unwrap_or("invalid privacy exec refusal record")
404                    .to_string(),
405            ),
406            Ok(Ok(_)) => match self.child.try_wait() {
407                // Empty EOF is the exec acknowledgement. A real module may exit
408                // immediately, including with a reserved trampoline status; no
409                // image is admitted, and its ordinary exit contract stays intact.
410                Ok(Some(_status)) => None,
411                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
412                Ok(None) => {
413                    let image = observe_spawned_image(self.pid);
414                    if let Some(image) = image.filter(|image| {
415                        image.executable.is_some()
416                            && image.executable != pending.trampoline
417                            && (image.executable == pending.expected || pending.script)
418                    }) {
419                        if let Some(guard) = &self.roster_guard {
420                            guard.confirm_executable(image);
421                        }
422                        info!(module_id = %pending.module_id, pid = self.pid,
423                            "module spawned with own privacy identity (responsibility disclaimed)");
424                        None
425                    } else if image.is_none()
426                        || image.is_some_and(|image| image.executable.is_none())
427                    {
428                        // A process can exit between try_wait and the kernel
429                        // image read. Empty EOF already acknowledged exec, so
430                        // preserve that module's ordinary exit rather than
431                        // mislabel a disappearing image as trampoline refusal.
432                        // Keep pending in self across await so cancellation does
433                        // not discard validation or reset its original deadline.
434                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
435                            Ok(Ok(_status)) => None,
436                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
437                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
438                        }
439                    } else {
440                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
441                    }
442                }
443            },
444        };
445        let pending = self.privacy_exec.take().expect("pending exec");
446        if let Some(cause) = cause {
447            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
448            self.spawn_failure = Some(cause);
449            // No image is admitted on failure. Reach the entire fresh process
450            // group, including a module which spawned a helper before refusal.
451            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
452                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
453            }
454            let _ = self.child.start_kill();
455        }
456    }
457
458    /// Releases this child's daemon-shutdown roster entry once its exit has
459    /// been recorded. The pid is already reaped and free for reuse, so the
460    /// entry must not outlive the record any longer than that.
461    fn release_roster(&mut self) {
462        self.roster_guard = None;
463    }
464
465    /// Kill the child and, where containment is available, its process tree.
466    ///
467    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
468    /// helper process leaked the helper — the Synapse embedding module's CUDA
469    /// worker holds the GPU allocation, so the leak cost VRAM until the next
470    /// restart of something else. Terminating the job reaches grandchildren that
471    /// a tree walk cannot, including one whose parent has already exited and
472    /// been reparented away.
473    ///
474    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
475    /// direct-child kill still decides the outcome, so containment can never
476    /// change whether a module is reported as stopped.
477    fn start_kill(&mut self) -> io::Result<()> {
478        #[cfg(windows)]
479        if let Some(job) = &self.job {
480            if let Err(error) = job.terminate() {
481                debug!(
482                    error = %error,
483                    "job termination failed; the direct-child kill still owns the outcome"
484                );
485            }
486        }
487        #[cfg(target_os = "linux")]
488        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
489        self.child.start_kill()
490    }
491
492    async fn drain_stderr(&mut self, module_id: &str) {
493        if let Some(mut pump) = self.stdout_pump.take() {
494            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
495                Ok(Ok(())) => {}
496                Ok(Err(error)) => {
497                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
498                }
499                Err(_) => {
500                    pump.abort();
501                    warn!(
502                        module_id,
503                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
504                        "stdout pump did not drain before restart; stopped it before the next process"
505                    );
506                }
507            }
508        }
509
510        let Some(pump) = self.stderr_pump.take() else {
511            return;
512        };
513        settle_stderr_pump(
514            module_id,
515            &self.stderr_ring,
516            pump,
517            STDERR_PUMP_DRAIN_TIMEOUT,
518        )
519        .await;
520    }
521}
522
523/// The reader task for one process's stderr, with the ring generation its
524/// lines are attributed to.
525struct StderrPump {
526    task: JoinHandle<()>,
527    generation: u64,
528}
529
530/// Retire an exited process's stderr reader and wait up to `bound` for it to
531/// reach EOF. A reader still running at the bound is detached, not stopped: it
532/// keeps filling the exited process's section of the ring until its pipe
533/// closes, and the tail reads `Incomplete` until then. See
534/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
535async fn settle_stderr_pump(
536    module_id: &str,
537    ring: &Arc<Mutex<StderrRing>>,
538    pump: StderrPump,
539    bound: Duration,
540) {
541    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
542    let StderrPump {
543        mut task,
544        generation,
545    } = pump;
546    lock().retire_pump(generation);
547    match timeout(bound, &mut task).await {
548        Ok(Ok(())) => {}
549        Ok(Err(err)) => {
550            let mut ring = lock();
551            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
552            ring.finish_pump(generation);
553            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
554        }
555        Err(_) => {
556            // Dropping the handle detaches the task; it ends at EOF on its pipe.
557            drop(task);
558            lock().mark_pump_late(
559                generation,
560                format!(
561                    "stderr of the exited process had not reached EOF {bound:?} after it was \
562                     retired (a descendant may still hold the pipe open); lines it still \
563                     writes are kept in that process's section"
564                ),
565            );
566            warn!(
567                module_id,
568                waited = ?bound,
569                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
570            );
571        }
572    }
573}
574
575fn registration_release_events() -> &'static watch::Sender<u64> {
576    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
577    EVENTS.get_or_init(|| {
578        let (sender, _receiver) = watch::channel(0);
579        sender
580    })
581}
582
583pub(crate) fn notify_registration_release() {
584    let events = registration_release_events();
585    let next_generation = (*events.borrow()).wrapping_add(1);
586    events.send_replace(next_generation);
587}
588
589/// How to launch one singleton module process.
590#[derive(Debug, Clone, PartialEq, Eq)]
591pub struct ModuleSpec {
592    pub module_id: String,
593    pub program: PathBuf,
594    pub args: Vec<String>,
595    pub env: Vec<(String, String)>,
596    /// When true this is a reserved module: each spawn gets a fresh one-time launch
597    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
598    /// process can register this module_id (a security-boundary module like the
599    /// credential vault must not be impersonable while it is down/restarting).
600    pub reserved: bool,
601    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
602    /// Prefixes come from daemon config and must end in `:` before they reach the
603    /// supervisor; the owner module's current spawn nonce authorizes claims under
604    /// each prefix.
605    pub reserved_prefixes: Vec<String>,
606    /// The wire protocol this module speaks, as DECLARED in daemon config.
607    ///
608    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
609    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
610    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
611    /// and NO launch nonce, and a clean exit the daemon did not request is
612    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
613    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
614    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
615    /// because a process ignores an environment variable it does not read.
616    ///
617    /// The argument is the part that cannot be "harmless to a process that
618    /// ignores it": a stock binary exits on an unknown flag before it listens
619    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
620    /// first conformance run against this mode found it. The nonce is withheld
621    /// because a process that will never present it gains nothing from holding
622    /// it, and a secret in the environment of a process that does not need it is
623    /// a leak surface for no benefit.
624    pub protocol: ModuleProtocol,
625    /// Whether two processes of this module may run at once, which is what a
626    /// blue/green swap does for the length of its overlap. Declared in daemon
627    /// config because the daemon must be able to answer it while the module is
628    /// down, and so a module cannot talk itself into it after registering.
629    pub overlap: ModuleOverlap,
630}
631
632/// Whether a module tolerates a second process of itself running alongside.
633///
634/// Most modules are single-writer on their store (a WAL, a capture log, a
635/// resident index behind a writer barrier), and two processes on one store
636/// corrupt it. So a swap, which overlaps the old and new process by design,
637/// is refused unless the module's config opts in with `overlap: "safe"`.
638#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
639pub enum ModuleOverlap {
640    /// Never run two processes of this module at once. The default.
641    #[default]
642    Exclusive,
643    /// The module has said a second process of itself is harmless for the
644    /// length of a swap.
645    ///
646    /// Declare it only if a second instance can run for a few seconds without
647    /// touching ANY single-writer store: every database, WAL, index, projector
648    /// and scheduled job the module owns. A lease on part of that state is not
649    /// enough. broca's session lease guards WAL appends while its run index, its
650    /// store projector and its archive fold timer (which unlinks live WAL files)
651    /// stay single-writer, so broca is exclusive despite holding a lease. The
652    /// refusal only fires after this has been decided, so the decision is the
653    /// check.
654    Safe,
655}
656
657impl ModuleOverlap {
658    pub fn as_str(self) -> &'static str {
659        match self {
660            Self::Exclusive => "exclusive",
661            Self::Safe => "safe",
662        }
663    }
664}
665
666/// Environment variable telling a spawned module which case it was started
667/// for, before it sends HELLO. Only a swap candidate carries it, as
668/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
669///
670/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
671/// longer because nobody waits on it, while a plain restart must flip ready
672/// quickly because callers see `module_warming` until it does. Absence means
673/// plain restart, the safe reading. The daemon trusts nothing about it; the
674/// candidate is proven by its launch nonce at HELLO.
675pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
676/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
677pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
678/// How long a swap waits for its candidate to register and declare itself
679/// ready when the operator does not say. A module warming as a swap candidate
680/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
681/// daemon allows that plus time to start the process and send HELLO.
682pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
683
684/// Bounded restart policy for crash exits.
685///
686/// `max_restarts` is the number of replacement processes allowed after the
687/// initial spawn WITHIN `window`. After that many crash restarts inside one
688/// window the module enters [`ModuleState::Failed`] and the supervisor stops
689/// the crash loop.
690///
691/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
692/// and that only survived because crashes were rare: a module that crashed
693/// three times across a week was disabled forever by crashes that had nothing
694/// to do with each other. That stopped being survivable once modules began
695/// exiting non-zero whenever the daemon's connection to them drops, because
696/// then every daemon-side connection drop spends a unit of the same budget and
697/// one flappy hour permanently stops a healthy module. Restarts older than
698/// `window` release their slot, so a module that crashed twice yesterday has a
699/// full budget today, while a genuine crash loop -- which is fast by
700/// definition -- still reaches the cap and stops.
701#[derive(Debug, Clone, Copy, PartialEq, Eq)]
702pub struct RestartPolicy {
703    pub max_restarts: u32,
704    /// Base delay before a crash replacement. The actual delay escalates with
705    /// the number of recent crash replacements and is capped by `max_backoff`.
706    pub backoff: Duration,
707    /// Maximum delay before a crash replacement.
708    pub max_backoff: Duration,
709    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
710    /// budget effectively infinite (nothing is ever in-window), which is why
711    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
712    pub window: Duration,
713}
714
715impl RestartPolicy {
716    /// A policy with the default crash window. Callers that care about the
717    /// window say so with [`Self::with_window`]; the ones that do not are
718    /// asking for the standard rate limit, not for no limit.
719    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
720        Self {
721            max_restarts,
722            backoff,
723            max_backoff: DEFAULT_MAX_BACKOFF,
724            window: DEFAULT_RESTART_WINDOW,
725        }
726    }
727
728    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
729        self.max_backoff = max_backoff;
730        self
731    }
732
733    pub fn with_window(mut self, window: Duration) -> Self {
734        self.window = window;
735        self
736    }
737
738    /// Calculate the capped exponential delay for the next crash replacement.
739    /// `restart_in_window` is zero for the first replacement after an operator
740    /// action (restart, reload, re-enable) cleared the crash ring, or after all
741    /// older crash replacements have aged out of the window.
742    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
743        if self.backoff.is_zero() || self.max_backoff.is_zero() {
744            return Duration::ZERO;
745        }
746
747        let mut delay = self.backoff;
748        for _ in 0..restart_in_window {
749            if delay >= self.max_backoff {
750                return self.max_backoff;
751            }
752            delay = delay
753                .checked_mul(10)
754                .unwrap_or(self.max_backoff)
755                .min(self.max_backoff);
756        }
757        delay.min(self.max_backoff)
758    }
759
760    /// The one sentence that explains a budget-exhausted stop, used for both the
761    /// log line and the terminal record so the two cannot drift. It names the
762    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
763    /// exactly what this budget is not.
764    fn budget_exhausted_detail(&self) -> String {
765        format!(
766            "crash budget exhausted: max_restarts={} within window_secs={}",
767            self.max_restarts,
768            self.window.as_secs()
769        )
770    }
771}
772
773impl Default for RestartPolicy {
774    fn default() -> Self {
775        Self {
776            max_restarts: DEFAULT_MAX_RESTARTS,
777            backoff: DEFAULT_BACKOFF,
778            max_backoff: DEFAULT_MAX_BACKOFF,
779            window: DEFAULT_RESTART_WINDOW,
780        }
781    }
782}
783
784#[derive(Debug, Clone, Copy, PartialEq, Eq)]
785struct CrashRestartSchedule {
786    restart_in_window: u32,
787    delay: Duration,
788}
789
790/// Whether the daemon itself will bring this module back after the exit being
791/// handled: it is enabled AND its in-window crash restarts are below the cap.
792///
793/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
794/// the window are dropped here rather than by a timer, so the count is right
795/// the moment somebody asks and no bookkeeping runs for idle modules.
796fn daemon_will_restart(
797    state: &mut SupervisorSnapshot,
798    policy: &RestartPolicy,
799    now: Instant,
800) -> bool {
801    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
802}
803
804const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
805const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
806const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
807const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
808
809#[derive(Debug, Clone, Copy, PartialEq, Eq)]
810pub enum HealthAction {
811    Report,
812    Restart,
813    Alert,
814}
815
816impl fmt::Display for HealthAction {
817    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
818        f.write_str(match self {
819            Self::Report => "report",
820            Self::Restart => "restart",
821            Self::Alert => "alert",
822        })
823    }
824}
825
826#[derive(Debug, Clone, PartialEq, Eq)]
827pub struct HealthConfig {
828    /// Optional loopback HTTP endpoint for a managed non-wire process.
829    /// Changing it applies live on rescan; the process protocol changes only
830    /// at its next spawn.
831    pub http: Option<String>,
832    pub cadence: Duration,
833    pub deadline: Duration,
834    pub failure_threshold: u32,
835    pub on_degraded: HealthAction,
836    pub on_failing: HealthAction,
837    pub critical: bool,
838}
839
840impl Default for HealthConfig {
841    fn default() -> Self {
842        Self {
843            http: None,
844            cadence: DEFAULT_HEALTH_CADENCE,
845            deadline: DEFAULT_HEALTH_DEADLINE,
846            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
847            on_degraded: HealthAction::Report,
848            on_failing: HealthAction::Report,
849            critical: false,
850        }
851    }
852}
853
854/// The supervisor's view of one module's health, relayed to clients over
855/// channel-0 and rendered by `ck health`.
856///
857/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
858/// stated here rather than only at the wire type a consumer reads. A reader can
859/// look up what `None` means; only a writer can silently change it, and the
860/// writer has no reason to go looking at a downstream contract before editing.
861///
862/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
863/// back to `None` on re-registration precisely so a respawned module does not
864/// carry its predecessor's timestamp — so an old value and an absent one call for
865/// opposite readings, and anything that defaulted this to a number would make a
866/// never-probed module indistinguishable from one probed at the epoch.
867///
868/// `detail` and `metrics` are `None` when the module published none on this
869/// probe, which does not mean it reported nothing wrong — it is also the shape
870/// when the probe never reached it. `last_probe_ms` is what separates those.
871#[derive(Debug, Clone, PartialEq)]
872pub struct ModuleHealthStatus {
873    pub status: SupervisorHealthStatus,
874    pub last_probe_ms: Option<u64>,
875    pub detail: Option<String>,
876    pub metrics: Option<Value>,
877    pub consecutive_failures: u32,
878    /// Number of replies received after a recurring health probe's deadline.
879    /// Unlike a timeout, every increment proves the module was alive.
880    pub late_answer_count: u64,
881    /// End-to-end latency of the newest late reply, measured from probe start.
882    pub last_late_answer_latency_ms: Option<u64>,
883    pub last_action: Option<String>,
884    /// Set together with `last_action`; the pair moves as one, and both being
885    /// absent means no escalation has ever been taken rather than that the last
886    /// one succeeded.
887    pub last_action_ms: Option<u64>,
888}
889
890impl Default for ModuleHealthStatus {
891    fn default() -> Self {
892        Self {
893            status: SupervisorHealthStatus::Unknown,
894            last_probe_ms: None,
895            detail: None,
896            metrics: None,
897            consecutive_failures: 0,
898            late_answer_count: 0,
899            last_late_answer_latency_ms: None,
900            last_action: None,
901            last_action_ms: None,
902        }
903    }
904}
905
906/// Typed lifecycle state for a supervised module.
907#[derive(Debug, Clone, Copy, PartialEq, Eq)]
908pub enum ModuleState {
909    Starting,
910    Running,
911    Unresponsive,
912    Restarting,
913    Draining,
914    Stopped,
915    Failed,
916    Disabled,
917}
918
919impl fmt::Display for ModuleState {
920    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
921        f.write_str(match self {
922            Self::Starting => "starting",
923            Self::Running => "running",
924            Self::Unresponsive => "unresponsive",
925            Self::Restarting => "restarting",
926            Self::Draining => "draining",
927            Self::Stopped => "stopped",
928            Self::Failed => "failed",
929            Self::Disabled => "disabled",
930        })
931    }
932}
933
934/// Supervisor classification of a child-process exit.
935#[derive(Debug, Clone, Copy, PartialEq, Eq)]
936pub enum ExitKind {
937    Clean,
938    Crash,
939    DeliberateSeverance,
940}
941
942impl From<ExitKind> for TerminalExitKind {
943    fn from(kind: ExitKind) -> Self {
944        match kind {
945            ExitKind::Clean => Self::Clean,
946            ExitKind::Crash => Self::Crash,
947            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
948        }
949    }
950}
951
952/// Exact process identity retained when a supervised module registers its
953/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub(crate) struct ProcessIdentity {
956    pub(crate) pid: u32,
957    pub(crate) start_time: u64,
958}
959
960/// Last observed child exit, if any.
961#[derive(Debug, Clone, PartialEq, Eq)]
962pub struct ExitReport {
963    pub kind: ExitKind,
964    pub code: Option<i32>,
965    pub signal: Option<i32>,
966    pub at_ms: u64,
967}
968
969/// Point-in-time module status answerable by subc without forwarding to the
970/// module process.
971#[derive(Debug, Clone, PartialEq)]
972pub struct ModuleStatus {
973    pub module_id: String,
974    pub state: ModuleState,
975    pub enabled: bool,
976    pub process_alive: bool,
977    pub registration_active: bool,
978    /// The module's declared wire protocol, carried beside `live` because it is
979    /// what makes `live` readable: the two fields answer one question together.
980    /// While a process is alive this is its launch declaration, not a later
981    /// pending-reload edit. When down it is the configured next launch protocol.
982    pub protocol: ModuleProtocol,
983    /// Whether the module is serving, under the strongest definition the daemon
984    /// can assert for its protocol.
985    ///
986    /// A subc module must also be REGISTERED: its process being alive says
987    /// nothing about whether it can take a request. A `protocol: "none"` module
988    /// never registers, so that term is dropped and this falls back to "enabled,
989    /// running, and the process the daemon launched is alive" -- which is all
990    /// the daemon observes about a process that speaks no subc wire. It stays a
991    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
992    /// rather than printing it bare.
993    pub live: bool,
994    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
995    /// restarts have already released their slot, so this count can go down
996    /// without anybody touching the module.
997    pub restart_count: u32,
998    /// Replacement processes spawned over this module's entire supervisor lifetime;
999    /// unlike `restart_count`, this value is never reset by an operator action
1000    /// and never falls out of a window.
1001    pub lifetime_restarts: u32,
1002    pub spawn_generation: u64,
1003    /// The budget `restart_count` is spent against. Carried alongside the count
1004    /// because the count alone does not say how close the module is to being
1005    /// disabled, and reporting one without the other is what makes an
1006    /// about-to-be-retired module look ordinary.
1007    pub max_restarts: u32,
1008    /// The span `restart_count` is counted over. Carried with the pair above for
1009    /// the same reason they are carried together: "2 of 3" means one thing for a
1010    /// ten-minute window and something else entirely for a lifetime.
1011    pub restart_window: Duration,
1012    /// Effective drain and restart timing policy used by this running module.
1013    /// These values are carried together with the restart budget so status
1014    /// readers can compare configured intent with what the supervisor applied.
1015    pub drain_timeout: Duration,
1016    pub restart_backoff: Duration,
1017    pub restart_max_backoff: Duration,
1018    pub pid: Option<u32>,
1019    pub spawned_at_ms: Option<u64>,
1020    pub spawned_from: Option<PathBuf>,
1021    pub process_start_time: Option<u64>,
1022    pub last_exit: Option<ExitReport>,
1023    pub health: ModuleHealthStatus,
1024}
1025
1026#[derive(Debug, Clone, PartialEq)]
1027struct SupervisorSnapshot {
1028    state: ModuleState,
1029    enabled: bool,
1030    process_alive: bool,
1031    spawned_protocol: Option<ModuleProtocol>,
1032    spawn_failure: Option<String>,
1033    /// When each crash restart was spent, oldest first. This IS the crash
1034    /// budget: its in-window length is the count an operator sees and the count
1035    /// the restart decision is made against, so there is no second counter that
1036    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1037    /// operator actions that used to zero the old lifetime counter.
1038    crash_restarts: VecDeque<Instant>,
1039    lifetime_restarts: u32,
1040    /// Successful child spawns in this daemon incarnation.
1041    ///
1042    /// `lifetime_restarts` was considered and rejected: it starts at zero
1043    /// (line 640), successful initial/operator spawns in `set_running` do not
1044    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1045    /// increments before a successful replacement exists (lines 604, 3846,
1046    /// and 3921), so a failed spawn can consume it. This counter moves only
1047    /// when a live PID is accepted below.
1048    spawn_generation: u64,
1049    pid: Option<u32>,
1050    /// Last reaped child, retained after current process facts are cleared.
1051    reaped_pid: Option<u32>,
1052    /// Whether the command-serving supervision loop has a scheduled respawn.
1053    respawn_pending: bool,
1054    /// A second restart is waiting for the replacement already scheduled.
1055    coalesced_restart_pending: bool,
1056    spawned_at_ms: Option<u64>,
1057    spawned_from: Option<PathBuf>,
1058    spawned_file_identity: Option<SpawnedFileIdentity>,
1059    process_start_time: Option<u64>,
1060    deliberate_severance: Option<ProcessIdentity>,
1061    last_exit: Option<ExitReport>,
1062    /// Diagnostic attached to the next drain's terminal record, if any.
1063    drain_disposition_detail: Option<String>,
1064    health: ModuleHealthStatus,
1065    /// Whether the current process was started as a swap candidate and so
1066    /// lives in the module's alternate cgroup. The next swap's candidate takes
1067    /// the other one, so the two processes of a swap never share a cgroup. A
1068    /// plain spawn always uses the primary cgroup.
1069    in_alternate_slot: bool,
1070    /// Whether the current `Draining` state ends in a replacement process
1071    /// (restart, reload, health restart) rather than a stop. Only meaningful
1072    /// while `state` is `Draining`; every entry into that state rewrites it.
1073    /// It is what lets route.open answer the retryable `module_reloading` to a
1074    /// consumer that reaches a still-registered process mid-restart, instead of
1075    /// the `supervisor_not_live` a stop or disable deserves.
1076    draining_to_replace: bool,
1077    /// Whether a configuration update has been applied since the current
1078    /// process was spawned, so that process runs an older spec than the one
1079    /// the supervisor now holds. A queued restart is only coalesced into a
1080    /// fresher process when this is false: a restart requested to pick up a
1081    /// new configuration must not be satisfied by a process that predates it.
1082    configuration_updated_since_spawn: bool,
1083}
1084
1085impl SupervisorSnapshot {
1086    fn starting() -> Self {
1087        Self::new(ModuleState::Starting, true)
1088    }
1089
1090    fn disabled() -> Self {
1091        Self::new(ModuleState::Disabled, false)
1092    }
1093
1094    fn failed() -> Self {
1095        Self::new(ModuleState::Failed, true)
1096    }
1097
1098    /// Crash restarts still inside `window`, having dropped the ones that are
1099    /// not. Pruning on read is what makes the budget a rate: an instant older
1100    /// than the window stops holding a slot the moment anybody counts.
1101    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1102        while let Some(oldest) = self.crash_restarts.front() {
1103            if now.duration_since(*oldest) > window {
1104                self.crash_restarts.pop_front();
1105            } else {
1106                break;
1107            }
1108        }
1109        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1110    }
1111
1112    /// Spend one unit of the crash budget and record the restart in the ledger.
1113    ///
1114    /// The ring is bounded by the cap because more than `max_restarts` in-window
1115    /// instants can never be reached (the caller refuses the restart first), so
1116    /// anything beyond that is an unbounded queue waiting to happen.
1117    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1118        self.crash_restarts.push_back(now);
1119        while self.crash_restarts.len() > policy.max_restarts as usize {
1120            self.crash_restarts.pop_front();
1121        }
1122        self.lifetime_restarts += 1;
1123    }
1124
1125    /// Reserve one crash-restart slot and calculate the delay before respawning.
1126    /// The count is captured before recording this restart, so the first retry
1127    /// uses the base delay and each later in-window retry escalates once.
1128    fn next_crash_restart(
1129        &mut self,
1130        policy: &RestartPolicy,
1131        now: Instant,
1132    ) -> Option<CrashRestartSchedule> {
1133        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1134        if restart_in_window >= policy.max_restarts {
1135            return None;
1136        }
1137        self.record_crash_restart(policy, now);
1138        Some(CrashRestartSchedule {
1139            restart_in_window,
1140            delay: policy.delay_for_restart(restart_in_window),
1141        })
1142    }
1143
1144    /// Give the module its full budget back, as an operator restart, reload, or
1145    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1146    /// ledger of what actually happened, and an operator action does not unmake
1147    /// the crashes.
1148    fn clear_crash_restarts(&mut self) {
1149        self.crash_restarts.clear();
1150    }
1151
1152    fn new(state: ModuleState, enabled: bool) -> Self {
1153        Self {
1154            state,
1155            enabled,
1156            process_alive: false,
1157            spawned_protocol: None,
1158            spawn_failure: None,
1159            crash_restarts: VecDeque::new(),
1160            lifetime_restarts: 0,
1161            spawn_generation: 0,
1162            pid: None,
1163            reaped_pid: None,
1164            respawn_pending: false,
1165            coalesced_restart_pending: false,
1166            spawned_at_ms: None,
1167            spawned_from: None,
1168            spawned_file_identity: None,
1169            process_start_time: None,
1170            deliberate_severance: None,
1171            last_exit: None,
1172            drain_disposition_detail: None,
1173            health: ModuleHealthStatus::default(),
1174            in_alternate_slot: false,
1175            draining_to_replace: false,
1176            configuration_updated_since_spawn: false,
1177        }
1178    }
1179}
1180
1181type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1182
1183type SpawnSubscriberKey = (ConnectionId, u64);
1184
1185#[derive(Debug)]
1186struct SpawnSubscriber {
1187    version: u8,
1188    frames: mpsc::Sender<Frame>,
1189    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1190    /// from which event. The full frame channel cannot carry that news, so it
1191    /// travels beside it; see `SpawnEventFeed::subscribe`.
1192    lagged: Option<oneshot::Sender<SpawnCursor>>,
1193}
1194
1195#[derive(Debug)]
1196struct SpawnEventState {
1197    daemon_incarnation: String,
1198    seq: u64,
1199    capacity: usize,
1200    live: HashMap<String, LiveSpawn>,
1201    generations: HashMap<String, u64>,
1202    events: VecDeque<SpawnEvent>,
1203    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1204}
1205
1206impl Default for SpawnEventState {
1207    fn default() -> Self {
1208        Self {
1209            daemon_incarnation: "unconfigured".to_string(),
1210            seq: 0,
1211            capacity: SPAWN_EVENT_RING_CAPACITY,
1212            live: HashMap::new(),
1213            generations: HashMap::new(),
1214            events: VecDeque::new(),
1215            subscribers: HashMap::new(),
1216        }
1217    }
1218}
1219
1220#[derive(Debug, Clone, Default)]
1221struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1222
1223#[derive(Debug, Clone, PartialEq, Eq)]
1224pub(crate) enum SpawnSubscribeRefusal {
1225    ForeignIncarnation { current: String },
1226    TooOld { oldest: SpawnCursor },
1227    Frame(String),
1228}
1229
1230impl SpawnEventFeed {
1231    fn configure_incarnation(&self, daemon_incarnation: String) {
1232        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1233        state.daemon_incarnation = daemon_incarnation;
1234        state.seq = 0;
1235        state.live.clear();
1236        state.generations.clear();
1237        state.events.clear();
1238        state.subscribers.clear();
1239    }
1240
1241    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1242        SpawnCursor {
1243            daemon_incarnation: state.daemon_incarnation.clone(),
1244            seq: state.seq,
1245        }
1246    }
1247
1248    fn snapshot(&self) -> SpawnSnapshot {
1249        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1250        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1251        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1252        SpawnSnapshot {
1253            cursor: Self::cursor(&state),
1254            ring_bound: state.capacity as u64,
1255            live,
1256        }
1257    }
1258
1259    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1260        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1261        let generation = state
1262            .generations
1263            .get(module_id)
1264            .copied()
1265            .unwrap_or(0)
1266            .checked_add(1)
1267            .expect("spawn generation exhausted");
1268        state.generations.insert(module_id.to_string(), generation);
1269        let live = LiveSpawn {
1270            module_id: module_id.to_string(),
1271            spawn_generation: generation,
1272            pid,
1273            spawned_at_ms,
1274        };
1275        state.live.insert(module_id.to_string(), live);
1276        Self::emit_locked(
1277            &mut state,
1278            SpawnEventKind::Spawned,
1279            module_id.to_string(),
1280            generation,
1281            pid,
1282            None,
1283            None,
1284        );
1285        generation
1286    }
1287
1288    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1289        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1290        let Some(live) = state.live.remove(module_id) else {
1291            warn!(
1292                module_id,
1293                "terminal record had no live spawn event identity"
1294            );
1295            return;
1296        };
1297        Self::emit_locked(
1298            &mut state,
1299            SpawnEventKind::Exited,
1300            module_id.to_string(),
1301            live.spawn_generation,
1302            live.pid,
1303            exit_code,
1304            exit_signal,
1305        );
1306    }
1307
1308    /// Report the exit of a process that a swap has already replaced.
1309    ///
1310    /// `emit_exited` removes the module's live entry, which after a swap's
1311    /// cutover describes the promoted candidate, not the old process now
1312    /// exiting. This emits the old generation's exit and leaves the live entry
1313    /// alone unless it still names that generation.
1314    fn emit_superseded_exited(
1315        &self,
1316        module_id: &str,
1317        spawn_generation: u64,
1318        pid: u32,
1319        exit_code: Option<i32>,
1320        exit_signal: Option<i32>,
1321    ) {
1322        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1323        if state
1324            .live
1325            .get(module_id)
1326            .is_some_and(|live| live.spawn_generation == spawn_generation)
1327        {
1328            state.live.remove(module_id);
1329        }
1330        Self::emit_locked(
1331            &mut state,
1332            SpawnEventKind::Exited,
1333            module_id.to_string(),
1334            spawn_generation,
1335            pid,
1336            exit_code,
1337            exit_signal,
1338        );
1339    }
1340
1341    #[allow(clippy::too_many_arguments)]
1342    fn emit_locked(
1343        state: &mut SpawnEventState,
1344        kind: SpawnEventKind,
1345        module_id: String,
1346        spawn_generation: u64,
1347        pid: u32,
1348        exit_code: Option<i32>,
1349        exit_signal: Option<i32>,
1350    ) {
1351        state.seq = state
1352            .seq
1353            .checked_add(1)
1354            .expect("spawn event sequence exhausted");
1355        let event = SpawnEvent {
1356            cursor: Self::cursor(state),
1357            kind,
1358            module_id,
1359            spawn_generation,
1360            pid,
1361            exit_code,
1362            exit_signal,
1363        };
1364        state.events.push_back(event.clone());
1365        while state.events.len() > state.capacity {
1366            state.events.pop_front();
1367        }
1368        let body = match serde_json::to_vec(&event) {
1369            Ok(body) => body,
1370            Err(error) => {
1371                error!(%error, "failed to serialize supervisor spawn event");
1372                return;
1373            }
1374        };
1375        state.subscribers.retain(|(connection_id, corr), subscriber| {
1376            let frame = Frame::build_with_version(
1377                subscriber.version,
1378                FrameType::StreamData,
1379                control_flags(),
1380                0,
1381                0,
1382                *corr,
1383                body.clone(),
1384            );
1385            match frame {
1386                Ok(frame) => {
1387                    if subscriber.frames.try_send(frame).is_ok() {
1388                        true
1389                    } else {
1390                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1391                        if let Some(lagged) = subscriber.lagged.take() {
1392                            let _ = lagged.send(event.cursor.clone());
1393                        }
1394                        false
1395                    }
1396                }
1397                Err(error) => {
1398                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1399                    false
1400                }
1401            }
1402        });
1403    }
1404
1405    fn subscribe(
1406        &self,
1407        connection_id: ConnectionId,
1408        corr: u64,
1409        version: u8,
1410        since: Option<SpawnCursor>,
1411        sink: FrameSink,
1412    ) -> Result<(), SpawnSubscribeRefusal> {
1413        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1414        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1415        {
1416            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1417            let replay = if let Some(since) = since {
1418                if since.daemon_incarnation != state.daemon_incarnation {
1419                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1420                        current: state.daemon_incarnation.clone(),
1421                    });
1422                }
1423                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1424                    if since.seq < oldest.seq.saturating_sub(1) {
1425                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1426                    }
1427                }
1428                state
1429                    .events
1430                    .iter()
1431                    .filter(|event| event.cursor.seq > since.seq)
1432                    .cloned()
1433                    .collect::<Vec<_>>()
1434            } else {
1435                Vec::new()
1436            };
1437            for event in replay {
1438                let body = serde_json::to_vec(&event)
1439                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1440                let frame = Frame::build_with_version(
1441                    version,
1442                    FrameType::StreamData,
1443                    control_flags(),
1444                    0,
1445                    0,
1446                    corr,
1447                    body,
1448                )
1449                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1450                frames
1451                    .try_send(frame)
1452                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1453            }
1454            state.subscribers.insert(
1455                (connection_id, corr),
1456                SpawnSubscriber {
1457                    version,
1458                    frames,
1459                    lagged: Some(lagged),
1460                },
1461            );
1462        }
1463        // The lagged terminal is sent here, by the forwarder, rather than by
1464        // the emitter: at the moment of the drop the subscriber's own channel
1465        // is full, and writing to the connection sink directly from the emitter
1466        // would put the Error AHEAD of the events still queued in that channel
1467        // (and the emitter holds the feed lock, so it cannot await the sink).
1468        // Dropping the subscriber drops the only sender, so `recv` drains every
1469        // queued event and then returns `None`; only then is the Error sent, so
1470        // the client sees each event it can keep, then the reason it was cut.
1471        // Cancel and connection removal drop the oneshot unsent, so they end
1472        // the stream with no Error.
1473        tokio::spawn(async move {
1474            while let Some(frame) = receiver.recv().await {
1475                if sink.send(frame).await.is_err() {
1476                    return;
1477                }
1478            }
1479            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1480                return;
1481            };
1482            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1483                Ok(frame) => {
1484                    let _ = sink.send(frame).await;
1485                }
1486                Err(error) => {
1487                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1488                }
1489            }
1490        });
1491        Ok(())
1492    }
1493
1494    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1495        let Some(subscriber) = self
1496            .0
1497            .lock()
1498            .unwrap_or_else(|p| p.into_inner())
1499            .subscribers
1500            .remove(&(connection_id, corr))
1501        else {
1502            return false;
1503        };
1504        if let Ok(frame) = Frame::build_with_version(
1505            subscriber.version,
1506            FrameType::StreamEnd,
1507            control_flags(),
1508            0,
1509            0,
1510            corr,
1511            Vec::new(),
1512        ) {
1513            tokio::spawn(async move {
1514                let _ = subscriber.frames.send(frame).await;
1515            });
1516        }
1517        true
1518    }
1519
1520    fn remove_connection(&self, connection_id: ConnectionId) {
1521        self.0
1522            .lock()
1523            .unwrap_or_else(|p| p.into_inner())
1524            .subscribers
1525            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1526    }
1527
1528    #[cfg(any(test, feature = "test-support"))]
1529    fn set_capacity(&self, capacity: usize) {
1530        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1531    }
1532
1533    #[cfg(any(test, feature = "test-support"))]
1534    fn subscriber_count(&self) -> usize {
1535        self.0
1536            .lock()
1537            .unwrap_or_else(|p| p.into_inner())
1538            .subscribers
1539            .len()
1540    }
1541}
1542
1543/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1544/// The terminal Error a lagged spawn subscriber receives after its queued events.
1545fn spawn_subscriber_lagged_frame(
1546    version: u8,
1547    corr: u64,
1548    first_undelivered: SpawnCursor,
1549) -> Result<Frame, String> {
1550    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1551        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1552        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1553            .to_string(),
1554        detail: Some(serde_json::json!({
1555            "first_undelivered_cursor": first_undelivered
1556        })),
1557    })
1558    .map_err(|error| error.to_string())?;
1559    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1560        .map_err(|error| error.to_string())
1561}
1562
1563pub trait ModuleProcessLiveness: Send + Sync {
1564    fn process_live(&self, module_id: &str) -> Option<bool>;
1565
1566    /// Whether the supervisor is replacing this module's process right now: an
1567    /// operator restart or reload, a health restart, or a crash respawn whose
1568    /// backoff is running. A module in that state is not live, but a consumer
1569    /// refused now should retry shortly rather than treat the target as gone.
1570    /// Stopped, failed, and disabled modules are not replacing.
1571    fn process_replacing(&self, _module_id: &str) -> bool {
1572        false
1573    }
1574}
1575
1576/// Shared process-liveness registry keyed by supervised `module_id`.
1577#[derive(Debug, Clone, Default)]
1578pub struct SupervisorProcessLiveness {
1579    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1580}
1581
1582impl SupervisorProcessLiveness {
1583    pub fn new() -> Self {
1584        Self::default()
1585    }
1586
1587    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1588        let mut snapshots = self
1589            .snapshots
1590            .lock()
1591            .unwrap_or_else(|poisoned| poisoned.into_inner());
1592        snapshots.insert(module_id, snapshot);
1593    }
1594
1595    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1596        let mut snapshots = self
1597            .snapshots
1598            .lock()
1599            .unwrap_or_else(|poisoned| poisoned.into_inner());
1600        let is_current = snapshots
1601            .get(module_id)
1602            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1603            .unwrap_or(false);
1604        if is_current {
1605            snapshots.remove(module_id);
1606        }
1607    }
1608}
1609
1610impl ModuleProcessLiveness for SupervisorProcessLiveness {
1611    fn process_live(&self, module_id: &str) -> Option<bool> {
1612        let snapshot = {
1613            let snapshots = self
1614                .snapshots
1615                .lock()
1616                .unwrap_or_else(|poisoned| poisoned.into_inner());
1617            snapshots.get(module_id).cloned()
1618        }?;
1619        let snapshot = snapshot
1620            .lock()
1621            .unwrap_or_else(|poisoned| poisoned.into_inner());
1622        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1623    }
1624
1625    fn process_replacing(&self, module_id: &str) -> bool {
1626        let Some(snapshot) = self
1627            .snapshots
1628            .lock()
1629            .unwrap_or_else(|poisoned| poisoned.into_inner())
1630            .get(module_id)
1631            .cloned()
1632        else {
1633            return false;
1634        };
1635        let snapshot = snapshot
1636            .lock()
1637            .unwrap_or_else(|poisoned| poisoned.into_inner());
1638        snapshot.enabled
1639            && match snapshot.state {
1640                ModuleState::Restarting => true,
1641                ModuleState::Draining => snapshot.draining_to_replace,
1642                ModuleState::Starting
1643                | ModuleState::Running
1644                | ModuleState::Unresponsive
1645                | ModuleState::Stopped
1646                | ModuleState::Failed
1647                | ModuleState::Disabled => false,
1648            }
1649    }
1650}
1651
1652#[cfg(test)]
1653#[derive(Debug, Default)]
1654struct ReloadExitRecordGate {
1655    reached: tokio::sync::Notify,
1656    resume: tokio::sync::Notify,
1657}
1658
1659#[derive(Debug, Clone, Copy)]
1660enum RespawnKind {
1661    Spawn,
1662    Reload,
1663}
1664
1665#[derive(Debug, Clone, Copy)]
1666struct PendingRespawn {
1667    deadline: Instant,
1668    kind: RespawnKind,
1669}
1670
1671type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1672
1673#[derive(Debug, Clone)]
1674struct SupervisorRuntimeConfig {
1675    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1676    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1677    /// A reload acknowledges completion only after its replacement registers.
1678    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1679    restart_policy: RestartPolicy,
1680    /// This module's RESOLVED drain budget: per-module config when present,
1681    /// else `default_drain_timeout`.
1682    drain_timeout: Duration,
1683    /// Shared with the status handle so the attested value changes atomically
1684    /// when a rescan updates the running drain policy.
1685    effective_drain_timeout: Arc<Mutex<Duration>>,
1686    /// The supervisor-wide fallback, kept so a configuration update that
1687    /// REMOVES the per-module override can re-resolve to it.
1688    default_drain_timeout: Duration,
1689    health: HealthConfig,
1690    connection_file_path: Option<PathBuf>,
1691    capture_logs_dir: Option<PathBuf>,
1692    forwarding: Option<Arc<ForwardingTable>>,
1693    /// The shared handle, so every spawn path (initial, restart, reload) records the
1694    /// reserved-module launch nonce the HELLO verifier checks against.
1695    supervisor_handle: Option<SupervisorHandle>,
1696    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1697    /// status queries.
1698    ///
1699    /// One ring per module, held across every respawn. The lines explaining an exit
1700    /// are written BEFORE that exit, so a ring recreated per process would be empty
1701    /// exactly when it is asked for.
1702    stderr_ring: Arc<Mutex<StderrRing>>,
1703    terminal_ring: Arc<Mutex<TerminalRing>>,
1704    spawn_events: SpawnEventFeed,
1705    child_roster: ChildRoster,
1706    #[cfg(target_os = "linux")]
1707    cgroup_placement: Option<subc_cgroup::Placement>,
1708    #[cfg(test)]
1709    test_seed_stale_facts_before_enable_spawn: bool,
1710    #[cfg(test)]
1711    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1712}
1713
1714#[derive(Debug, Clone, PartialEq, Eq)]
1715struct SupervisedConfiguration {
1716    spec: ModuleSpec,
1717    health: HealthConfig,
1718}
1719
1720/// Shared daemon lookup table for supervised module handles.
1721///
1722/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1723/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1724/// launch nonces recorded at spawn are checked by the same daemon instance.
1725#[derive(Debug, Clone, Default)]
1726pub struct SupervisorHandle {
1727    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1728    /// Module ids the supervisor has taken on. An id is added BEFORE the
1729    /// module's first process is spawned and removed only when the module
1730    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1731    /// the keys of `modules`.
1732    ///
1733    /// `modules` cannot answer "is this module configured?" on its own: a
1734    /// [`SupervisedModule`] only exists once its process has been spawned, and
1735    /// a fast child can connect, register, sync its scopes and ask about them
1736    /// before the supervisor has inserted it. Answering "not configured" in that
1737    /// gap makes scope admission refuse with the terminal "will never sync"
1738    /// instead of the retryable "has not synced yet".
1739    configured_ids: Arc<Mutex<HashSet<String>>>,
1740    spawn_events: SpawnEventFeed,
1741    /// The current expected launch nonce for each reserved module_id. Set when the
1742    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1743    /// non-reserved module never has an entry here and is never nonce-checked.
1744    /// Reserved module ids and the nonce that authorizes their next HELLO.
1745    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1746    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1747    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1748    /// had NO entry and admitted anyone: the reservation protected the nonce
1749    /// holder, not the NAME (found live by CKCRED's canary probe registering
1750    /// against a reserved scratch id).
1751    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1752    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1753    ///
1754    /// This is deliberately in-memory only: subc is state-free across daemon
1755    /// restarts, and the tombstone only explains the hours-after-removal window
1756    /// while this executing daemon is still alive. Do not persist it in a store.
1757    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1758    /// The current launch nonce for every supervised spawn. This is separate from
1759    /// reserved_nonces because consumer route.open attestation applies to all spawned
1760    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1761    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1762    /// Reserved namespace prefixes mapped to the supervised owner module whose
1763    /// current spawn nonce authorizes HELLO claims below the prefix.
1764    ///
1765    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1766    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1767    /// accidental collisions and lower-trust processes from squatting protected
1768    /// namespaces.
1769    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1770    /// Blue/green swaps in progress, by module id. An entry exists from just
1771    /// before the candidate process is spawned until the swap has failed, or
1772    /// has cut over and the old process is gone. While it exists, HELLO for the
1773    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1774    /// consumer attestation accepts both processes' nonces.
1775    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1776    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1777    promotion_observer: PromotionObserverSlot,
1778    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1779    /// this daemon-wide ordering, a rescan could retire or update a module while a
1780    /// concurrent reload still held its old handle and launch specification.
1781    operation_lock: Arc<AsyncMutex<()>>,
1782}
1783
1784/// Told when a swap has promoted its candidate to be the module's active
1785/// registration.
1786///
1787/// An ordinary HELLO runs the control plane's registration side effects (the
1788/// capability cache, the deny census, the requirement recompute) as it
1789/// registers. A swap candidate's HELLO does not, because it is not routable;
1790/// promotion is when those must run instead, and promotion happens in the
1791/// supervisor, which has no other way into the control handler.
1792pub(crate) trait SwapPromotionObserver: Send + Sync {
1793    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1794}
1795
1796/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1797/// control handler) owns this handle, so a strong reference back would be a
1798/// cycle that keeps both alive.
1799#[derive(Clone, Default)]
1800struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1801
1802impl fmt::Debug for PromotionObserverSlot {
1803    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1804        f.write_str("PromotionObserverSlot")
1805    }
1806}
1807
1808/// The nonces of one open swap.
1809#[derive(Debug, Clone)]
1810struct OpenSwap {
1811    /// The launch nonce minted for the candidate process. It is the swap
1812    /// token: the only thing that admits a HELLO into the candidate slot.
1813    candidate_nonce: String,
1814    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1815    /// here because cutover moves the module's recorded spawn nonce to the
1816    /// candidate while the incumbent is still draining and its consumers are
1817    /// still attesting with this one.
1818    incumbent_nonce: Option<String>,
1819    /// Set once a HELLO has been admitted with the swap token, so the token
1820    /// admits one registration and cannot be replayed after cutover empties
1821    /// the candidate slot.
1822    candidate_admitted: bool,
1823}
1824
1825/// What the swap gate says about a HELLO. See
1826/// [`SupervisorHandle::swap_hello_admission`].
1827#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1828pub(crate) enum SwapHelloAdmission {
1829    /// No swap is open for the id (or the HELLO carries the incumbent's own
1830    /// nonce); the ordinary gates decide.
1831    NotSwapping,
1832    /// The HELLO carries the swap token: register it into the candidate slot.
1833    Candidate,
1834    /// A swap is open and the HELLO carries a nonce the supervisor did not
1835    /// mint for this id, no nonce, or a token already used.
1836    Refused,
1837}
1838
1839#[derive(Debug, Clone, PartialEq, Eq)]
1840pub(crate) enum ReservedHelloRejection {
1841    Exact {
1842        module_id: String,
1843    },
1844    Prefix {
1845        prefix: String,
1846        owner_module_id: String,
1847    },
1848}
1849
1850impl SupervisorHandle {
1851    pub fn new() -> Self {
1852        Self::default()
1853    }
1854
1855    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1856        self.spawn_events.snapshot()
1857    }
1858
1859    pub(crate) fn subscribe_spawns(
1860        &self,
1861        connection_id: ConnectionId,
1862        corr: u64,
1863        version: u8,
1864        since: Option<SpawnCursor>,
1865        sink: FrameSink,
1866    ) -> Result<(), SpawnSubscribeRefusal> {
1867        self.spawn_events
1868            .subscribe(connection_id, corr, version, since, sink)
1869    }
1870
1871    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1872        self.spawn_events.cancel(connection_id, corr)
1873    }
1874
1875    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1876        self.spawn_events.remove_connection(connection_id);
1877    }
1878
1879    #[cfg(any(test, feature = "test-support"))]
1880    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1881        assert!(capacity > 0, "spawn event capacity must be non-zero");
1882        self.spawn_events.set_capacity(capacity);
1883    }
1884
1885    #[cfg(any(test, feature = "test-support"))]
1886    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1887        self.spawn_events.subscriber_count()
1888    }
1889
1890    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1891    /// a respawn invalidates stale consumer identities.
1892    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1893        self.spawn_nonces
1894            .lock()
1895            .unwrap_or_else(|poisoned| poisoned.into_inner())
1896            .insert(module_id.to_string(), nonce);
1897    }
1898
1899    /// Record the launch nonce expected from the next HELLO for a reserved module,
1900    /// replacing any prior nonce (a respawn invalidates the previous one).
1901    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1902        self.reserved_nonces
1903            .lock()
1904            .unwrap_or_else(|poisoned| poisoned.into_inner())
1905            .insert(module_id.to_string(), Some(nonce));
1906    }
1907
1908    /// Record namespace prefixes owned by a supervised module.
1909    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1910        let mut owners = self
1911            .reserved_prefix_owners
1912            .lock()
1913            .unwrap_or_else(|poisoned| poisoned.into_inner());
1914        owners.retain(|_, owner| owner != owner_module_id);
1915        for prefix in prefixes {
1916            owners.insert(prefix.clone(), owner_module_id.to_string());
1917        }
1918    }
1919
1920    /// The launch nonce most recently minted for a module's spawn, if any.
1921    #[cfg(test)]
1922    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1923        self.spawn_nonces
1924            .lock()
1925            .unwrap_or_else(|poisoned| poisoned.into_inner())
1926            .get(module_id)
1927            .cloned()
1928    }
1929
1930    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1931        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1932        let spawn_nonce = self
1933            .spawn_nonces
1934            .lock()
1935            .unwrap_or_else(|poisoned| poisoned.into_inner())
1936            .get(&spec.module_id)
1937            .cloned();
1938        let mut reserved_nonces = self
1939            .reserved_nonces
1940            .lock()
1941            .unwrap_or_else(|poisoned| poisoned.into_inner());
1942        if spec.reserved {
1943            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1944            // reserved name whose module has never spawned has no legitimate
1945            // holder, and the entry's absence is what used to leave the name
1946            // open to the first claimant.
1947            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1948        }
1949        drop(reserved_nonces);
1950        // A later unreserved declaration must not silently unreserve an id that
1951        // was retained after its reserved configuration was removed. The explicit
1952        // release ceremony is the only operation that retires that gate.
1953        self.removal_tombstones
1954            .lock()
1955            .unwrap_or_else(|poisoned| poisoned.into_inner())
1956            .remove(&spec.module_id);
1957    }
1958
1959    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
1960    /// authorized only by its expected nonce; otherwise a matching reserved prefix
1961    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
1962    /// with no matching prefix are always authorized.
1963    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1964        self.reserved_hello_rejection(module_id, presented)
1965            .is_none()
1966    }
1967
1968    pub(crate) fn reserved_hello_rejection(
1969        &self,
1970        module_id: &str,
1971        presented: Option<&str>,
1972    ) -> Option<ReservedHelloRejection> {
1973        let nonces = self
1974            .reserved_nonces
1975            .lock()
1976            .unwrap_or_else(|poisoned| poisoned.into_inner());
1977        if let Some(expected) = nonces.get(module_id) {
1978            // `None` = reserved with no legitimate holder: refuse every
1979            // presentation, because no process can hold a nonce that was never
1980            // minted. Only a real minted nonce admits, in constant time.
1981            let authorized = match expected {
1982                Some(expected) => {
1983                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1984                }
1985                None => false,
1986            };
1987            if authorized {
1988                return None;
1989            }
1990            return Some(ReservedHelloRejection::Exact {
1991                module_id: module_id.to_string(),
1992            });
1993        }
1994        drop(nonces);
1995
1996        let matched_prefix = self
1997            .reserved_prefix_owners
1998            .lock()
1999            .unwrap_or_else(|poisoned| poisoned.into_inner())
2000            .iter()
2001            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2002            .max_by_key(|(prefix, _)| prefix.len())
2003            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2004        let (prefix, owner_module_id) = matched_prefix?;
2005
2006        let authorized = presented.is_some_and(|presented| {
2007            self.spawn_nonces
2008                .lock()
2009                .unwrap_or_else(|poisoned| poisoned.into_inner())
2010                .get(&owner_module_id)
2011                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2012                // While the owner is being swapped, children started by
2013                // either of its two processes hold that process's nonce.
2014                || self.swap_nonce_matches(&owner_module_id, presented)
2015        });
2016        if authorized {
2017            None
2018        } else {
2019            Some(ReservedHelloRejection::Prefix {
2020                prefix,
2021                owner_module_id,
2022            })
2023        }
2024    }
2025
2026    /// Whether a consumer connection proved it came from a daemon-spawned module.
2027    ///
2028    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2029    /// accepted only for module ids the supervisor has spawned.
2030    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2031        if presented.is_empty() {
2032            return false;
2033        }
2034        let nonces = self
2035            .spawn_nonces
2036            .lock()
2037            .unwrap_or_else(|poisoned| poisoned.into_inner());
2038        let current = nonces
2039            .get(module_id)
2040            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2041        drop(nonces);
2042        // During a swap two processes of the module are alive, and a consumer
2043        // started by either one presents that process's nonce. Accepting only
2044        // the recorded one would fail the incumbent's consumers for the whole
2045        // overlap once cutover moves the record to the candidate.
2046        current || self.swap_nonce_matches(module_id, presented)
2047    }
2048
2049    /// Whether `presented` is either nonce of an open swap for `module_id`.
2050    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2051        let swaps = self
2052            .swaps
2053            .lock()
2054            .unwrap_or_else(|poisoned| poisoned.into_inner());
2055        swaps.get(module_id).is_some_and(|swap| {
2056            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2057                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2058                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2059                })
2060        })
2061    }
2062
2063    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2064    /// Called before the candidate process exists.
2065    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2066        let incumbent_nonce = self
2067            .spawn_nonces
2068            .lock()
2069            .unwrap_or_else(|poisoned| poisoned.into_inner())
2070            .get(module_id)
2071            .cloned();
2072        self.swaps
2073            .lock()
2074            .unwrap_or_else(|poisoned| poisoned.into_inner())
2075            .insert(
2076                module_id.to_string(),
2077                OpenSwap {
2078                    candidate_nonce,
2079                    incumbent_nonce,
2080                    candidate_admitted: false,
2081                },
2082            );
2083    }
2084
2085    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2086    /// the module's recorded one.
2087    pub(crate) fn close_swap(&self, module_id: &str) {
2088        self.swaps
2089            .lock()
2090            .unwrap_or_else(|poisoned| poisoned.into_inner())
2091            .remove(module_id);
2092    }
2093
2094    /// Install the observer told about swap promotions, replacing any earlier
2095    /// one.
2096    pub(crate) fn set_swap_promotion_observer(
2097        &self,
2098        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2099    ) {
2100        *self
2101            .promotion_observer
2102            .0
2103            .lock()
2104            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2105    }
2106
2107    /// Tell the installed observer, if it is still alive, that a swap promoted
2108    /// `registration`.
2109    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2110        let observer = self
2111            .promotion_observer
2112            .0
2113            .lock()
2114            .unwrap_or_else(|poisoned| poisoned.into_inner())
2115            .as_ref()
2116            .and_then(std::sync::Weak::upgrade);
2117        if let Some(observer) = observer {
2118            observer.swap_promoted(registration);
2119        }
2120    }
2121
2122    /// Whether a swap is open for `module_id`.
2123    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2124        self.swaps
2125            .lock()
2126            .unwrap_or_else(|poisoned| poisoned.into_inner())
2127            .contains_key(module_id)
2128    }
2129
2130    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2131    /// respawn would, once cutover has made the candidate the module's process.
2132    /// The swap stays open so the incumbent's nonce keeps attesting until the
2133    /// incumbent has drained and exited.
2134    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2135        let candidate_nonce = self
2136            .swaps
2137            .lock()
2138            .unwrap_or_else(|poisoned| poisoned.into_inner())
2139            .get(module_id)
2140            .map(|swap| swap.candidate_nonce.clone());
2141        let Some(nonce) = candidate_nonce else {
2142            return;
2143        };
2144        self.set_spawn_nonce(module_id, nonce.clone());
2145        if reserved {
2146            self.set_reserved_nonce(module_id, nonce);
2147        }
2148    }
2149
2150    /// The swap gate for a HELLO claiming `module_id`.
2151    ///
2152    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2153    /// presents the candidate nonce, which the reserved gate (holding the
2154    /// incumbent's nonce) would refuse as `reserved_module` before swap
2155    /// admission was ever reached. And it applies to unreserved ids too: for an
2156    /// unreserved id the only thing that ever stopped a second process claiming
2157    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2158    /// refusal a swap lifts for its candidate.
2159    ///
2160    /// The incumbent's own nonce falls through to the ordinary gates, which
2161    /// treat it as they always have (a live incumbent is refused as a
2162    /// duplicate). Anything else while a swap is open is refused, including an
2163    /// absent nonce.
2164    pub(crate) fn swap_hello_admission(
2165        &self,
2166        module_id: &str,
2167        presented: Option<&str>,
2168    ) -> SwapHelloAdmission {
2169        let swaps = self
2170            .swaps
2171            .lock()
2172            .unwrap_or_else(|poisoned| poisoned.into_inner());
2173        let Some(swap) = swaps.get(module_id) else {
2174            return SwapHelloAdmission::NotSwapping;
2175        };
2176        let Some(presented) = presented else {
2177            return SwapHelloAdmission::Refused;
2178        };
2179        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2180            return if swap.candidate_admitted {
2181                SwapHelloAdmission::Refused
2182            } else {
2183                SwapHelloAdmission::Candidate
2184            };
2185        }
2186        if swap
2187            .incumbent_nonce
2188            .as_deref()
2189            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2190        {
2191            return SwapHelloAdmission::NotSwapping;
2192        }
2193        SwapHelloAdmission::Refused
2194    }
2195
2196    /// Record that the swap token has registered a candidate, so it admits no
2197    /// second HELLO.
2198    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2199        if let Some(swap) = self
2200            .swaps
2201            .lock()
2202            .unwrap_or_else(|poisoned| poisoned.into_inner())
2203            .get_mut(module_id)
2204        {
2205            swap.candidate_admitted = true;
2206        }
2207    }
2208
2209    /// Test/support lookup for the current launch nonce of a supervised spawn.
2210    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2211        self.spawn_nonces
2212            .lock()
2213            .unwrap_or_else(|poisoned| poisoned.into_inner())
2214            .get(module_id)
2215            .cloned()
2216    }
2217
2218    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2219    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2220        self.reserved_nonces
2221            .lock()
2222            .unwrap_or_else(|poisoned| poisoned.into_inner())
2223            .get(module_id)
2224            .cloned()
2225            .flatten()
2226    }
2227
2228    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2229        // Normally already marked before the process was spawned; marking here
2230        // too keeps `configured_ids` a superset of the roster for any caller
2231        // that inserts a module directly.
2232        self.mark_configured(module.module_id());
2233        let mut modules = self
2234            .modules
2235            .lock()
2236            .unwrap_or_else(|poisoned| poisoned.into_inner());
2237        modules.insert(module.module_id().to_string(), module)
2238    }
2239
2240    /// Record that the supervisor has taken on `module_id`. Called before the
2241    /// module's first process is spawned, so that by the time that process can
2242    /// register, [`Self::is_configured`] already answers true.
2243    fn mark_configured(&self, module_id: &str) {
2244        self.configured_ids
2245            .lock()
2246            .unwrap_or_else(|poisoned| poisoned.into_inner())
2247            .insert(module_id.to_string());
2248    }
2249
2250    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2251    /// before it was ever put on the roster. A module already on the roster
2252    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2253    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2254        let modules = self
2255            .modules
2256            .lock()
2257            .unwrap_or_else(|poisoned| poisoned.into_inner());
2258        if !modules.contains_key(module_id) {
2259            self.configured_ids
2260                .lock()
2261                .unwrap_or_else(|poisoned| poisoned.into_inner())
2262                .remove(module_id);
2263        }
2264    }
2265
2266    /// Whether `module_id` is a module this daemon supervises: on the roster,
2267    /// or about to be (its process is being spawned right now).
2268    ///
2269    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2270    /// for scopes: a supervised module's process can register and sync before
2271    /// [`Self::get`] can return it, and in that window it is still a module
2272    /// that will sync, not one that never will.
2273    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2274        self.configured_ids
2275            .lock()
2276            .unwrap_or_else(|poisoned| poisoned.into_inner())
2277            .contains(module_id)
2278    }
2279
2280    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2281        let modules = self
2282            .modules
2283            .lock()
2284            .unwrap_or_else(|poisoned| poisoned.into_inner());
2285        modules.get(module_id).cloned()
2286    }
2287
2288    pub(crate) fn record_late_health_answer(
2289        &self,
2290        module_id: &str,
2291        latency_ms: u64,
2292    ) -> Result<bool, SuperviseError> {
2293        let Some(module) = self.get(module_id) else {
2294            return Ok(false);
2295        };
2296        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2297            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2298            state.health.last_late_answer_latency_ms = Some(latency_ms);
2299            // A late answer is an answer: the module served the probe, just past
2300            // the deadline. Leaving the miss streak in place while logging
2301            // "proves the module is alive" is how a CPU-starved module that
2302            // answers every probe a few seconds late still marches to the
2303            // threshold and gets killed — the exact kill class `NoAnswer` is
2304            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2305            // is degradation, and degradation reports; it does not restart.
2306            state.health.consecutive_failures = 0;
2307        })?;
2308        Ok(true)
2309    }
2310
2311    /// Arm the one-shot marker for the module process that this caller
2312    /// deliberately initiated severance against. Generic connection teardown
2313    /// must not call this:
2314    /// a surviving process would otherwise retain an exemption for a later
2315    /// genuine crash.
2316    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2317        let Some(module) = self.get(module_id) else {
2318            return Ok(false);
2319        };
2320        let status = module.status()?;
2321        let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
2322            return Ok(false);
2323        };
2324        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2325    }
2326
2327    pub fn list(&self) -> Vec<SupervisedModule> {
2328        let modules = self
2329            .modules
2330            .lock()
2331            .unwrap_or_else(|poisoned| poisoned.into_inner());
2332        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2333        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2334        modules
2335    }
2336
2337    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2338        self.spawn_nonces
2339            .lock()
2340            .unwrap_or_else(|poisoned| poisoned.into_inner())
2341            .remove(module_id);
2342        self.close_swap(module_id);
2343        let mut reserved_nonces = self
2344            .reserved_nonces
2345            .lock()
2346            .unwrap_or_else(|poisoned| poisoned.into_inner());
2347        if reserved_nonces.contains_key(module_id) {
2348            // The old nonce must die with the removed process, but the exact-id
2349            // gate remains until an operator explicitly releases it.
2350            reserved_nonces.insert(module_id.to_string(), None);
2351        }
2352        drop(reserved_nonces);
2353        self.reserved_prefix_owners
2354            .lock()
2355            .unwrap_or_else(|poisoned| poisoned.into_inner())
2356            .retain(|_, owner| owner != module_id);
2357        let removed = self
2358            .modules
2359            .lock()
2360            .unwrap_or_else(|poisoned| poisoned.into_inner())
2361            .remove(module_id);
2362        self.configured_ids
2363            .lock()
2364            .unwrap_or_else(|poisoned| poisoned.into_inner())
2365            .remove(module_id);
2366        removed
2367    }
2368
2369    /// Remember a module removed by a non-preview rescan so route.open can
2370    /// distinguish that intentional removal from an unknown id.
2371    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2372        self.removal_tombstones
2373            .lock()
2374            .unwrap_or_else(|poisoned| poisoned.into_inner())
2375            .insert(module_id.to_string(), unix_ms_now());
2376    }
2377
2378    /// Return how long ago a rescan removed this module in milliseconds.
2379    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2380        self.removal_tombstones
2381            .lock()
2382            .unwrap_or_else(|poisoned| poisoned.into_inner())
2383            .get(module_id)
2384            .copied()
2385            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2386    }
2387
2388    /// Retire a reserved-id gate only after its module has left supervision.
2389    ///
2390    /// A retained gate has no live nonce (`None`), so releasing any other entry
2391    /// would weaken a currently configured or otherwise active reservation.
2392    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2393        if self.get(module_id).is_some() {
2394            return false;
2395        }
2396        let mut reserved_nonces = self
2397            .reserved_nonces
2398            .lock()
2399            .unwrap_or_else(|poisoned| poisoned.into_inner());
2400        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2401            return false;
2402        }
2403        reserved_nonces.remove(module_id);
2404        true
2405    }
2406
2407    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2408        Arc::clone(&self.operation_lock)
2409    }
2410}
2411
2412/// Process supervisor for subc-owned singleton modules.
2413#[derive(Debug, Clone)]
2414pub struct Supervisor {
2415    registry: Arc<Registry>,
2416    restart_policy: RestartPolicy,
2417    drain_timeout: Duration,
2418    connection_file_path: Option<PathBuf>,
2419    capture_logs_dir: Option<PathBuf>,
2420    forwarding: Option<Arc<ForwardingTable>>,
2421    process_liveness: Arc<SupervisorProcessLiveness>,
2422    supervisor_handle: Option<SupervisorHandle>,
2423    health: HealthConfig,
2424    daemon_start_clock: crate::clock::StartClock,
2425    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2426    spawn_events: SpawnEventFeed,
2427    provenance_probe: ExecutableIdentityProbe,
2428    /// Every process spawned through this supervisor (and its clones) and not
2429    /// yet reaped, so daemon shutdown can end them.
2430    child_roster: ChildRoster,
2431    #[cfg(target_os = "linux")]
2432    cgroup_placement: Option<subc_cgroup::Placement>,
2433    #[cfg(test)]
2434    test_after_first_spawn: AfterFirstSpawnHook,
2435}
2436
2437/// Test-only hook run on the path that takes on a new module, right after its
2438/// first `spawn_child` returns (the process exists and could already be
2439/// registering) and before that process is handed to the module's supervise
2440/// loop and put on the roster. Lets a test observe what a fast child would see
2441/// in that window without racing a real one.
2442#[cfg(test)]
2443#[derive(Clone, Default)]
2444struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2445
2446#[cfg(test)]
2447type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2448
2449#[cfg(test)]
2450impl fmt::Debug for AfterFirstSpawnHook {
2451    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2452        f.write_str("AfterFirstSpawnHook")
2453    }
2454}
2455
2456#[cfg(test)]
2457impl AfterFirstSpawnHook {
2458    fn run(&self, module_id: &str) {
2459        if let Some(hook) = &self.0 {
2460            hook(module_id);
2461        }
2462    }
2463}
2464
2465impl Supervisor {
2466    #[cfg(test)]
2467    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2468        let supervisor = Self::new(registry, policy);
2469        #[cfg(target_os = "macos")]
2470        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2471        supervisor
2472    }
2473    /// Verify the trampoline once when configured. Missing private OS support
2474    /// refuses every macOS launch by name but does not stop the daemon's control
2475    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2476    /// entry point; the library must not exec an arbitrary hosting program.
2477    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2478        let path = path.into();
2479        #[cfg(target_os = "macos")]
2480        {
2481            let result = probe_privacy_trampoline(&path).map(|()| path);
2482            if let Err(cause) = &result {
2483                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2484            }
2485            self.child_roster.set_privacy_trampoline(result);
2486        }
2487        #[cfg(not(target_os = "macos"))]
2488        let _ = path;
2489        self
2490    }
2491    /// The first step of an announced daemon shutdown, before the notice and
2492    /// before any connection is closed.
2493    ///
2494    /// Sets the daemon-shutdown flag first: from here on no module is
2495    /// respawned (crash restart, operator restart, or swap), and every child
2496    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2497    /// the module exits on the EOF this shutdown gives it or is signalled by a
2498    /// service manager that kills the whole cgroup. Then writes the journal's
2499    /// shutdown marker, which records the instant and closes this daemon
2500    /// incarnation's stretch of the journal.
2501    #[cfg(unix)]
2502    pub(crate) fn begin_daemon_shutdown(&self) {
2503        self.child_roster.close();
2504        if let Some(journal) = &self.terminal_journal {
2505            journal.stamp_shutdown();
2506        }
2507    }
2508
2509    /// Announce a cut while established connections can still carry replies.
2510    /// These budgets promise notice and a bounded wait, not child completion;
2511    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2512    #[cfg(unix)]
2513    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2514        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2515        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2516        let Some(forwarding) = &self.forwarding else {
2517            return Ok(());
2518        };
2519        let module_ids = forwarding
2520            .begin_daemon_drain()
2521            .map_err(SuperviseError::Forwarding)?;
2522        let deadline_ms =
2523            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2524        let mut notices = tokio::task::JoinSet::new();
2525        let mut drains = Vec::new();
2526        for module_id in module_ids {
2527            let Some(target) = forwarding
2528                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2529                .map_err(SuperviseError::Forwarding)?
2530            else {
2531                continue;
2532            };
2533            let routes = forwarding
2534                .endpoint_routes(target.endpoint)
2535                .map_err(SuperviseError::Forwarding)?;
2536            // Restart allows deployed consumers to reopen after the new daemon
2537            // appears. The wire reason stays `restart`; what tells a daemon cut
2538            // apart from a module restart afterwards is the terminal record
2539            // itself, whose disposition is `daemon_shutdown` for every exit
2540            // observed once `begin_daemon_shutdown` has run.
2541            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2542                reason: RouteCloseReason::Restart,
2543                deadline_ms,
2544            })
2545            .expect("module draining serializes");
2546            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2547            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2548            for route in routes {
2549                let client = route.goodbye_target;
2550                if let Some((_, channels)) = clients
2551                    .iter_mut()
2552                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2553                {
2554                    channels.push(client.channel);
2555                } else {
2556                    let channel = client.channel;
2557                    clients.push((client, vec![channel]));
2558                }
2559            }
2560            for (client, mut channels) in clients {
2561                channels.sort_unstable();
2562                channels.dedup();
2563                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2564                    module_id: module_id.clone(),
2565                    channels,
2566                    reason: RouteCloseReason::Restart,
2567                })
2568                .expect("route closing serializes");
2569                recipients.push((client.sink, client.negotiated_ver, closing));
2570            }
2571            for (sink, version, body) in recipients {
2572                notices.spawn(async move {
2573                    let frame = Frame::build_with_version(
2574                        version,
2575                        FrameType::Push,
2576                        control_flags(),
2577                        0,
2578                        0,
2579                        0,
2580                        body,
2581                    )
2582                    .expect("bounded lifecycle notice frame builds");
2583                    sink.send_flushed(frame).await
2584                });
2585            }
2586            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2587            drains.push((module_id, target.endpoint, gauges));
2588        }
2589        // A quiet forwarding table is not proof that queued notices reached the
2590        // socket. Wait for writer flush acknowledgements before testing quiescence.
2591        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2592        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2593            if !matches!(result, Ok(Ok(()))) {
2594                warn!(?result, "daemon shutdown notice delivery failed");
2595            }
2596        }
2597        notices.abort_all();
2598        let deadline = Instant::now() + DRAIN_BUDGET;
2599        let mut waits = tokio::task::JoinSet::new();
2600        for (module_id, endpoint, gauges) in drains {
2601            let forwarding = Arc::clone(forwarding);
2602            let mut runtime = self.runtime_config();
2603            runtime.health.cadence = Duration::from_millis(100);
2604            waits.spawn(async move {
2605                wait_for_forwarding_quiescence(
2606                    &forwarding,
2607                    &module_id,
2608                    &runtime,
2609                    endpoint,
2610                    deadline,
2611                    &gauges,
2612                    DrainScope::Active,
2613                )
2614                .await
2615            });
2616        }
2617        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2618            if !matches!(result, Ok(Ok(true))) {
2619                warn!(?result, "daemon shutdown drain did not reach quiescence");
2620            }
2621        }
2622        Ok(())
2623    }
2624
2625    /// The last step of an announced daemon shutdown, after the notice and the
2626    /// drain: send every registered module a module GOODBYE, the same planned
2627    /// stop signal `ck module stop` gives, then close every connection so each
2628    /// subc module sees EOF and starts its own teardown, then end every
2629    /// supervised child that has not exited
2630    /// by its own deadline (its drain budget, capped). Modules lead their own
2631    /// process groups, so a
2632    /// service manager's group kill no longer reaches them; without this a
2633    /// child that does not stop on EOF (every `protocol: "none"` child, which
2634    /// has no connection) would outlive the daemon. Every wait is bounded (see
2635    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2636    #[cfg(unix)]
2637    pub(crate) async fn end_children_for_daemon_shutdown(
2638        &self,
2639        already_escalated: bool,
2640        escalate: impl std::future::Future<Output = ()>,
2641    ) {
2642        tokio::pin!(escalate);
2643        let mut escalated = already_escalated;
2644        if let Some(forwarding) = &self.forwarding {
2645            let reason = CloseReason::new(
2646                "daemon_shutdown",
2647                "the daemon is exiting after its shutdown notice and drain",
2648            );
2649            if escalated {
2650                // The operator asked to stop waiting: queue the GOODBYEs but
2651                // do not wait for them to be written.
2652                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2653            } else {
2654                tokio::select! {
2655                    biased;
2656                    _ = escalate.as_mut() => {
2657                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2658                        escalated = true;
2659                    }
2660                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2661                }
2662            }
2663            let closed = forwarding.close_all_connections(&reason);
2664            debug!(closed, "closed established connections for daemon shutdown");
2665        }
2666        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2667        // already completed and must not be polled again; the child shutdown
2668        // wait is told it is escalated and gets a future that never fires.
2669        let escalated_here = escalated && !already_escalated;
2670        let remaining_escalate = async move {
2671            if escalated_here {
2672                std::future::pending::<()>().await;
2673            } else {
2674                escalate.await;
2675            }
2676        };
2677        crate::child_roster::end_children_for_daemon_shutdown(
2678            &self.child_roster,
2679            escalated,
2680            remaining_escalate,
2681        )
2682        .await;
2683    }
2684
2685    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2686        Self {
2687            registry,
2688            restart_policy,
2689            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2690            connection_file_path: None,
2691            capture_logs_dir: None,
2692            forwarding: None,
2693            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2694            supervisor_handle: None,
2695            health: HealthConfig::default(),
2696            daemon_start_clock: crate::clock::StartClock::capture(),
2697            terminal_journal: None,
2698            spawn_events: SpawnEventFeed::default(),
2699            provenance_probe: ExecutableIdentityProbe::default(),
2700            child_roster: ChildRoster::default(),
2701            #[cfg(target_os = "linux")]
2702            cgroup_placement: None,
2703            #[cfg(test)]
2704            test_after_first_spawn: AfterFirstSpawnHook::default(),
2705        }
2706    }
2707
2708    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2709        self.drain_timeout = drain_timeout;
2710        self
2711    }
2712
2713    pub fn with_process_liveness(
2714        mut self,
2715        process_liveness: Arc<SupervisorProcessLiveness>,
2716    ) -> Self {
2717        self.process_liveness = process_liveness;
2718        self
2719    }
2720
2721    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2722        self.connection_file_path = Some(connection_file_path.into());
2723        self
2724    }
2725
2726    /// Enables daemon-owned capture files for supervised stdout and stderr.
2727    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2728        self.capture_logs_dir = Some(logs_dir.into());
2729        self
2730    }
2731
2732    /// Names this daemon lifetime in spawn events, independently of whether a
2733    /// terminal journal is configured.
2734    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2735        // A millisecond start stamp can repeat after clock rollback or a rapid
2736        // restart. Use the connection file's random daemon_id instead: it already
2737        // identifies this daemon lifetime independently of the wall clock.
2738        self.spawn_events.configure_incarnation(daemon_incarnation);
2739        self
2740    }
2741
2742    /// Enables best-effort history shared by every supervised module. Without
2743    /// it, terminal history is kept only in each module's in-memory ring.
2744    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2745        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2746        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2747            path,
2748            daemon_incarnation,
2749        )));
2750        this
2751    }
2752
2753    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2754        self.forwarding = Some(forwarding);
2755        self
2756    }
2757
2758    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2759        self.spawn_events = supervisor_handle.spawn_events.clone();
2760        self.supervisor_handle = Some(supervisor_handle);
2761        self
2762    }
2763
2764    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2765        self.health = health;
2766        self
2767    }
2768
2769    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2770    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2771    /// record is kept.
2772    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2773        self.child_roster.record_to(path.into());
2774        self
2775    }
2776
2777    #[cfg(target_os = "linux")]
2778    pub fn with_cgroup_placement(
2779        mut self,
2780        cgroup_placement: Option<subc_cgroup::Placement>,
2781    ) -> Self {
2782        self.cgroup_placement = cgroup_placement;
2783        self
2784    }
2785
2786    /// Spawn `spec.program` and start monitoring it.
2787    ///
2788    /// The child is expected to parse `--subc <connection-file-path>`, read the
2789    /// TCP+key connection file, authenticate to the already-running listener, and
2790    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2791    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2792        validate_spec(&spec)?;
2793        self.establish_identity(&spec);
2794
2795        let runtime = self.runtime_config();
2796        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2797        let spawned = spawn_child(
2798            &spec,
2799            runtime.connection_file_path.as_deref(),
2800            self.supervisor_handle.as_ref(),
2801            &runtime.stderr_ring,
2802            runtime.capture_logs_dir.as_deref(),
2803            &runtime.child_roster,
2804            #[cfg(target_os = "linux")]
2805            runtime.cgroup_placement.as_ref(),
2806        );
2807        #[cfg(test)]
2808        self.test_after_first_spawn.run(&spec.module_id);
2809        let child = match spawned {
2810            Ok(child) => child,
2811            Err(err) => {
2812                // Unlike the configured paths, a failed `spawn` leaves nothing
2813                // on the roster, so the module must not stay marked configured.
2814                self.abandon_unrostered(&spec.module_id);
2815                return Err(err);
2816            }
2817        };
2818        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2819
2820        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2821    }
2822
2823    /// Make `spec`'s module count as configured, with its identity gates
2824    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2825    /// exists.
2826    ///
2827    /// Every path that takes on a new module calls this before `spawn_child`.
2828    /// The order is the point: the child can connect, register, sync its
2829    /// scopes and ask about them as soon as it is spawned, and the module is
2830    /// only put on the roster after `spawn_child` returns. Were the mark set
2831    /// with the roster entry, a fast child would see its own owner reported
2832    /// as not configured, and a scoped `route.open` in that window would be
2833    /// refused as terminal `scope_not_live` ("will never sync") instead of
2834    /// retryable `scope_not_synced`.
2835    fn establish_identity(&self, spec: &ModuleSpec) {
2836        if let Some(supervisor_handle) = &self.supervisor_handle {
2837            supervisor_handle.apply_identity_configuration(spec);
2838            supervisor_handle.mark_configured(&spec.module_id);
2839        }
2840    }
2841
2842    /// Take back [`Self::establish_identity`]'s configured mark when the
2843    /// module will not be put on the roster after all.
2844    fn abandon_unrostered(&self, module_id: &str) {
2845        if let Some(supervisor_handle) = &self.supervisor_handle {
2846            supervisor_handle.unmark_configured_unless_rostered(module_id);
2847        }
2848    }
2849
2850    /// Record a freshly spawned first process as running. On failure the
2851    /// module never reaches the roster, so its configured mark is taken back.
2852    fn mark_first_process_running(
2853        &self,
2854        spec: &ModuleSpec,
2855        runtime: &SupervisorRuntimeConfig,
2856        snapshot: &SharedSnapshot,
2857        child: &SupervisedChild,
2858    ) -> Result<(), SuperviseError> {
2859        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2860            self.abandon_unrostered(&spec.module_id);
2861            return Err(err);
2862        }
2863        self.process_liveness
2864            .track(spec.module_id.clone(), Arc::clone(snapshot));
2865        Ok(())
2866    }
2867
2868    /// Start supervising a module declared in daemon configuration.
2869    ///
2870    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2871    /// failures in the supervisor handle so operator-facing `supervisor.list`
2872    /// reflects every configured module while daemon startup continues.
2873    pub fn supervise_configured(
2874        &self,
2875        spec: ModuleSpec,
2876        enabled: bool,
2877    ) -> Result<SupervisedModule, SuperviseError> {
2878        validate_spec(&spec)?;
2879        self.establish_identity(&spec);
2880
2881        let runtime = self.runtime_config();
2882        if !enabled {
2883            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2884            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2885        }
2886
2887        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2888        let spawned = spawn_child(
2889            &spec,
2890            runtime.connection_file_path.as_deref(),
2891            self.supervisor_handle.as_ref(),
2892            &runtime.stderr_ring,
2893            runtime.capture_logs_dir.as_deref(),
2894            &runtime.child_roster,
2895            #[cfg(target_os = "linux")]
2896            runtime.cgroup_placement.as_ref(),
2897        );
2898        #[cfg(test)]
2899        self.test_after_first_spawn.run(&spec.module_id);
2900        match spawned {
2901            Ok(child) => {
2902                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2903                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2904            }
2905            Err(err) => {
2906                error!(
2907                    module_id = %spec.module_id,
2908                    program = %spec.program.display(),
2909                    error = %err,
2910                    "configured module failed to spawn; marking failed and continuing"
2911                );
2912                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2913                Ok(self.supervised_module(spec, runtime, snapshot, None))
2914            }
2915        }
2916    }
2917
2918    /// Supervise a configured module with its own health, drain, and crash
2919    /// budget. The restart policy is per-module because the config file is:
2920    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2921    /// module that is expensive to restart should not be forced onto the same
2922    /// budget as one that is cheap.
2923    pub fn supervise_configured_with_health(
2924        &self,
2925        spec: ModuleSpec,
2926        enabled: bool,
2927        health: HealthConfig,
2928        drain_timeout_ms: Option<u64>,
2929        restart_policy: RestartPolicy,
2930    ) -> Result<SupervisedModule, SuperviseError> {
2931        validate_spec(&spec)?;
2932        self.establish_identity(&spec);
2933
2934        let mut runtime = self.runtime_config();
2935        runtime.health = health.clone();
2936        runtime.restart_policy = restart_policy;
2937        if let Some(ms) = drain_timeout_ms {
2938            runtime.drain_timeout = Duration::from_millis(ms);
2939            *runtime
2940                .effective_drain_timeout
2941                .lock()
2942                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2943        }
2944        if !enabled {
2945            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2946            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2947        }
2948
2949        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2950        let spawned = spawn_child(
2951            &spec,
2952            runtime.connection_file_path.as_deref(),
2953            self.supervisor_handle.as_ref(),
2954            &runtime.stderr_ring,
2955            runtime.capture_logs_dir.as_deref(),
2956            &runtime.child_roster,
2957            #[cfg(target_os = "linux")]
2958            runtime.cgroup_placement.as_ref(),
2959        );
2960        #[cfg(test)]
2961        self.test_after_first_spawn.run(&spec.module_id);
2962        match spawned {
2963            Ok(child) => {
2964                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2965                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2966            }
2967            Err(err) => {
2968                if health.critical {
2969                    error!(
2970                        module_id = %spec.module_id,
2971                        program = %spec.program.display(),
2972                        error = %err,
2973                        "critical configured module failed to spawn; marking failed and alerting"
2974                    );
2975                } else {
2976                    error!(
2977                        module_id = %spec.module_id,
2978                        program = %spec.program.display(),
2979                        error = %err,
2980                        "configured module failed to spawn; marking failed and continuing"
2981                    );
2982                }
2983                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2984                Ok(self.supervised_module(spec, runtime, snapshot, None))
2985            }
2986        }
2987    }
2988
2989    fn runtime_config(&self) -> SupervisorRuntimeConfig {
2990        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2991        SupervisorRuntimeConfig {
2992            scheduled_respawn: Arc::default(),
2993            deferred_reload_reply: Arc::default(),
2994            restart_policy: self.restart_policy,
2995            drain_timeout: self.drain_timeout,
2996            // Shared with this module's roster copy: daemon shutdown waits on
2997            // each child for the module's own drain budget, as resolved now.
2998            child_roster: self
2999                .child_roster
3000                .for_module(Arc::clone(&effective_drain_timeout)),
3001            effective_drain_timeout,
3002            default_drain_timeout: self.drain_timeout,
3003            health: self.health.clone(),
3004            connection_file_path: self.connection_file_path.clone(),
3005            capture_logs_dir: self.capture_logs_dir.clone(),
3006            forwarding: self.forwarding.clone(),
3007            supervisor_handle: self.supervisor_handle.clone(),
3008            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3009            terminal_ring: Arc::new(Mutex::new(
3010                TerminalRing::new(
3011                    TerminalRingConfig::default(),
3012                    self.daemon_start_clock.started_at_ms(),
3013                )
3014                .with_start_clock(self.daemon_start_clock)
3015                .with_journal(self.terminal_journal.clone())
3016                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3017            )),
3018            spawn_events: self.spawn_events.clone(),
3019            #[cfg(target_os = "linux")]
3020            cgroup_placement: self.cgroup_placement.clone(),
3021            #[cfg(test)]
3022            test_seed_stale_facts_before_enable_spawn: false,
3023            #[cfg(test)]
3024            test_reload_exit_record_gate: None,
3025        }
3026    }
3027
3028    fn supervised_module(
3029        &self,
3030        spec: ModuleSpec,
3031        runtime: SupervisorRuntimeConfig,
3032        snapshot: SharedSnapshot,
3033        child: Option<SupervisedChild>,
3034    ) -> SupervisedModule {
3035        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3036            spec: spec.clone(),
3037            health: runtime.health.clone(),
3038        }));
3039        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3040        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3041        // The module's OWN policy, which may be its per-module config rather than
3042        // the supervisor-wide one; status must report the budget the supervise
3043        // loop actually enforces.
3044        let restart_policy = runtime.restart_policy;
3045        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3046        let (tx, rx) = mpsc::channel(4);
3047        let monitor = tokio::spawn(supervise_loop(
3048            spec.clone(),
3049            runtime,
3050            Arc::clone(&self.registry),
3051            Arc::clone(&self.process_liveness),
3052            Arc::clone(&snapshot),
3053            child,
3054            rx,
3055        ));
3056
3057        let module_id = spec.module_id.clone();
3058        let module = SupervisedModule {
3059            inner: Arc::new(SupervisedModuleInner {
3060                module_id: module_id.clone(),
3061                registry: Arc::clone(&self.registry),
3062                snapshot,
3063                configuration,
3064                stderr_ring,
3065                terminal_ring,
3066                commands: tx,
3067                monitor: Mutex::new(Some(monitor)),
3068                restart_policy,
3069                effective_drain_timeout,
3070                provenance_probe: self.provenance_probe.clone(),
3071            }),
3072        };
3073        // The identity gates and the configured mark were set by
3074        // `establish_identity` before any process was spawned; only the roster
3075        // entry waits for the module handle, which needs the spawned child.
3076        if let Some(supervisor_handle) = &self.supervisor_handle {
3077            supervisor_handle.insert(module.clone());
3078        }
3079        module
3080    }
3081}
3082
3083impl Default for Supervisor {
3084    fn default() -> Self {
3085        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3086    }
3087}
3088
3089/// Handle to one supervised child process.
3090#[derive(Clone)]
3091pub struct SupervisedModule {
3092    inner: Arc<SupervisedModuleInner>,
3093}
3094
3095struct SupervisedModuleInner {
3096    module_id: String,
3097    registry: Arc<Registry>,
3098    snapshot: SharedSnapshot,
3099    configuration: Arc<Mutex<SupervisedConfiguration>>,
3100    stderr_ring: Arc<Mutex<StderrRing>>,
3101    terminal_ring: Arc<Mutex<TerminalRing>>,
3102    commands: mpsc::Sender<SupervisorCommand>,
3103    monitor: Mutex<Option<JoinHandle<()>>>,
3104    /// Copied from the supervisor's runtime config at spawn so `status()` can
3105    /// report the restart budget without reaching back into the supervisor. The
3106    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3107    restart_policy: RestartPolicy,
3108    effective_drain_timeout: Arc<Mutex<Duration>>,
3109    provenance_probe: ExecutableIdentityProbe,
3110}
3111
3112impl fmt::Debug for SupervisedModule {
3113    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3114        f.debug_struct("SupervisedModule")
3115            .field("module_id", &self.inner.module_id)
3116            .field("status", &self.status())
3117            .finish_non_exhaustive()
3118    }
3119}
3120
3121impl SupervisedModule {
3122    pub fn module_id(&self) -> &str {
3123        &self.inner.module_id
3124    }
3125
3126    /// Test-only: put one probe miss on the streak, the way
3127    /// `handle_health_probe_failure` does, so tests can assert what a later
3128    /// event does to the streak without driving the whole probe loop.
3129    #[cfg(test)]
3130    pub(crate) fn record_health_probe_failure_for_test(
3131        &self,
3132        detail: &str,
3133    ) -> Result<(), SuperviseError> {
3134        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3135            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3136            state.health.detail = Some(detail.to_string());
3137        })
3138    }
3139
3140    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3141        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3142    }
3143
3144    /// The module's retained stderr, newest lines last.
3145    ///
3146    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3147    /// module, `supervisor.list` renders every module, and putting it in the
3148    /// shared snapshot would make each status read carry a payload almost nobody
3149    /// asked for. Callers that want the text ask for it.
3150    pub fn stderr_tail(
3151        &self,
3152        max_lines: Option<usize>,
3153        max_bytes: Option<usize>,
3154    ) -> StderrTailSnapshot {
3155        self.inner
3156            .stderr_ring
3157            .lock()
3158            .unwrap_or_else(|poisoned| poisoned.into_inner())
3159            .snapshot(max_lines, max_bytes)
3160    }
3161
3162    /// The module's bounded terminal history, oldest retained exit first.
3163    ///
3164    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3165    /// daemon whose in-memory history was necessarily reset.
3166    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3167        self.inner
3168            .terminal_ring
3169            .lock()
3170            .unwrap_or_else(|poisoned| poisoned.into_inner())
3171            .snapshot()
3172    }
3173
3174    /// Retained observations from the current ring and all journal generations.
3175    ///
3176    /// Blocking: this reads the journal files. Async callers use
3177    /// [`Self::read_durable_terminal_history`].
3178    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3179        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3180    }
3181
3182    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3183    /// read (up to every retained generation) never occupies a runtime worker.
3184    /// Fails only if the blocking task could not finish (runtime shutdown or a
3185    /// panic in the read).
3186    pub(crate) async fn read_durable_terminal_history(
3187        &self,
3188    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3189        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3190        let module_id = self.inner.module_id.clone();
3191        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3192            .await
3193    }
3194
3195    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3196        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3197    }
3198
3199    pub(crate) fn record_deliberate_severance(
3200        &self,
3201        identity: ProcessIdentity,
3202    ) -> Result<bool, SuperviseError> {
3203        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3204        if snapshot.pid != Some(identity.pid)
3205            || snapshot.process_start_time != Some(identity.start_time)
3206        {
3207            return Ok(false);
3208        }
3209        snapshot.deliberate_severance = Some(identity);
3210        Ok(true)
3211    }
3212
3213    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3214    ///
3215    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3216    /// does not produce reader-observability logs.
3217    pub(crate) fn status_for_control(
3218        &self,
3219        caller: &'static str,
3220    ) -> Result<ModuleStatus, SuperviseError> {
3221        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3222    }
3223
3224    fn status_with_snapshot_lock(
3225        &self,
3226        snapshot: &SharedSnapshot,
3227        caller: Option<&'static str>,
3228    ) -> Result<ModuleStatus, SuperviseError> {
3229        let mut guard = match caller {
3230            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3231            None => lock_snapshot(snapshot)?,
3232        };
3233        // Read the budget through the pruning path so a reader sees the same
3234        // in-window count the restart decision would use, not a stale total.
3235        let restart_count =
3236            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3237        let snapshot = guard.clone();
3238        drop(guard);
3239        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3240            SuperviseError::StatePoisoned {
3241                module_id: Some(self.inner.module_id.clone()),
3242            }
3243        })?;
3244        let registration_active = self
3245            .inner
3246            .registry
3247            .get_module(&self.inner.module_id)
3248            .map_err(SuperviseError::Registry)?
3249            .is_some();
3250        let protocol = snapshot
3251            .spawned_protocol
3252            .unwrap_or(self.declared_protocol()?);
3253        let running_process =
3254            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3255        // Registration is the difference between the two protocols and the only
3256        // one: a subc module that has not registered cannot serve a request even
3257        // though its process is up, and a `none` module never registers at all,
3258        // so requiring it there would pin `live` to false for the whole life of
3259        // a perfectly healthy process.
3260        let live = match protocol {
3261            ModuleProtocol::Subc => running_process && registration_active,
3262            ModuleProtocol::None => running_process,
3263        };
3264
3265        Ok(ModuleStatus {
3266            module_id: self.inner.module_id.clone(),
3267            state: snapshot.state,
3268            enabled: snapshot.enabled,
3269            process_alive: snapshot.process_alive,
3270            registration_active,
3271            protocol,
3272            live,
3273            restart_count,
3274            lifetime_restarts: snapshot.lifetime_restarts,
3275            spawn_generation: snapshot.spawn_generation,
3276            max_restarts: self.inner.restart_policy.max_restarts,
3277            restart_window: self.inner.restart_policy.window,
3278            drain_timeout,
3279            restart_backoff: self.inner.restart_policy.backoff,
3280            restart_max_backoff: self.inner.restart_policy.max_backoff,
3281            pid: snapshot.pid,
3282            spawned_at_ms: snapshot.spawned_at_ms,
3283            spawned_from: snapshot.spawned_from,
3284            process_start_time: snapshot.process_start_time,
3285            last_exit: snapshot.last_exit,
3286            health: snapshot.health,
3287        })
3288    }
3289
3290    #[cfg(test)]
3291    pub(crate) fn hold_snapshot_for_test(
3292        &self,
3293        acquired: std::sync::mpsc::Sender<()>,
3294        hold: Duration,
3295    ) -> std::thread::JoinHandle<()> {
3296        let snapshot = Arc::clone(&self.inner.snapshot);
3297        std::thread::spawn(move || {
3298            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3299            acquired
3300                .send(())
3301                .expect("test receiver waits for snapshot lock");
3302            std::thread::sleep(hold);
3303        })
3304    }
3305
3306    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3307        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3308            Ok(snapshot) => snapshot.clone(),
3309            Err(_) => {
3310                return subc_control::RunningImageAgreement::Unavailable {
3311                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3312                };
3313            }
3314        };
3315        self.inner
3316            .provenance_probe
3317            .observe(
3318                snapshot.pid,
3319                snapshot.spawned_from.as_deref(),
3320                snapshot.spawned_file_identity,
3321                snapshot.process_start_time,
3322            )
3323            .await
3324    }
3325
3326    /// Memory and CPU time of the module's current process, read now. Only the
3327    /// process the supervisor spawned is read, not processes it has started.
3328    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3329        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3330            Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
3331            Err(_) => {
3332                return subc_control::ChildResourceUsage::Unavailable {
3333                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3334                }
3335            }
3336        };
3337        crate::child_resources::read(pid, start_time)
3338    }
3339
3340    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3341        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3342        Ok(match snapshot.state {
3343            ModuleState::Restarting => true,
3344            ModuleState::Failed | ModuleState::Disabled => false,
3345            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3346        })
3347    }
3348
3349    #[cfg(test)]
3350    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3351        self.is_warming_with_snapshot_lock(None)
3352    }
3353
3354    pub(crate) fn is_warming_for_control(
3355        &self,
3356        caller: &'static str,
3357    ) -> Result<bool, SuperviseError> {
3358        self.is_warming_with_snapshot_lock(Some(caller))
3359    }
3360
3361    fn is_warming_with_snapshot_lock(
3362        &self,
3363        caller: Option<&'static str>,
3364    ) -> Result<bool, SuperviseError> {
3365        let snapshot = match caller {
3366            Some(caller) => {
3367                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3368            }
3369            None => lock_snapshot(&self.inner.snapshot)?,
3370        }
3371        .clone();
3372        Ok(matches!(
3373            snapshot.state,
3374            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3375        ))
3376    }
3377
3378    /// Drain the module and stop monitoring it.
3379    pub async fn drain(&self) -> Result<(), SuperviseError> {
3380        self.stop().await
3381    }
3382
3383    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3384        match self.state()? {
3385            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3386            ModuleState::Starting
3387            | ModuleState::Running
3388            | ModuleState::Unresponsive
3389            | ModuleState::Restarting
3390            | ModuleState::Draining
3391            | ModuleState::Disabled => {}
3392        }
3393
3394        let (reply_tx, reply_rx) = oneshot::channel();
3395        self.inner
3396            .commands
3397            .send(SupervisorCommand::Retire { reply: reply_tx })
3398            .await
3399            .map_err(|_| SuperviseError::CommandClosed {
3400                module_id: self.inner.module_id.clone(),
3401            })?;
3402        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3403            module_id: self.inner.module_id.clone(),
3404        })?
3405    }
3406
3407    pub async fn stop(&self) -> Result<(), SuperviseError> {
3408        match self.state()? {
3409            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3410            ModuleState::Starting
3411            | ModuleState::Running
3412            | ModuleState::Unresponsive
3413            | ModuleState::Restarting
3414            | ModuleState::Draining
3415            | ModuleState::Disabled => {}
3416        }
3417
3418        let (reply_tx, reply_rx) = oneshot::channel();
3419        self.inner
3420            .commands
3421            .send(SupervisorCommand::Drain { reply: reply_tx })
3422            .await
3423            .map_err(|_| SuperviseError::CommandClosed {
3424                module_id: self.inner.module_id.clone(),
3425            })?;
3426        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3427            module_id: self.inner.module_id.clone(),
3428        })?
3429    }
3430
3431    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3432        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3433        let (reply_tx, reply_rx) = oneshot::channel();
3434        self.inner
3435            .commands
3436            .send(SupervisorCommand::Restart {
3437                drain_timeout_ms,
3438                received_at_generation,
3439                queued_at: Instant::now(),
3440                reply: reply_tx,
3441            })
3442            .await
3443            .map_err(|_| SuperviseError::CommandClosed {
3444                module_id: self.inner.module_id.clone(),
3445            })?;
3446        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3447            module_id: self.inner.module_id.clone(),
3448        })?
3449    }
3450
3451    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3452    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3453    /// process then drains in the background of the supervise loop) or has
3454    /// failed, leaving the old process serving.
3455    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3456        let (reply_tx, reply_rx) = oneshot::channel();
3457        self.inner
3458            .commands
3459            .send(SupervisorCommand::Swap {
3460                ready_timeout,
3461                reply: reply_tx,
3462            })
3463            .await
3464            .map_err(|_| SuperviseError::CommandClosed {
3465                module_id: self.inner.module_id.clone(),
3466            })?;
3467        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3468            module_id: self.inner.module_id.clone(),
3469        })?
3470    }
3471
3472    pub async fn reload(&self) -> Result<(), SuperviseError> {
3473        let (reply_tx, reply_rx) = oneshot::channel();
3474        self.inner
3475            .commands
3476            .send(SupervisorCommand::Reload { reply: reply_tx })
3477            .await
3478            .map_err(|_| SuperviseError::CommandClosed {
3479                module_id: self.inner.module_id.clone(),
3480            })?;
3481        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3482            module_id: self.inner.module_id.clone(),
3483        })?
3484    }
3485
3486    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3487        let (reply_tx, reply_rx) = oneshot::channel();
3488        self.inner
3489            .commands
3490            .send(SupervisorCommand::SetEnabled {
3491                enabled,
3492                reply: reply_tx,
3493            })
3494            .await
3495            .map_err(|_| SuperviseError::CommandClosed {
3496                module_id: self.inner.module_id.clone(),
3497            })?;
3498        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3499            module_id: self.inner.module_id.clone(),
3500        })?
3501    }
3502
3503    /// The current process's protocol, or the configured protocol when down.
3504    /// A rescan stores the next launch spec without changing how an existing
3505    /// process registers, serves routes, is probed, or exits.
3506    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3507        let configured = self
3508            .inner
3509            .configuration
3510            .lock()
3511            .map_err(|_| SuperviseError::StatePoisoned {
3512                module_id: Some(self.inner.module_id.clone()),
3513            })?
3514            .spec
3515            .protocol;
3516        let state = lock_snapshot(&self.inner.snapshot)?;
3517        Ok(state.spawned_protocol.unwrap_or(configured))
3518    }
3519
3520    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3521        let configuration =
3522            self.inner
3523                .configuration
3524                .lock()
3525                .map_err(|_| SuperviseError::StatePoisoned {
3526                    module_id: Some(self.inner.module_id.clone()),
3527                })?;
3528        Ok((configuration.spec.clone(), configuration.health.clone()))
3529    }
3530
3531    /// Replace this module's launch spec, keeping its health and drain policy,
3532    /// the way a rescan does for a changed config entry. The running process is
3533    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3534    #[cfg(any(test, feature = "test-support"))]
3535    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3536        let (_, health) = self.configuration()?;
3537        let drain_timeout_ms = u64::try_from(
3538            self.inner
3539                .effective_drain_timeout
3540                .lock()
3541                .unwrap_or_else(|poisoned| poisoned.into_inner())
3542                .as_millis(),
3543        )
3544        .ok();
3545        self.update_configuration(spec, health, drain_timeout_ms)
3546            .await
3547    }
3548
3549    pub(crate) async fn update_configuration(
3550        &self,
3551        spec: ModuleSpec,
3552        health: HealthConfig,
3553        drain_timeout_ms: Option<u64>,
3554    ) -> Result<(), SuperviseError> {
3555        if spec.module_id != self.inner.module_id {
3556            return Err(SuperviseError::InvalidSpec {
3557                reason: "a supervised module's module_id cannot be changed".to_string(),
3558            });
3559        }
3560        validate_spec(&spec)?;
3561        let (reply_tx, reply_rx) = oneshot::channel();
3562        self.inner
3563            .commands
3564            .send(SupervisorCommand::UpdateConfiguration {
3565                spec: spec.clone(),
3566                health: health.clone(),
3567                drain_timeout_ms,
3568                reply: reply_tx,
3569            })
3570            .await
3571            .map_err(|_| SuperviseError::CommandClosed {
3572                module_id: self.inner.module_id.clone(),
3573            })?;
3574        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3575            module_id: self.inner.module_id.clone(),
3576        })?;
3577        let mut configuration =
3578            self.inner
3579                .configuration
3580                .lock()
3581                .map_err(|_| SuperviseError::StatePoisoned {
3582                    module_id: Some(self.inner.module_id.clone()),
3583                })?;
3584        configuration.spec = spec;
3585        configuration.health = health;
3586        Ok(())
3587    }
3588}
3589
3590impl Drop for SupervisedModuleInner {
3591    fn drop(&mut self) {
3592        let Ok(mut monitor) = self.monitor.lock() else {
3593            return;
3594        };
3595        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3596            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3597                state.state = ModuleState::Stopped;
3598                clear_current_process_facts(state);
3599            });
3600            monitor.abort();
3601        }
3602        let _ = monitor.take();
3603    }
3604}
3605
3606#[derive(Debug)]
3607enum SupervisorCommand {
3608    Drain {
3609        reply: oneshot::Sender<Result<(), SuperviseError>>,
3610    },
3611    Retire {
3612        reply: oneshot::Sender<Result<(), SuperviseError>>,
3613    },
3614    Restart {
3615        /// Operator override for this one restart's drain budget, in ms. `None`
3616        /// uses the module's configured/default budget; `Some(0)` cuts
3617        /// immediately (wedge bounce: a stuck request never settles, so
3618        /// waiting only delays recovery).
3619        drain_timeout_ms: Option<u64>,
3620        /// The module's `spawn_generation` when the request was received, before
3621        /// it waited in the command queue. A queued restart whose module has
3622        /// since spawned a newer process is already satisfied (see the handler).
3623        received_at_generation: u64,
3624        /// When the request entered the command queue, so the handler can log
3625        /// how long it waited behind the loop's other work.
3626        queued_at: Instant,
3627        reply: oneshot::Sender<Result<(), SuperviseError>>,
3628    },
3629    Reload {
3630        reply: oneshot::Sender<Result<(), SuperviseError>>,
3631    },
3632    SetEnabled {
3633        enabled: bool,
3634        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3635    },
3636    UpdateConfiguration {
3637        spec: ModuleSpec,
3638        health: HealthConfig,
3639        /// Per-module drain override from the new config; `None` re-resolves to
3640        /// the supervisor-wide default.
3641        drain_timeout_ms: Option<u64>,
3642        reply: oneshot::Sender<()>,
3643    },
3644    Swap {
3645        /// How long the candidate may take to register and declare itself
3646        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3647        ready_timeout: Option<Duration>,
3648        /// Answered at cutover or failure; the incumbent's drain follows.
3649        reply: oneshot::Sender<Result<(), SuperviseError>>,
3650    },
3651}
3652
3653#[derive(Debug)]
3654pub enum SuperviseError {
3655    InvalidSpec {
3656        reason: String,
3657    },
3658    Spawn {
3659        program: PathBuf,
3660        source: io::Error,
3661        cgroup_path: Option<PathBuf>,
3662    },
3663    Cgroup {
3664        module_id: String,
3665        source: io::Error,
3666    },
3667    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3668    /// than spawn a reserved module without its identity binding.
3669    LaunchNonce {
3670        reason: String,
3671    },
3672    Wait {
3673        module_id: String,
3674        source: io::Error,
3675    },
3676    Kill {
3677        module_id: String,
3678        source: io::Error,
3679    },
3680    Forwarding(ForwardingError),
3681    Registry(RegistryError),
3682    ReloadUnavailable {
3683        module_id: String,
3684        reason: String,
3685    },
3686    /// An operator restart/reload was requested for a module that is currently
3687    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3688    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3689    /// by a restart, so these commands are rejected instead of re-enabling it.
3690    Disabled {
3691        module_id: String,
3692    },
3693    ReloadFailed {
3694        module_id: String,
3695        reason: String,
3696    },
3697    RegistrationStillActive {
3698        module_id: String,
3699        waited: Duration,
3700    },
3701    StatePoisoned {
3702        module_id: Option<String>,
3703    },
3704    CommandClosed {
3705        module_id: String,
3706    },
3707    /// A restart or reload arrived while a swap's candidate was warming. The
3708    /// swap owns the module until it cuts over or fails; a stop or disable
3709    /// would have aborted it instead.
3710    SwapInProgress {
3711        module_id: String,
3712    },
3713    /// A swap was refused before anything was spawned.
3714    SwapRefused {
3715        module_id: String,
3716        reason: SwapRefusal,
3717    },
3718    /// A swap spawned a candidate and gave up on it. The candidate has been
3719    /// killed and its slot freed; the incumbent was left serving and was never
3720    /// drained, except in the one `CutoverLost` case described on that arm.
3721    SwapFailed {
3722        module_id: String,
3723        arm: SwapFailureArm,
3724        detail: String,
3725        /// How the candidate exited, when it exited on its own before the
3726        /// supervisor gave up on it.
3727        candidate_exit: Option<ExitReport>,
3728    },
3729}
3730
3731/// Why a swap was refused before a candidate was spawned.
3732#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3733pub enum SwapRefusal {
3734    /// The module's config does not declare `overlap: "safe"`.
3735    OverlapExclusive,
3736    /// The module is not registered, so there is no incumbent to keep serving
3737    /// and nothing a swap would improve on; a plain restart is the tool.
3738    NotRegistered,
3739    /// The module does not speak the subc wire, so a candidate could never
3740    /// register or declare itself ready.
3741    ProtocolNone,
3742    /// The supervisor lacks the forwarding table (to cut routes over) or the
3743    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3744    NotConfigured,
3745    /// A swap is already open for this module.
3746    AlreadySwapping,
3747}
3748
3749impl SwapRefusal {
3750    pub fn as_str(self) -> &'static str {
3751        match self {
3752            Self::OverlapExclusive => "overlap_exclusive",
3753            Self::NotRegistered => "not_registered",
3754            Self::ProtocolNone => "protocol_none",
3755            Self::NotConfigured => "not_configured",
3756            Self::AlreadySwapping => "already_swapping",
3757        }
3758    }
3759}
3760
3761/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3762/// serving and undrained; see `CutoverLost`.
3763#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3764pub enum SwapFailureArm {
3765    /// The candidate process could not be started.
3766    SpawnFailed,
3767    /// The candidate did not register within the readiness budget.
3768    NeverRegistered,
3769    /// The candidate registered but did not declare itself ready in time.
3770    NeverReady,
3771    /// The candidate exited before cutover.
3772    CandidateExited,
3773    /// The candidate declared itself ready but failed its health probe.
3774    CandidateUnhealthy,
3775    /// An operator stop, disable or retire arrived while the candidate warmed.
3776    /// The candidate was killed and the operator's command then carried out on
3777    /// the incumbent.
3778    Interrupted,
3779    /// The candidate's connection closed at the moment of cutover. If it
3780    /// closed before forwarding moved, the incumbent is untouched. If it closed
3781    /// between the forwarding and registry halves of cutover, forwarding can no
3782    /// longer route to the incumbent, so the module is restarted plainly.
3783    CutoverLost,
3784}
3785
3786impl SwapFailureArm {
3787    pub fn as_str(self) -> &'static str {
3788        match self {
3789            Self::SpawnFailed => "spawn_failed",
3790            Self::NeverRegistered => "never_registered",
3791            Self::NeverReady => "never_ready",
3792            Self::CandidateExited => "candidate_exited",
3793            Self::CandidateUnhealthy => "candidate_unhealthy",
3794            Self::Interrupted => "interrupted",
3795            Self::CutoverLost => "cutover_lost",
3796        }
3797    }
3798}
3799
3800impl fmt::Display for SuperviseError {
3801    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3802        match self {
3803            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3804            Self::Spawn {
3805                program,
3806                source,
3807                cgroup_path: Some(cgroup_path),
3808            } => write!(
3809                f,
3810                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3811                cgroup_path.display(),
3812                program.display()
3813            ),
3814            Self::Spawn {
3815                program,
3816                source,
3817                cgroup_path: None,
3818            } => write!(
3819                f,
3820                "failed to spawn module '{}': {source}",
3821                program.display()
3822            ),
3823            Self::Cgroup { module_id, source } => {
3824                write!(
3825                    f,
3826                    "failed to prepare cgroup for module '{module_id}': {source}"
3827                )
3828            }
3829            Self::LaunchNonce { reason } => {
3830                write!(
3831                    f,
3832                    "failed to generate reserved-module launch nonce: {reason}"
3833                )
3834            }
3835            Self::Wait { module_id, source } => {
3836                write!(f, "failed to wait for module '{module_id}': {source}")
3837            }
3838            Self::Kill { module_id, source } => {
3839                write!(f, "failed to kill module '{module_id}': {source}")
3840            }
3841            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3842            Self::Registry(err) => write!(f, "registry error: {err}"),
3843            Self::ReloadUnavailable { module_id, reason } => {
3844                write!(f, "reload unavailable for module '{module_id}': {reason}")
3845            }
3846            Self::Disabled { module_id } => {
3847                write!(
3848                    f,
3849                    "module '{module_id}' is disabled; enable it before restart or reload"
3850                )
3851            }
3852            Self::ReloadFailed { module_id, reason } => {
3853                write!(f, "reload failed for module '{module_id}': {reason}")
3854            }
3855            Self::RegistrationStillActive { module_id, waited } => write!(
3856                f,
3857                "module '{module_id}' registration remained active after waiting {waited:?}"
3858            ),
3859            Self::StatePoisoned { module_id } => match module_id {
3860                Some(module_id) => {
3861                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3862                }
3863                None => write!(f, "supervisor state was poisoned"),
3864            },
3865            Self::CommandClosed { module_id } => {
3866                write!(
3867                    f,
3868                    "supervisor command channel for module '{module_id}' is closed"
3869                )
3870            }
3871            Self::SwapInProgress { module_id } => write!(
3872                f,
3873                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3874            ),
3875            Self::SwapRefused { module_id, reason } => match reason {
3876                SwapRefusal::OverlapExclusive => write!(
3877                    f,
3878                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3879                ),
3880                SwapRefusal::NotRegistered => write!(
3881                    f,
3882                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3883                ),
3884                SwapRefusal::ProtocolNone => write!(
3885                    f,
3886                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3887                ),
3888                SwapRefusal::NotConfigured => write!(
3889                    f,
3890                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3891                ),
3892                SwapRefusal::AlreadySwapping => {
3893                    write!(f, "module '{module_id}' is already being swapped")
3894                }
3895            },
3896            Self::SwapFailed {
3897                module_id,
3898                arm,
3899                detail,
3900                ..
3901            } => write!(
3902                f,
3903                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3904                arm.as_str()
3905            ),
3906        }
3907    }
3908}
3909
3910impl Error for SuperviseError {
3911    fn source(&self) -> Option<&(dyn Error + 'static)> {
3912        match self {
3913            Self::Spawn { source, .. }
3914            | Self::Cgroup { source, .. }
3915            | Self::Wait { source, .. }
3916            | Self::Kill { source, .. } => Some(source),
3917            Self::Forwarding(err) => Some(err),
3918            Self::Registry(err) => Some(err),
3919            Self::LaunchNonce { .. }
3920            | Self::InvalidSpec { .. }
3921            | Self::ReloadUnavailable { .. }
3922            | Self::Disabled { .. }
3923            | Self::ReloadFailed { .. }
3924            | Self::RegistrationStillActive { .. }
3925            | Self::StatePoisoned { .. }
3926            | Self::CommandClosed { .. }
3927            | Self::SwapInProgress { .. }
3928            | Self::SwapRefused { .. }
3929            | Self::SwapFailed { .. } => None,
3930        }
3931    }
3932}
3933
3934pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3935    if spec.module_id.trim().is_empty() {
3936        return Err(SuperviseError::InvalidSpec {
3937            reason: "module_id must not be empty".to_string(),
3938        });
3939    }
3940
3941    Ok(())
3942}
3943
3944#[derive(Debug, Default)]
3945struct HealthProbeRuntime {
3946    configured_health: Option<HealthConfig>,
3947    registered_connection: Option<crate::ConnectionId>,
3948    advertised: bool,
3949    next_probe_at: Option<Instant>,
3950    probe_index: u64,
3951}
3952
3953fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
3954    lock_snapshot(snapshot)
3955        .ok()
3956        .and_then(|state| state.spawned_protocol)
3957        .unwrap_or(spec.protocol)
3958}
3959
3960impl HealthProbeRuntime {
3961    fn refresh_registration(
3962        &mut self,
3963        spec: &ModuleSpec,
3964        runtime: &SupervisorRuntimeConfig,
3965        registry: &Registry,
3966        snapshot: &SharedSnapshot,
3967    ) {
3968        if self.configured_health.as_ref() != Some(&runtime.health) {
3969            self.configured_health = Some(runtime.health.clone());
3970            self.next_probe_at = None;
3971            self.registered_connection = None;
3972            self.probe_index = 0;
3973        }
3974        // A non-wire process never registers. Only an explicitly configured
3975        // HTTP endpoint can arm its health probe; an absent HELLO is not a
3976        // health failure for that kind of process.
3977        if running_protocol(spec, snapshot) == ModuleProtocol::None {
3978            self.registered_connection = None;
3979            self.advertised = runtime.health.http.is_some();
3980            if !self.advertised {
3981                self.next_probe_at = None;
3982                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3983                    state.health = ModuleHealthStatus::default();
3984                });
3985            } else if self.next_probe_at.is_none() {
3986                self.next_probe_at = Some(
3987                    Instant::now()
3988                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3989                );
3990            }
3991            return;
3992        }
3993
3994        let registration = match registry.get_module(&spec.module_id) {
3995            Ok(registration) => registration,
3996            Err(err) => {
3997                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3998                self.advertised = false;
3999                self.next_probe_at = None;
4000                return;
4001            }
4002        };
4003
4004        let Some(registration) = registration else {
4005            self.registered_connection = None;
4006            self.advertised = false;
4007            self.next_probe_at = None;
4008            return;
4009        };
4010
4011        let advertised = registration
4012            .control_ops
4013            .iter()
4014            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4015        if !advertised {
4016            self.registered_connection = Some(registration.connection_id);
4017            self.advertised = false;
4018            self.next_probe_at = None;
4019            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4020                state.health.status = SupervisorHealthStatus::Unknown;
4021                state.health.consecutive_failures = 0;
4022                state.health.last_probe_ms = None;
4023                state.health.detail = None;
4024                state.health.metrics = None;
4025            });
4026            return;
4027        }
4028
4029        let reregistered = self.registered_connection != Some(registration.connection_id);
4030        self.registered_connection = Some(registration.connection_id);
4031        self.advertised = true;
4032        if reregistered || self.next_probe_at.is_none() {
4033            self.probe_index = 0;
4034            self.next_probe_at = Some(
4035                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4036            );
4037            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4038                state.health.status = SupervisorHealthStatus::Unknown;
4039                state.health.consecutive_failures = 0;
4040                state.health.detail = None;
4041                state.health.metrics = None;
4042            });
4043        }
4044    }
4045
4046    fn wake_after(&self) -> Duration {
4047        if !self.advertised {
4048            return REGISTRY_RELEASE_POLL;
4049        }
4050        self.next_probe_at
4051            .map(|next| next.saturating_duration_since(Instant::now()))
4052            .unwrap_or(REGISTRY_RELEASE_POLL)
4053    }
4054
4055    fn due(&self) -> bool {
4056        self.advertised
4057            && self
4058                .next_probe_at
4059                .is_some_and(|next| Instant::now() >= next)
4060    }
4061
4062    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4063        self.probe_index = self.probe_index.wrapping_add(1);
4064        self.next_probe_at = Some(
4065            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4066        );
4067    }
4068}
4069
4070/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4071///
4072/// This was a struct with a single `message: String`, and every one of the
4073/// fifteen construction sites collapsed into it. Each site knows exactly what it
4074/// saw -- the lane is gone, the module did not answer in time, the module
4075/// answered with the wrong thing -- and `handle_health_probe_failure` then
4076/// treated all of them identically: increment a counter, compare to a threshold,
4077/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4078/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4079///
4080/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4081///
4082/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4083///   answer on it again.
4084/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4085///   AND with a perfectly healthy one that lost a CPU race -- which is what
4086///   happens under machine load, and is how this supervisor killed a healthy
4087///   module three times in one day.
4088/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4089///   Restarting on it is defensible, but it is not the silence case and should
4090///   never be counted as one.
4091/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4092///   anything, so it cannot be evidence about the module at all.
4093///
4094/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4095/// one that fires most often, and while every variant collapsed into one string
4096/// it carried the same weight as the strongest.
4097///
4098/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4099/// DESIGN and a reader stopping at it gets the build backwards: the restart
4100/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4101/// probes still increment the failure streak and drive escalation at the
4102/// threshold (see `is_proof_of_death` below for why that is deliberate and
4103/// what gates the change). Absence of evidence restarts modules today.
4104#[derive(Debug)]
4105enum HealthProbeEvidence {
4106    /// The module's control lane is gone. Proof of death.
4107    LaneDead,
4108    /// No reply within the deadline. Proves nothing about the module's state.
4109    NoAnswer,
4110    /// The module replied, but not with a usable health report. Proves it is alive.
4111    BadAnswer,
4112    /// The daemon could not ask. Says nothing about the module.
4113    Misconfigured,
4114}
4115
4116#[derive(Debug)]
4117struct HealthProbeError {
4118    evidence: HealthProbeEvidence,
4119    message: String,
4120}
4121
4122impl HealthProbeError {
4123    fn lane_dead(message: impl Into<String>) -> Self {
4124        Self::with(HealthProbeEvidence::LaneDead, message)
4125    }
4126
4127    fn no_answer(message: impl Into<String>) -> Self {
4128        Self::with(HealthProbeEvidence::NoAnswer, message)
4129    }
4130
4131    fn bad_answer(message: impl Into<String>) -> Self {
4132        Self::with(HealthProbeEvidence::BadAnswer, message)
4133    }
4134
4135    fn misconfigured(message: impl Into<String>) -> Self {
4136        Self::with(HealthProbeEvidence::Misconfigured, message)
4137    }
4138
4139    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4140        Self {
4141            evidence,
4142            message: message.into(),
4143        }
4144    }
4145
4146    /// Whether this observation is proof the module cannot serve.
4147    ///
4148    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4149    /// variant that fires under CPU starvation, and treating it as proof is the
4150    /// defect this enum exists to make impossible to reintroduce silently.
4151    ///
4152    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4153    /// to restart also needs a bound for the case it excludes -- a genuinely
4154    /// wedged module, alive but never answering -- and that bound must come from
4155    /// the distribution of real late-answer latencies, which nothing measures
4156    /// yet. Landing the classification first makes the later change a one-line
4157    /// decision against evidence that already exists, rather than two unproven
4158    /// changes at once.
4159    #[allow(dead_code)]
4160    fn is_proof_of_death(&self) -> bool {
4161        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4162    }
4163
4164    /// Short stable label for logs and the health snapshot.
4165    ///
4166    /// An operator reading `ck health` currently cannot tell "the module is gone"
4167    /// from "the module did not answer in five seconds", because both render as
4168    /// prose in the same field. These labels are what make the two
4169    /// distinguishable at a glance, and they are what a later restart-policy
4170    /// change will be argued from.
4171    fn label(&self) -> &'static str {
4172        match self.evidence {
4173            HealthProbeEvidence::LaneDead => "lane-dead",
4174            HealthProbeEvidence::NoAnswer => "no-answer",
4175            HealthProbeEvidence::BadAnswer => "bad-answer",
4176            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4177        }
4178    }
4179}
4180
4181impl fmt::Display for HealthProbeError {
4182    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4183        f.write_str(&self.message)
4184    }
4185}
4186
4187async fn run_health_probe_cycle(
4188    spec: &ModuleSpec,
4189    runtime: &SupervisorRuntimeConfig,
4190    registry: &Registry,
4191    process_liveness: &SupervisorProcessLiveness,
4192    snapshot: &SharedSnapshot,
4193    child: &mut Option<SupervisedChild>,
4194) {
4195    let now_ms = unix_ms_now();
4196    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4197        .then_some(runtime.health.http.as_deref())
4198        .flatten();
4199    let result = match http {
4200        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4201        None => probe_module_health(&spec.module_id, runtime, None).await,
4202    };
4203    match result {
4204        Ok(report) => {
4205            handle_health_report(
4206                spec,
4207                runtime,
4208                registry,
4209                process_liveness,
4210                snapshot,
4211                child,
4212                report,
4213                now_ms,
4214            )
4215            .await;
4216        }
4217        Err(err) => {
4218            if http.is_some() {
4219                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4220                    state.health.status = SupervisorHealthStatus::Failing;
4221                });
4222            }
4223            handle_health_probe_failure(
4224                spec,
4225                runtime,
4226                registry,
4227                process_liveness,
4228                snapshot,
4229                child,
4230                err,
4231                now_ms,
4232            )
4233            .await;
4234        }
4235    }
4236}
4237
4238pub(crate) struct HttpProbeTarget<'a> {
4239    address: std::net::SocketAddr,
4240    localhost: bool,
4241    authority: &'a str,
4242    path: String,
4243}
4244
4245/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4246/// or TLS. A URL cannot turn a local health check into an outbound connection.
4247pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4248    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4249        return Err("must not contain whitespace, controls, or a fragment".into());
4250    }
4251    let rest = url
4252        .strip_prefix("http://")
4253        .ok_or("must use plain http://")?;
4254    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4255    let (authority, suffix) = rest.split_at(split);
4256    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4257        ("::1", rest)
4258    } else {
4259        let split = authority.find(':').unwrap_or(authority.len());
4260        authority.split_at(split)
4261    };
4262    let ip: std::net::IpAddr = match host {
4263        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4264        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4265        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4266    };
4267    let port = if port.is_empty() {
4268        80
4269    } else {
4270        port.strip_prefix(':')
4271            .and_then(|p| p.parse::<u16>().ok())
4272            .filter(|p| *p > 0)
4273            .ok_or("must have a valid nonzero TCP port")?
4274    };
4275    let path = if suffix.is_empty() {
4276        "/".into()
4277    } else if suffix.starts_with('?') {
4278        format!("/{suffix}")
4279    } else {
4280        suffix.into()
4281    };
4282    Ok(HttpProbeTarget {
4283        address: std::net::SocketAddr::new(ip, port),
4284        localhost: host == "localhost",
4285        authority,
4286        path,
4287    })
4288}
4289
4290async fn probe_http_health(
4291    url: &str,
4292    deadline: Duration,
4293) -> Result<HealthReport, HealthProbeError> {
4294    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4295    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4296    // Keep partial diagnostics outside the timed future so cancellation does
4297    // not discard a status line or body bytes already received.
4298    let mut response_status = String::new();
4299    let mut body = Vec::new();
4300    let probe = async {
4301        // Resolve localhost ourselves so a hosts-file override cannot turn
4302        // this into an outbound request, while IPv6-only local servers work.
4303        let connection = match tokio::net::TcpStream::connect(target.address).await {
4304            Err(_) if target.localhost => {
4305                tokio::net::TcpStream::connect((
4306                    std::net::Ipv6Addr::LOCALHOST,
4307                    target.address.port(),
4308                ))
4309                .await
4310            }
4311            result => result,
4312        };
4313        let mut stream = connection.map_err(|error| {
4314            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4315        })?;
4316        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4317            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4318        let mut reader = BufReader::new(stream);
4319        let mut budget = 16 * 1024;
4320        let status = http_line(&mut reader, &mut budget).await?;
4321        let mut words = status.split_ascii_whitespace();
4322        let version = words.next();
4323        let code = words
4324            .next()
4325            .filter(|word| word.len() == 3)
4326            .and_then(|word| word.parse::<u16>().ok());
4327        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4328            || !code.is_some_and(|code| (100..600).contains(&code))
4329        {
4330            return Err(HealthProbeError::bad_answer(format!(
4331                "invalid HTTP status: {status}"
4332            )));
4333        }
4334        let code = code.expect("validated status code");
4335        response_status = status.clone();
4336        let mut length = None;
4337        let mut chunked = false;
4338        loop {
4339            let line = http_line(&mut reader, &mut budget).await?;
4340            if line.is_empty() {
4341                break;
4342            }
4343            if let Some((name, value)) = line.split_once(':') {
4344                if name.eq_ignore_ascii_case("content-length") {
4345                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4346                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4347                    })?);
4348                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4349                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4350                }
4351            }
4352        }
4353        if chunked {
4354            while body.len() < 200 {
4355                let line = http_line(&mut reader, &mut budget).await?;
4356                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4357                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4358                if size == 0 {
4359                    break;
4360                }
4361                let count = size.min((200 - body.len()) as u64) as usize;
4362                let start = body.len();
4363                (&mut reader)
4364                    .take(count as u64)
4365                    .read_to_end(&mut body)
4366                    .await
4367                    .map_err(|error| {
4368                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4369                    })?;
4370                if body.len() - start != count {
4371                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4372                }
4373                if size > count as u64 || body.len() == 200 {
4374                    break;
4375                }
4376                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4377                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4378                }
4379            }
4380        } else {
4381            reader
4382                .take(length.unwrap_or(200).min(200))
4383                .read_to_end(&mut body)
4384                .await
4385                .map_err(|error| {
4386                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4387                })?;
4388        }
4389        if (200..300).contains(&code) {
4390            Ok(HealthReport::ok())
4391        } else {
4392            Err(HealthProbeError::bad_answer(
4393                "HTTP health endpoint returned non-2xx",
4394            ))
4395        }
4396    };
4397    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4398        Err(HealthProbeError::no_answer(format!(
4399            "HTTP probe timed out after {deadline:?}"
4400        )))
4401    });
4402    if let Err(error) = &mut result {
4403        if !response_status.is_empty() {
4404            error.message = format!(
4405                "{}; {response_status}: {}",
4406                error.message,
4407                String::from_utf8_lossy(&body)
4408            );
4409        }
4410    }
4411    result
4412}
4413
4414async fn http_line(
4415    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4416    remaining: &mut usize,
4417) -> Result<String, HealthProbeError> {
4418    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4419    let mut line = Vec::new();
4420    (&mut *reader)
4421        .take(*remaining as u64)
4422        .read_until(b'\n', &mut line)
4423        .await
4424        .map_err(|error| {
4425            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4426        })?;
4427    *remaining -= line.len();
4428    if !line.ends_with(b"\r\n") {
4429        return Err(HealthProbeError::bad_answer(
4430            "HTTP headers are incomplete or exceed 16 KiB",
4431        ));
4432    }
4433    line.truncate(line.len() - 2);
4434    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4435}
4436
4437async fn probe_module_health(
4438    module_id: &str,
4439    runtime: &SupervisorRuntimeConfig,
4440    drain_deadline: Option<Instant>,
4441) -> Result<HealthReport, HealthProbeError> {
4442    let Some(forwarding) = runtime.forwarding.as_ref() else {
4443        return Err(HealthProbeError::misconfigured(
4444            "supervisor was not configured with a forwarding table",
4445        ));
4446    };
4447    let probe_started_at = Instant::now();
4448    let mut deadline = probe_started_at + runtime.health.deadline;
4449    if let Some(drain_deadline) = drain_deadline {
4450        deadline = deadline.min(drain_deadline);
4451    }
4452    let pending = if drain_deadline.is_some() {
4453        forwarding.begin_drain_health_probe_rpc_for(
4454            module_id,
4455            MODULE_CONTROL_OP_HEALTH_CHECK,
4456            probe_started_at,
4457            deadline,
4458        )
4459    } else {
4460        forwarding.begin_health_probe_rpc_for(
4461            module_id,
4462            MODULE_CONTROL_OP_HEALTH_CHECK,
4463            probe_started_at,
4464            deadline,
4465        )
4466    }
4467    .map_err(|err| {
4468        // The endpoint is not registered, so there is no live control lane to
4469        // ask. That is the module being absent, not slow.
4470        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4471    })?;
4472    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4473}
4474
4475/// [`probe_module_health`] for one endpoint rather than the id's active one.
4476///
4477/// A swap probes two processes that no by-id lookup reaches: its candidate
4478/// before cutover, and its superseded incumbent (for busy gauges) while the
4479/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4480/// bounds the by-id drain probe.
4481async fn probe_endpoint_health(
4482    endpoint: crate::ModuleEndpointId,
4483    runtime: &SupervisorRuntimeConfig,
4484    deadline_cap: Option<Instant>,
4485) -> Result<HealthReport, HealthProbeError> {
4486    let Some(forwarding) = runtime.forwarding.as_ref() else {
4487        return Err(HealthProbeError::misconfigured(
4488            "supervisor was not configured with a forwarding table",
4489        ));
4490    };
4491    let probe_started_at = Instant::now();
4492    let mut deadline = probe_started_at + runtime.health.deadline;
4493    if let Some(cap) = deadline_cap {
4494        deadline = deadline.min(cap);
4495    }
4496    let pending = forwarding
4497        .begin_endpoint_health_probe_rpc_for(
4498            endpoint,
4499            MODULE_CONTROL_OP_HEALTH_CHECK,
4500            probe_started_at,
4501            deadline,
4502        )
4503        .map_err(|err| {
4504            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4505        })?;
4506    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4507}
4508
4509/// Send a begun health probe and classify its answer.
4510async fn await_health_probe(
4511    forwarding: &ForwardingTable,
4512    pending: PendingModuleControlRpc,
4513    deadline: Instant,
4514    probe_budget: Duration,
4515) -> Result<HealthReport, HealthProbeError> {
4516    let PendingModuleControlRpc {
4517        endpoint,
4518        module_sink,
4519        negotiated_ver,
4520        corr,
4521        receiver,
4522    } = pending;
4523    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4524        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4525    })?;
4526    let frame = Frame::build_with_version(
4527        negotiated_ver,
4528        FrameType::Request,
4529        control_flags(),
4530        0,
4531        0,
4532        corr,
4533        body,
4534    )
4535    .map_err(|err| {
4536        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4537    })?;
4538
4539    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4540    // blocks waiting for capacity when the module's egress queue is full, and an
4541    // unbounded await here freezes the whole supervision actor (it stops polling
4542    // Child::wait and supervisor commands), making the module unrecoverable
4543    // in-band. On timeout the probe fails like any transport failure.
4544    match timeout_at(deadline, module_sink.send(frame)).await {
4545        Ok(Ok(())) => {}
4546        Ok(Err(err)) => {
4547            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4548            // A closed sink means the module's egress channel is gone -- the
4549            // receiving half is dropped when its connection tears down. Proof.
4550            return Err(HealthProbeError::lane_dead(format!(
4551                "failed to send health.check: {err}"
4552            )));
4553        }
4554        Err(_elapsed) => {
4555            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4556            // A full egress queue means the module is not draining its socket, which
4557            // is consistent with a wedged module AND with one whose reader is merely
4558            // starved. Silence, not proof.
4559            return Err(HealthProbeError::no_answer(
4560                "health.check send timed out before enqueue (module egress full)",
4561            ));
4562        }
4563    }
4564
4565    match timeout_at(deadline, receiver).await {
4566        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4567        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4568        // and those prove it is alive even though the probe failed.
4569        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4570            response.health_report().ok_or_else(|| {
4571                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4572            })
4573        }
4574        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4575            format!("health.check rejected: {}", body.message),
4576        )),
4577        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4578            Err(HealthProbeError::lane_dead(message))
4579        }
4580        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4581            Err(HealthProbeError::bad_answer(message))
4582        }
4583        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4584            Err(HealthProbeError::bad_answer(format!(
4585                "expected module-control op '{expected}', got '{actual}'"
4586            )))
4587        }
4588        // A reply that crosses the deadline before this waiter observes it is
4589        // still proof of life. The forwarding path records its end-to-end latency
4590        // before delivering this classification.
4591        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4592            "module answered health.check after its daemon deadline",
4593        )),
4594        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4595            "health.check waiter was canceled before the module responded",
4596        )),
4597        Err(_) => {
4598            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4599            Err(HealthProbeError::no_answer(format!(
4600                "module did not answer health.check within {probe_budget:?}"
4601            )))
4602        }
4603    }
4604}
4605
4606#[allow(clippy::too_many_arguments)]
4607async fn handle_health_report(
4608    spec: &ModuleSpec,
4609    runtime: &SupervisorRuntimeConfig,
4610    registry: &Registry,
4611    process_liveness: &SupervisorProcessLiveness,
4612    snapshot: &SharedSnapshot,
4613    child: &mut Option<SupervisedChild>,
4614    report: HealthReport,
4615    now_ms: u64,
4616) {
4617    let status = supervisor_health_status(report.status);
4618    let detail = report.detail.clone();
4619    let metrics = truncate_health_metrics(report.metrics);
4620    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4621        state.health.status = status;
4622        state.health.last_probe_ms = Some(now_ms);
4623        state.health.detail = detail.clone();
4624        state.health.metrics = metrics.clone();
4625        state.health.consecutive_failures = 0;
4626    });
4627
4628    let action = match report.status {
4629        HealthStatus::Ok => return,
4630        HealthStatus::Degraded => runtime.health.on_degraded,
4631        HealthStatus::Failing => runtime.health.on_failing,
4632    };
4633    apply_l3_health_action(
4634        spec,
4635        runtime,
4636        registry,
4637        process_liveness,
4638        snapshot,
4639        child,
4640        status,
4641        detail.as_deref(),
4642        action,
4643        now_ms,
4644    )
4645    .await;
4646}
4647
4648#[allow(clippy::too_many_arguments)]
4649async fn handle_health_probe_failure(
4650    spec: &ModuleSpec,
4651    runtime: &SupervisorRuntimeConfig,
4652    registry: &Registry,
4653    process_liveness: &SupervisorProcessLiveness,
4654    snapshot: &SharedSnapshot,
4655    child: &mut Option<SupervisedChild>,
4656    err: HealthProbeError,
4657    now_ms: u64,
4658) {
4659    let threshold = runtime.health.failure_threshold.max(1);
4660    let mut failures = 0;
4661    // Carry the evidence class into the operator-visible detail. Without it,
4662    // "module did not answer within 5s" and "the control lane is gone" are two
4663    // prose strings in the same field, and the reader has to know the codebase to
4664    // tell which one is proof of anything.
4665    let detail = format!("[{}] {err}", err.label());
4666    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4667        state.health.last_probe_ms = Some(now_ms);
4668        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4669        state.health.detail = Some(detail.clone());
4670        state.health.metrics = None;
4671        failures = state.health.consecutive_failures;
4672    });
4673
4674    if failures < threshold {
4675        warn!(
4676            module_id = %spec.module_id,
4677            consecutive_failures = failures,
4678            threshold,
4679            evidence = err.label(),
4680            detail = %detail,
4681            "health.check probe failed"
4682        );
4683        return;
4684    }
4685
4686    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4687        state.state = ModuleState::Unresponsive;
4688        state.health.status = SupervisorHealthStatus::Unresponsive;
4689    });
4690    // The evidence class is logged at the kill site because this is the line an
4691    // operator reads after an unexplained restart. A streak of `no-answer` under
4692    // machine load is the known false-positive shape; a `lane-dead` is not.
4693    if runtime.health.critical {
4694        error!(
4695            module_id = %spec.module_id,
4696            status = "unresponsive",
4697            evidence = err.label(),
4698            detail = %detail,
4699            "critical module health alert"
4700        );
4701    } else {
4702        warn!(
4703            module_id = %spec.module_id,
4704            status = "unresponsive",
4705            evidence = err.label(),
4706            detail = %detail,
4707            "module health threshold breached"
4708        );
4709    }
4710    if let Err(err) = health_restart_child(
4711        spec,
4712        runtime,
4713        registry,
4714        process_liveness,
4715        snapshot,
4716        child,
4717        SupervisorHealthStatus::Unresponsive,
4718        Some(&detail),
4719        now_ms,
4720    )
4721    .await
4722    {
4723        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4724    }
4725}
4726
4727#[allow(clippy::too_many_arguments)]
4728async fn apply_l3_health_action(
4729    spec: &ModuleSpec,
4730    runtime: &SupervisorRuntimeConfig,
4731    registry: &Registry,
4732    process_liveness: &SupervisorProcessLiveness,
4733    snapshot: &SharedSnapshot,
4734    child: &mut Option<SupervisedChild>,
4735    status: SupervisorHealthStatus,
4736    detail: Option<&str>,
4737    action: HealthAction,
4738    now_ms: u64,
4739) {
4740    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4741    match action {
4742        HealthAction::Report => {
4743            info!(
4744                module_id = %spec.module_id,
4745                status = ?status,
4746                detail,
4747                "module reported non-ok health"
4748            );
4749        }
4750        HealthAction::Alert => {
4751            error!(
4752                module_id = %spec.module_id,
4753                status = ?status,
4754                detail,
4755                "module health alert"
4756            );
4757        }
4758        HealthAction::Restart => {
4759            if let Err(err) = health_restart_child(
4760                spec,
4761                runtime,
4762                registry,
4763                process_liveness,
4764                snapshot,
4765                child,
4766                status,
4767                detail,
4768                now_ms,
4769            )
4770            .await
4771            {
4772                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4773            }
4774        }
4775    }
4776}
4777
4778#[allow(clippy::too_many_arguments)]
4779async fn health_restart_child(
4780    spec: &ModuleSpec,
4781    runtime: &SupervisorRuntimeConfig,
4782    registry: &Registry,
4783    process_liveness: &SupervisorProcessLiveness,
4784    snapshot: &SharedSnapshot,
4785    child: &mut Option<SupervisedChild>,
4786    status: SupervisorHealthStatus,
4787    detail: Option<&str>,
4788    now_ms: u64,
4789) -> Result<(), SuperviseError> {
4790    let (enabled, schedule) = {
4791        let mut state = lock_snapshot(snapshot)?;
4792        let enabled = state.enabled;
4793        let schedule = if enabled {
4794            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4795        } else {
4796            None
4797        };
4798        (enabled, schedule)
4799    };
4800
4801    if !enabled {
4802        return Err(SuperviseError::Disabled {
4803            module_id: spec.module_id.clone(),
4804        });
4805    }
4806
4807    if schedule.is_none() {
4808        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4809        error!(
4810            module_id = %spec.module_id,
4811            status = ?status,
4812            detail,
4813            max_restarts = runtime.restart_policy.max_restarts,
4814            window_secs = runtime.restart_policy.window.as_secs(),
4815            reason = %runtime.restart_policy.budget_exhausted_detail(),
4816            "health restart budget exhausted; marking module failed"
4817        );
4818        let stop_notice = begin_forwarding_drain_if_configured(
4819            spec,
4820            runtime,
4821            registry,
4822            snapshot,
4823            Some(true),
4824            RouteCloseReason::Disable,
4825        )
4826        .await?;
4827        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4828            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4829        })?;
4830        drain_optional_child(
4831            &spec.module_id,
4832            spec.protocol,
4833            stop_notice,
4834            registry,
4835            runtime.forwarding.as_deref(),
4836            snapshot,
4837            &runtime.terminal_ring,
4838            &runtime.spawn_events,
4839            child,
4840            runtime.drain_timeout,
4841            ModuleState::Failed,
4842            Some(true),
4843        )
4844        .await?;
4845        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4846        return Ok(());
4847    }
4848
4849    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4850    let mut restart_count = 0;
4851    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4852        restart_count = state.crash_restarts.len();
4853        state.state = ModuleState::Unresponsive;
4854        state.health.status = status;
4855        state.health.last_action = Some(HealthAction::Restart.to_string());
4856        state.health.last_action_ms = Some(now_ms);
4857    })?;
4858    warn!(
4859        module_id = %spec.module_id,
4860        status = ?status,
4861        detail,
4862        restart_count,
4863        restart_in_window = schedule.restart_in_window,
4864        delay_ms = schedule.delay.as_millis() as u64,
4865        "health-triggered module restart"
4866    );
4867
4868    let stop_notice = begin_forwarding_drain_if_configured(
4869        spec,
4870        runtime,
4871        registry,
4872        snapshot,
4873        Some(true),
4874        RouteCloseReason::Restart,
4875    )
4876    .await?;
4877    drain_optional_child(
4878        &spec.module_id,
4879        spec.protocol,
4880        stop_notice,
4881        registry,
4882        runtime.forwarding.as_deref(),
4883        snapshot,
4884        &runtime.terminal_ring,
4885        &runtime.spawn_events,
4886        child,
4887        runtime.drain_timeout,
4888        ModuleState::Restarting,
4889        Some(true),
4890    )
4891    .await?;
4892    schedule_respawn(
4893        runtime,
4894        snapshot,
4895        &spec.module_id,
4896        schedule.delay,
4897        RespawnKind::Spawn,
4898    )
4899}
4900
4901fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4902    if let Some(reply) = runtime
4903        .deferred_reload_reply
4904        .lock()
4905        .unwrap_or_else(|p| p.into_inner())
4906        .take()
4907    {
4908        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4909            module_id: module_id.to_string(),
4910            reason: reason.to_string(),
4911        }));
4912    }
4913}
4914
4915fn schedule_respawn(
4916    runtime: &SupervisorRuntimeConfig,
4917    snapshot: &SharedSnapshot,
4918    module_id: &str,
4919    delay: Duration,
4920    kind: RespawnKind,
4921) -> Result<(), SuperviseError> {
4922    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4923    update_snapshot(snapshot, Some(module_id), |state| {
4924        state.respawn_pending = true
4925    })?;
4926    *runtime
4927        .scheduled_respawn
4928        .lock()
4929        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
4930        deadline: Instant::now() + delay,
4931        kind,
4932    });
4933    Ok(())
4934}
4935
4936fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4937    let _ = update_snapshot(snapshot, Some(module_id), |state| {
4938        state.health.last_action = Some(action);
4939        state.health.last_action_ms = Some(now_ms);
4940    });
4941}
4942
4943fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4944    match status {
4945        HealthStatus::Ok => SupervisorHealthStatus::Ok,
4946        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4947        HealthStatus::Failing => SupervisorHealthStatus::Failing,
4948    }
4949}
4950
4951/// Caps the metrics blob stored in the cached supervisor snapshot, which is
4952/// returned to every `supervisor.list` and `supervisor.health` caller.
4953///
4954/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
4955/// path: that request exists to return a module's complete metrics object, and
4956/// `ck health <module-id>` documents it as the way to see what the cached view
4957/// truncates. The asymmetry is the feature.
4958///
4959/// So a new caller must decide which side it is on rather than assume the cap is
4960/// universal. Reaching for it on a fresh-probe path would silently reintroduce
4961/// the truncation that path exists to avoid.
4962fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4963    let metrics = metrics?;
4964    match serde_json::to_vec(&metrics) {
4965        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4966            "truncated": true,
4967            "original_bytes": encoded.len(),
4968        })),
4969        Ok(_) | Err(_) => Some(metrics),
4970    }
4971}
4972
4973/// Spread health probes so a fleet-wide restart does not converge them.
4974///
4975/// The delay is derived from the module id and probe index rather than a random
4976/// source, so it is deterministic per module: a module keeps its own offset
4977/// across daemon restarts instead of re-rolling into a collision.
4978fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4979    if cadence.is_zero() {
4980        return Duration::ZERO;
4981    }
4982    let cadence_ms = cadence.as_millis() as u64;
4983    // This early return is REDUNDANT, deliberately, and a mutation run will show
4984    // it surviving removal. Recording why here so the next person to notice does
4985    // not have to re-derive it:
4986    //
4987    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
4988    //   a zero cadence and builds the Duration from whole milliseconds, so a
4989    //   sub-millisecond cadence cannot come from config.
4990    // - Even if reached it changes no answer. The `.max(1)` below makes the span
4991    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
4992    //   -- exactly what this returns.
4993    //
4994    // Kept as a guard against a future widening of the config parser (accepting
4995    // microseconds, say), which would make the sub-millisecond case reachable.
4996    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
4997    // divides by zero. Remove this and nothing changes.
4998    if cadence_ms == 0 {
4999        return cadence;
5000    }
5001    // Note that this never returns less than one cadence, including for the FIRST
5002    // probe. So a freshly registered module reports health `unknown` for a full
5003    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5004    // ready to answer.
5005    //
5006    // That is a property of the supervisor's schedule, not of any module: an
5007    // operator watching a restart sees `unknown` and cannot tell it from a module
5008    // that is slow to warm. Measured on two unrelated modules, both flipping to
5009    // `ok` between 22s and 32s after restart.
5010    //
5011    // Left as-is because spreading the first probe is what keeps a fleet-wide
5012    // restart from firing fourteen simultaneous probes into a cold machine. The
5013    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5014    // that thundering herd for a faster first reading.
5015    let jitter_span = (cadence_ms / 10).max(1);
5016    let hash = module_id.as_bytes().iter().fold(
5017        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5018        |acc, byte| {
5019            acc.wrapping_mul(1099511628211)
5020                .wrapping_add(u64::from(*byte))
5021        },
5022    );
5023    cadence + Duration::from_millis(hash % jitter_span)
5024}
5025
5026#[cfg(test)]
5027mod tests {
5028    use super::*;
5029
5030    #[test]
5031    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5032        let handle = SupervisorHandle::new();
5033        let module_id = "readded-tombstone";
5034        handle.record_rescan_removal(module_id);
5035        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5036
5037        handle.apply_identity_configuration(&ModuleSpec {
5038            module_id: module_id.to_string(),
5039            program: PathBuf::from("/test/module"),
5040            args: Vec::new(),
5041            env: Vec::new(),
5042            reserved: false,
5043            reserved_prefixes: Vec::new(),
5044            protocol: ModuleProtocol::Subc,
5045            overlap: Default::default(),
5046        });
5047
5048        assert!(
5049            handle.removal_tombstone_age_ms(module_id).is_none(),
5050            "a re-added module must not retain a stale removal tombstone"
5051        );
5052    }
5053
5054    /// What one module's owner looked like from the control plane at the
5055    /// instant after its first process was spawned.
5056    #[derive(Debug, PartialEq, Eq)]
5057    struct OwnerInSpawnWindow {
5058        module_id: String,
5059        configured: bool,
5060        on_roster: bool,
5061        admission_refusal: Option<&'static str>,
5062    }
5063
5064    /// A supervised module's process can connect, register, sync its scopes
5065    /// and describe them as soon as it is spawned, which is BEFORE the
5066    /// supervisor puts the module on the roster. In that window the owner must
5067    /// already read as configured, so a scoped `route.open` against it is
5068    /// refused as retryable `scope_not_synced` and not as terminal
5069    /// `scope_not_live` ("will never sync").
5070    ///
5071    /// The hook runs in exactly that window on every path that takes on a new
5072    /// module, so no race with a real child is needed: `on_roster: false`
5073    /// proves each observation was taken before the roster insert.
5074    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5075    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5076        use crate::scopes::ScopeTable;
5077        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5078
5079        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5080        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5081            module_id: module_id.to_string(),
5082            program,
5083            args: Vec::new(),
5084            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5085                .into_iter()
5086                .map(|key| (key.to_string(), dir.path().display().to_string()))
5087                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5088                .collect(),
5089            reserved: false,
5090            reserved_prefixes: Vec::new(),
5091            protocol: ModuleProtocol::Subc,
5092            overlap: Default::default(),
5093        };
5094        let live = super::terminal_history_tests::fake_aft_stub_path();
5095        let missing = dir.path().join("definitely-missing-module");
5096
5097        let handle = SupervisorHandle::new();
5098        let mut supervisor =
5099            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5100                .with_handle(handle.clone());
5101        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5102        let hook_handle = handle.clone();
5103        let hook_observed = Arc::clone(&observed);
5104        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5105            // Exactly what the control plane computes for a scoped route.open
5106            // naming this module as the owner of a scope it has not synced.
5107            let configured = hook_handle.is_configured(module_id);
5108            let selector = ScopeSelector {
5109                owner: Principal::Reserved {
5110                    module_id: module_id.to_string(),
5111                },
5112                scope_ref: "s".to_string(),
5113                scope_epoch: Some(1),
5114            };
5115            let carrier = Principal::Reserved {
5116                module_id: "carrier".to_string(),
5117            };
5118            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5119                .admit(&carrier, module_id, &selector, configured)
5120            {
5121                Ok(_) => None,
5122                Err(refusal) => Some(refusal.code),
5123            };
5124            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5125                module_id: module_id.to_string(),
5126                configured,
5127                on_roster: hook_handle.get(module_id).is_some(),
5128                admission_refusal,
5129            });
5130        })));
5131
5132        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5133        let configured = supervisor
5134            .supervise_configured(stub("configured", live.clone()), true)
5135            .unwrap();
5136        let with_health = supervisor
5137            .supervise_configured_with_health(
5138                stub("with-health", live.clone()),
5139                true,
5140                HealthConfig::default(),
5141                None,
5142                RestartPolicy::default(),
5143            )
5144            .unwrap();
5145        // The failed-spawn path still puts the module on the roster (as
5146        // failed), so it is configured throughout.
5147        let failed = supervisor
5148            .supervise_configured_with_health(
5149                stub("failed-spawn", missing.clone()),
5150                true,
5151                HealthConfig::default(),
5152                None,
5153                RestartPolicy::default(),
5154            )
5155            .unwrap();
5156        // A failed plain `spawn` puts nothing on the roster, so its mark is
5157        // taken back once the spawn has failed.
5158        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5159
5160        let expected = [
5161            "plain",
5162            "configured",
5163            "with-health",
5164            "failed-spawn",
5165            "spawn-error",
5166        ]
5167        .into_iter()
5168        .map(|module_id| OwnerInSpawnWindow {
5169            module_id: module_id.to_string(),
5170            configured: true,
5171            on_roster: false,
5172            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5173        })
5174        .collect::<Vec<_>>();
5175        assert_eq!(*observed.lock().unwrap(), expected);
5176
5177        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5178            assert!(
5179                handle.get(module_id).is_some(),
5180                "{module_id} is on the roster"
5181            );
5182            assert!(
5183                handle.is_configured(module_id),
5184                "{module_id} stays configured"
5185            );
5186        }
5187        assert!(handle.get("spawn-error").is_none());
5188        assert!(
5189            !handle.is_configured("spawn-error"),
5190            "a plain spawn that failed must not leave its module marked configured"
5191        );
5192
5193        // Leaving the roster clears the mark with it.
5194        handle.retire("failed-spawn");
5195        assert!(!handle.is_configured("failed-spawn"));
5196
5197        for module in [plain, configured, with_health] {
5198            module.stop().await.unwrap();
5199        }
5200        drop(failed);
5201    }
5202
5203    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5204        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5205        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5206            snapshot.process_alive = true;
5207            snapshot.pid = Some(41);
5208            snapshot.spawned_at_ms = Some(42);
5209            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5210            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5211                device: 43,
5212                inode: 44,
5213            });
5214        })
5215        .unwrap();
5216        snapshot
5217    }
5218
5219    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5220        let snapshot = lock_snapshot(snapshot).unwrap();
5221        assert!(!snapshot.process_alive);
5222        assert_eq!(snapshot.pid, None);
5223        assert_eq!(snapshot.spawned_at_ms, None);
5224        assert_eq!(snapshot.spawned_from, None);
5225        assert_eq!(snapshot.spawned_file_identity, None);
5226    }
5227
5228    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5229    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5230        let supervisor =
5231            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5232        let mut runtime = supervisor.runtime_config();
5233        runtime.test_seed_stale_facts_before_enable_spawn = true;
5234        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5235        let mut child = None;
5236        let spec = ModuleSpec {
5237            module_id: "failed-enable-clears-facts".to_string(),
5238            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5239            args: Vec::new(),
5240            env: Vec::new(),
5241            reserved: false,
5242            reserved_prefixes: Vec::new(),
5243            protocol: ModuleProtocol::Subc,
5244            overlap: Default::default(),
5245        };
5246
5247        let result = set_child_enabled(
5248            &spec,
5249            &runtime,
5250            &supervisor.registry,
5251            &supervisor.process_liveness,
5252            &snapshot,
5253            &mut child,
5254            true,
5255        )
5256        .await;
5257
5258        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5259        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5260        assert_snapshot_process_facts_cleared(&snapshot);
5261    }
5262
5263    #[tokio::test]
5264    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5265        let supervisor =
5266            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5267        let runtime = supervisor.runtime_config();
5268        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5269            ModuleState::Restarting,
5270            true,
5271        )));
5272        let spec = ModuleSpec {
5273            module_id: "start-stranded-restarting".to_string(),
5274            program: super::terminal_history_tests::fake_aft_stub_path(),
5275            args: Vec::new(),
5276            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5277            reserved: false,
5278            reserved_prefixes: Vec::new(),
5279            protocol: ModuleProtocol::None,
5280            overlap: Default::default(),
5281        };
5282        let mut child = None;
5283        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5284        assert!(!super::set_child_enabled(
5285            &spec,
5286            &runtime,
5287            &Registry::default(),
5288            &supervisor.process_liveness,
5289            &snapshot,
5290            &mut child,
5291            true
5292        )
5293        .await
5294        .unwrap());
5295        assert!(child.is_none());
5296        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5297        assert!(super::set_child_enabled(
5298            &spec,
5299            &runtime,
5300            &Registry::default(),
5301            &supervisor.process_liveness,
5302            &snapshot,
5303            &mut child,
5304            true
5305        )
5306        .await
5307        .unwrap());
5308        assert_eq!(
5309            lock_snapshot(&snapshot).unwrap().state,
5310            ModuleState::Running
5311        );
5312        let mut child = child.unwrap();
5313        child.start_kill().unwrap();
5314        child.wait().await.unwrap();
5315    }
5316
5317    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5318    async fn failed_reload_spawn_clears_current_process_facts() {
5319        let supervisor =
5320            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5321        let mut runtime = supervisor.runtime_config();
5322        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5323        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5324        let mut child = None;
5325        let spec = ModuleSpec {
5326            module_id: "failed-reload-clears-facts".to_string(),
5327            program: PathBuf::from("/unused/failed-reload-module"),
5328            args: Vec::new(),
5329            env: Vec::new(),
5330            reserved: false,
5331            reserved_prefixes: Vec::new(),
5332            protocol: ModuleProtocol::Subc,
5333            overlap: Default::default(),
5334        };
5335
5336        let result = handle_reload_spawn_failure(
5337            &spec,
5338            &runtime,
5339            &supervisor.process_liveness,
5340            &snapshot,
5341            &mut child,
5342            "forced reload spawn failure".to_string(),
5343        )
5344        .await;
5345
5346        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5347        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5348        assert_snapshot_process_facts_cleared(&snapshot);
5349    }
5350
5351    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5352    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5353        let supervisor =
5354            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5355        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5356        let module = supervisor.supervised_module(
5357            ModuleSpec {
5358                module_id: "drop-clears-facts".to_string(),
5359                program: PathBuf::from("/unused/drop-module"),
5360                args: Vec::new(),
5361                env: Vec::new(),
5362                reserved: false,
5363                reserved_prefixes: Vec::new(),
5364                protocol: ModuleProtocol::Subc,
5365                overlap: Default::default(),
5366            },
5367            supervisor.runtime_config(),
5368            Arc::clone(&snapshot),
5369            None,
5370        );
5371        assert!(!module
5372            .inner
5373            .monitor
5374            .lock()
5375            .unwrap()
5376            .as_ref()
5377            .unwrap()
5378            .is_finished());
5379
5380        drop(module);
5381
5382        assert_eq!(
5383            lock_snapshot(&snapshot).unwrap().state,
5384            ModuleState::Stopped
5385        );
5386        assert_snapshot_process_facts_cleared(&snapshot);
5387    }
5388
5389    #[cfg(unix)]
5390    #[tokio::test]
5391    async fn rescan_preserves_running_protocol_until_respawn() {
5392        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5393        let initial = ModuleSpec {
5394            module_id: "rescan-protocol".into(),
5395            program: PathBuf::from("/bin/sleep"),
5396            args: vec!["60".into()],
5397            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5398                .into_iter()
5399                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5400                .collect(),
5401            reserved: false,
5402            reserved_prefixes: vec![],
5403            protocol: ModuleProtocol::None,
5404            overlap: Default::default(),
5405        };
5406        let supervisor =
5407            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5408        let module = supervisor.spawn(initial.clone()).unwrap();
5409        assert!(module.status().unwrap().live);
5410        let mut next = initial;
5411        next.protocol = ModuleProtocol::Subc;
5412        module
5413            .update_configuration(next.clone(), HealthConfig::default(), None)
5414            .await
5415            .unwrap();
5416        assert!(
5417            module.status().unwrap().live,
5418            "rescan must not require HELLO from the old non-wire process"
5419        );
5420        let runtime = supervisor.runtime_config();
5421        let action = on_child_exit(
5422            &next,
5423            RestartPolicy::default(),
5424            &supervisor.registry,
5425            &module.inner.snapshot,
5426            &runtime.terminal_ring,
5427            &runtime.spawn_events,
5428            &runtime.child_roster,
5429            ExitReport {
5430                kind: ExitKind::Clean,
5431                code: Some(0),
5432                signal: None,
5433                at_ms: unix_ms_now(),
5434            },
5435        )
5436        .await;
5437        assert!(
5438            matches!(action, NextAction::Restart { .. }),
5439            "the old non-wire process's clean exit must restart"
5440        );
5441        module.drain().await.unwrap();
5442    }
5443
5444    #[cfg(unix)]
5445    #[tokio::test]
5446    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5447        use std::os::unix::fs::PermissionsExt;
5448        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5449        let script = dir.join("module.sh");
5450        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5451        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5452        let record_path = dir.join("live-children.json");
5453        let supervisor =
5454            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5455                .with_live_children_record(&record_path);
5456        for (program, args) in [
5457            (PathBuf::from("sleep"), vec!["60".into()]),
5458            (script, vec![]),
5459        ] {
5460            let spec = ModuleSpec {
5461                module_id: "image-identity".into(),
5462                program,
5463                args,
5464                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5465                    .into_iter()
5466                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5467                    .collect(),
5468                reserved: false,
5469                reserved_prefixes: vec![],
5470                protocol: ModuleProtocol::None,
5471                overlap: Default::default(),
5472            };
5473            let module = supervisor.spawn(spec).unwrap();
5474            #[cfg(target_os = "macos")]
5475            {
5476                // SETEXEC confirmation is asynchronous; the orphan record must
5477                // identify the final image, never the intermediate trampoline.
5478                let deadline = Instant::now() + Duration::from_secs(5);
5479                while crate::live_children::read_record(&record_path)
5480                    .unwrap()
5481                    .iter()
5482                    .all(|entry| entry.executable.is_none())
5483                {
5484                    assert!(Instant::now() < deadline, "module image was not confirmed");
5485                    tokio::time::sleep(Duration::from_millis(5)).await;
5486                }
5487            }
5488            let entry = crate::live_children::read_record(&record_path)
5489                .unwrap()
5490                .pop()
5491                .unwrap();
5492            let observed = subc_os::Process::open(entry.pid)
5493                .unwrap()
5494                .unwrap()
5495                .observe()
5496                .unwrap();
5497            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5498            module.drain().await.unwrap();
5499            assert_eq!(verdict, crate::live_children::IdentityVerdict::Matches);
5500        }
5501    }
5502
5503    #[cfg(unix)]
5504    fn http_fixture(
5505        dir: &std::path::Path,
5506        url: &str,
5507        threshold: u32,
5508    ) -> crate::daemon_config::ConfiguredModule {
5509        let path = dir.join("subc.jsonc");
5510        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5511            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5512            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5513            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5514            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5515        }}}).to_string()).unwrap();
5516        crate::daemon_config::load(&path)
5517            .unwrap()
5518            .unwrap()
5519            .modules
5520            .pop()
5521            .unwrap()
5522    }
5523
5524    #[cfg(unix)]
5525    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5526        timeout(Duration::from_secs(5), async {
5527            loop {
5528                if module.status().unwrap().health.status == status {
5529                    break;
5530                }
5531                sleep(Duration::from_millis(5)).await;
5532            }
5533        })
5534        .await
5535        .unwrap_or_else(|_| {
5536            panic!(
5537                "expected {status:?}, got {:?}",
5538                module.status().unwrap().health
5539            )
5540        });
5541    }
5542
5543    #[cfg(unix)]
5544    #[tokio::test]
5545    async fn http_health_status_flips_ok_failing_ok() {
5546        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5547        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5548        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5549        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5550        let serving_status = status.clone();
5551        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5552        let server = tokio::spawn(async move {
5553            loop {
5554                let (mut stream, _) = listener.accept().await.unwrap();
5555                let mut request = [0u8; 2048];
5556                let count = stream.read(&mut request).await.unwrap();
5557                assert!(count > 0, "a probe must send an HTTP request");
5558                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5559                let body = if code == 200 {
5560                    "ready"
5561                } else {
5562                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5563                };
5564                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5565                let _ = stream.write_all(response.as_bytes()).await;
5566            }
5567        });
5568        let configured = http_fixture(&dir, &url, 1000);
5569        let module =
5570            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5571                .supervise_configured_with_health(
5572                    configured.module_spec(),
5573                    true,
5574                    configured.health,
5575                    configured.drain_timeout_ms,
5576                    configured.restart,
5577                )
5578                .unwrap();
5579        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5580        status.store(503, std::sync::atomic::Ordering::SeqCst);
5581        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5582        assert!(module
5583            .status()
5584            .unwrap()
5585            .health
5586            .detail
5587            .unwrap()
5588            .contains("scratch failure"));
5589        status.store(200, std::sync::atomic::Ordering::SeqCst);
5590        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5591        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5592        let before = module.status().unwrap();
5593        let (spec, mut health) = module.configuration().unwrap();
5594        health.http = None;
5595        module
5596            .update_configuration(spec.clone(), health.clone(), Some(10))
5597            .await
5598            .unwrap();
5599        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5600        health.http = Some(url);
5601        module
5602            .update_configuration(spec, health, Some(10))
5603            .await
5604            .unwrap();
5605        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5606        assert_eq!(
5607            module.status().unwrap().pid,
5608            before.pid,
5609            "changing a probe must apply live, not restart its process"
5610        );
5611        let (spec, mut health) = module.configuration().unwrap();
5612        health.failure_threshold = 2;
5613        module
5614            .update_configuration(spec, health, Some(10))
5615            .await
5616            .unwrap();
5617        status.store(503, std::sync::atomic::Ordering::SeqCst);
5618        timeout(Duration::from_secs(5), async {
5619            while module.status().unwrap().spawn_generation == before.spawn_generation {
5620                sleep(Duration::from_millis(5)).await;
5621            }
5622        })
5623        .await
5624        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5625        module.drain().await.unwrap();
5626        server.abort();
5627    }
5628
5629    #[cfg(unix)]
5630    #[tokio::test]
5631    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5632        let dir = subc_test_support::TestTempDir::new("http-refused");
5633        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5634        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5635        drop(unused);
5636        let configured = http_fixture(&dir, &url, 2);
5637        let module =
5638            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5639                .supervise_configured_with_health(
5640                    configured.module_spec(),
5641                    true,
5642                    configured.health,
5643                    configured.drain_timeout_ms,
5644                    configured.restart,
5645                )
5646                .unwrap();
5647        let before = module.status().unwrap().spawn_generation;
5648        timeout(Duration::from_secs(5), async {
5649            loop {
5650                let status = module.status().unwrap();
5651                if status.spawn_generation > before {
5652                    assert!(status.lifetime_restarts > 0);
5653                    break;
5654                }
5655                sleep(Duration::from_millis(5)).await;
5656            }
5657        })
5658        .await
5659        .expect("sustained HTTP refusal must trigger the health restart policy");
5660        module.drain().await.unwrap();
5661    }
5662
5663    #[cfg(unix)]
5664    #[tokio::test]
5665    async fn http_health_timeout_honours_deadline() {
5666        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5667        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5668        let server = tokio::spawn(async move {
5669            let _held = listener.accept().await.unwrap();
5670            std::future::pending::<()>().await;
5671        });
5672        let error = timeout(
5673            Duration::from_secs(1),
5674            probe_http_health(&url, Duration::from_millis(10)),
5675        )
5676        .await
5677        .expect("the probe must enforce its own deadline")
5678        .unwrap_err();
5679        server.abort();
5680        assert!(error.to_string().contains("timed out"));
5681    }
5682
5683    #[cfg(unix)]
5684    #[tokio::test]
5685    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5686        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5687        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5688        let url = format!(
5689            "http://localhost:{}/healthz",
5690            listener.local_addr().unwrap().port()
5691        );
5692        let server = tokio::spawn(async move {
5693            let (mut stream, _) = listener.accept().await.unwrap();
5694            let mut request = [0u8; 2048];
5695            assert!(stream.read(&mut request).await.unwrap() > 0);
5696            stream
5697                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5698                .await
5699                .unwrap();
5700        });
5701        // The deadline only bounds a hang. A probe that never tried the IPv6
5702        // address would be refused on 127.0.0.1 and fail at once, so a longer
5703        // deadline does not weaken the assertion; one second timed out under a
5704        // loaded parallel test run.
5705        assert_eq!(
5706            probe_http_health(&url, Duration::from_secs(10))
5707                .await
5708                .unwrap()
5709                .status,
5710            HealthStatus::Ok
5711        );
5712        server.await.unwrap();
5713    }
5714
5715    #[cfg(unix)]
5716    #[tokio::test]
5717    async fn http_health_timeout_keeps_partial_status_and_body() {
5718        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5719        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5720        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5721        let server = tokio::spawn(async move {
5722            let (mut stream, _) = listener.accept().await.unwrap();
5723            let mut request = [0u8; 2048];
5724            assert!(stream.read(&mut request).await.unwrap() > 0);
5725            stream
5726                .write_all(
5727                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5728                )
5729                .await
5730                .unwrap();
5731            std::future::pending::<()>().await;
5732        });
5733        let error = probe_http_health(&url, Duration::from_secs(1))
5734            .await
5735            .unwrap_err()
5736            .to_string();
5737        server.abort();
5738        assert!(
5739            error.contains("timed out")
5740                && error.contains("503 Unavailable")
5741                && error.contains("partial diagnostic"),
5742            "{error}"
5743        );
5744    }
5745
5746    #[cfg(unix)]
5747    #[tokio::test]
5748    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5749        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5750        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5751        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5752        let server = tokio::spawn(async move {
5753            let (mut stream, _) = listener.accept().await.unwrap();
5754            let mut request = [0u8; 2048];
5755            assert!(stream.read(&mut request).await.unwrap() > 0);
5756            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5757            let response = format!(
5758                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5759                body.len()
5760            );
5761            stream.write_all(response.as_bytes()).await.unwrap();
5762        });
5763        let error = probe_http_health(&url, Duration::from_secs(1))
5764            .await
5765            .unwrap_err()
5766            .to_string();
5767        server.await.unwrap();
5768        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5769        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5770        assert!(!error.contains("not-in-diagnostic"));
5771    }
5772
5773    #[cfg(unix)]
5774    #[tokio::test]
5775    async fn http_health_real_nats_server_monitoring() {
5776        if std::process::Command::new("nats-server")
5777            .arg("--version")
5778            .env("XDG_DATA_HOME", std::env::temp_dir())
5779            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5780            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5781            .output()
5782            .is_err()
5783        {
5784            eprintln!(
5785                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5786            );
5787            return;
5788        }
5789        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5790        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5791        let port = monitor.local_addr().unwrap().port();
5792        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5793        let client_port = client.local_addr().unwrap().port();
5794        let config = dir.join("server.conf");
5795        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5796        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5797        configured.program = PathBuf::from("nats-server");
5798        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5799        drop(monitor);
5800        drop(client);
5801        let module =
5802            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5803                .supervise_configured_with_health(
5804                    configured.module_spec(),
5805                    true,
5806                    configured.health,
5807                    configured.drain_timeout_ms,
5808                    configured.restart,
5809                )
5810                .unwrap();
5811        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5812        module.drain().await.unwrap();
5813    }
5814
5815    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5816    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5817        let supervisor =
5818            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5819        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5820        let initial = ModuleSpec {
5821            module_id: "rescan-preserves-spawn-facts".to_string(),
5822            program: PathBuf::from("/spawned/module"),
5823            args: Vec::new(),
5824            env: Vec::new(),
5825            reserved: false,
5826            reserved_prefixes: Vec::new(),
5827            protocol: ModuleProtocol::Subc,
5828            overlap: Default::default(),
5829        };
5830        let module = supervisor.supervised_module(
5831            initial.clone(),
5832            supervisor.runtime_config(),
5833            snapshot,
5834            None,
5835        );
5836        let before = module.status().unwrap();
5837        let mut replacement = initial;
5838        replacement.program = PathBuf::from("/rescanned/replacement-module");
5839
5840        module
5841            .update_configuration(replacement, HealthConfig::default(), None)
5842            .await
5843            .unwrap();
5844
5845        let after = module.status().unwrap();
5846        assert_eq!(after.pid, before.pid);
5847        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5848        assert_eq!(after.spawned_from, before.spawned_from);
5849        drop(module);
5850    }
5851}
5852
5853fn unix_ms_now() -> u64 {
5854    SystemTime::now()
5855        .duration_since(UNIX_EPOCH)
5856        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5857        .unwrap_or(0)
5858}
5859
5860async fn supervise_loop(
5861    mut spec: ModuleSpec,
5862    mut runtime: SupervisorRuntimeConfig,
5863    registry: Arc<Registry>,
5864    process_liveness: Arc<SupervisorProcessLiveness>,
5865    snapshot: SharedSnapshot,
5866    mut child: Option<SupervisedChild>,
5867    mut commands: mpsc::Receiver<SupervisorCommand>,
5868) {
5869    let mut health_probe = HealthProbeRuntime::default();
5870    // All restart backoffs run here, including health and operator requests.
5871    // While one is pending the loop serves commands, so disable or drain can
5872    // cancel the replacement without spawning a process just to stop it.
5873    let mut pending_respawn: Option<PendingRespawn> = None;
5874    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5875    // before anything else so a stop that interrupted a swap runs at once.
5876    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5877    loop {
5878        #[cfg(target_os = "macos")]
5879        if let Some(active) = child.as_mut() {
5880            active.confirm_privacy_exec().await;
5881        }
5882        if let Some(scheduled) = runtime
5883            .scheduled_respawn
5884            .lock()
5885            .unwrap_or_else(|p| p.into_inner())
5886            .take()
5887        {
5888            pending_respawn = Some(scheduled);
5889        }
5890        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5891            pending_respawn = None;
5892            cancel_deferred_reload(
5893                &runtime,
5894                &spec.module_id,
5895                "respawn cancelled by a supervisor command",
5896            );
5897        }
5898        if child.is_none() && pending_respawn.is_none() {
5899            cancel_deferred_reload(
5900                &runtime,
5901                &spec.module_id,
5902                "respawn cancelled before a replacement was spawned",
5903            );
5904            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5905                state.respawn_pending = false;
5906                state.coalesced_restart_pending = false;
5907                if matches!(
5908                    state.state,
5909                    ModuleState::Restarting
5910                        | ModuleState::Starting
5911                        | ModuleState::Draining
5912                        | ModuleState::Unresponsive
5913                ) {
5914                    error!(module_id = %spec.module_id, state = ?state.state,
5915                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5916                    state.state = ModuleState::Failed;
5917                    clear_current_process_facts(state);
5918                }
5919            });
5920        }
5921        if let Some(command) = requeued.pop_front() {
5922            if !handle_supervisor_command(
5923                command,
5924                &mut spec,
5925                &mut runtime,
5926                &registry,
5927                &process_liveness,
5928                &snapshot,
5929                &mut child,
5930                &mut commands,
5931                &mut requeued,
5932            )
5933            .await
5934            {
5935                return;
5936            }
5937            if child.is_some() || !respawn_still_pending(&snapshot) {
5938                pending_respawn = None;
5939                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5940                    state.respawn_pending = false
5941                });
5942            }
5943            continue;
5944        }
5945        if child.is_some() {
5946            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
5947            let probe_sleep = sleep(health_probe.wake_after());
5948            tokio::pin!(probe_sleep);
5949            let active_child = child.as_mut().expect("child checked above");
5950            tokio::select! {
5951                wait_result = active_child.wait() => {
5952                    // Every arm below that gives up on the CHILD must keep the
5953                    // supervision task itself alive (child = None, loop
5954                    // continues into command-serving mode). Returning here
5955                    // closes the command channel, which makes the module
5956                    // permanently unrestartable in-band: a clean child exit
5957                    // of an enabled module once wedged the fleet this way
5958                    // ('supervisor command channel is closed') and required a
5959                    // full daemon restart to recover.
5960                    let exit_report = match wait_result {
5961                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
5962                        Err(err) => {
5963                            active_child.drain_stderr(&spec.module_id).await;
5964                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
5965                            // Every other exit path (on_child_exit's Clean/Crash arms,
5966                            // the reload-registration-failure path) records a terminal
5967                            // before moving on. Without one here, a module whose wait()
5968                            // itself errored (e.g. already reaped) leaves no terminal
5969                            // record at all -- an empty ring reads as "nothing died".
5970                            record_wait_error_terminal(
5971                                &spec.module_id,
5972                                &runtime.terminal_ring,
5973                                &runtime.spawn_events,
5974                            );
5975                            untrack_if_registration_released(
5976                                &process_liveness,
5977                                &registry,
5978                                &spec.module_id,
5979                                &snapshot,
5980                            );
5981                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
5982                            child = None;
5983                            continue;
5984                        }
5985                    };
5986                    active_child.drain_stderr(&spec.module_id).await;
5987
5988                    let next = on_child_exit(
5989                        &spec,
5990                        runtime.restart_policy,
5991                        &registry,
5992                        &snapshot,
5993                        &runtime.terminal_ring,
5994                        &runtime.spawn_events,
5995                        &runtime.child_roster,
5996                        exit_report,
5997                    ).await;
5998                    // The exit is recorded, so a daemon shutdown may stop
5999                    // waiting for this child (see `SupervisedChild::wait`).
6000                    active_child.release_roster();
6001                    match next {
6002                        NextAction::Stop { registration_released } => {
6003                            if registration_released {
6004                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6005                            }
6006                            child = None;
6007                        }
6008                        NextAction::Restart { schedule } => {
6009                            let delay = schedule.map_or(
6010                                runtime.restart_policy.delay_for_restart(0),
6011                                |schedule| schedule.delay,
6012                            );
6013                            if let Some(schedule) = schedule {
6014                                log_crash_respawn(&spec.module_id, schedule);
6015                            }
6016                            // The exited child is fully recorded at this point,
6017                            // so release it and count the backoff down in the
6018                            // command-serving branch below rather than sleeping
6019                            // here: commands cannot be received from inside this
6020                            // select arm, and an operator disable or drain that
6021                            // arrives during the backoff must cancel the pending
6022                            // respawn instead of waiting for it to spawn first.
6023                            child = None;
6024                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6025                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6026                        }
6027                    }
6028                }
6029                command = commands.recv() => {
6030                    let Some(command) = command else {
6031                        return;
6032                    };
6033                    if !handle_supervisor_command(
6034                        command,
6035                        &mut spec,
6036                        &mut runtime,
6037                        &registry,
6038                        &process_liveness,
6039                        &snapshot,
6040                        &mut child,
6041                        &mut commands,
6042                        &mut requeued,
6043                    ).await {
6044                        return;
6045                    }
6046                }
6047                _ = &mut probe_sleep => {
6048                    if health_probe.due() {
6049                        run_health_probe_cycle(
6050                            &spec,
6051                            &runtime,
6052                            &registry,
6053                            &process_liveness,
6054                            &snapshot,
6055                            &mut child,
6056                        ).await;
6057                        if child.is_some() {
6058                            health_probe.schedule_next(&spec, runtime.health.cadence);
6059                        }
6060                    }
6061                }
6062            }
6063        } else if let Some(pending) = pending_respawn {
6064            tokio::select! {
6065                _ = sleep_until(pending.deadline) => {
6066                    pending_respawn = None;
6067                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6068                    // A command handled below while the backoff elapsed may
6069                    // have stopped the module; never respawn past an operator's
6070                    // disable or drain.
6071                    if !respawn_still_pending(&snapshot) {
6072                        continue;
6073                    }
6074                    // The daemon began shutting down during the backoff: the
6075                    // spawn would be refused anyway, and refusing it here
6076                    // leaves the module stopped instead of reporting a
6077                    // failed restart.
6078                    if runtime.child_roster.is_closed() {
6079                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6080                            state.state = ModuleState::Stopped;
6081                        });
6082                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6083                        continue;
6084                    }
6085                    if let Err(err) = release_dead_registration(
6086                        &registry,
6087                        runtime.forwarding.as_deref(),
6088                        &snapshot,
6089                        &spec.module_id,
6090                    ).await {
6091                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6092                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6093                        continue;
6094                    }
6095
6096                    if matches!(pending.kind, RespawnKind::Reload) {
6097                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6098                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6099                        if let Some(reply) = reply { let _ = reply.send(result); }
6100                        continue;
6101                    }
6102                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6103                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6104                        Ok(next_child) => {
6105                            child = Some(next_child);
6106                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6107                        }
6108                        Err(err) => {
6109                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6110                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6111                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6112                        }
6113                    }
6114                }
6115                command = commands.recv() => {
6116                    let Some(command) = command else {
6117                        return;
6118                    };
6119                    if !handle_supervisor_command(
6120                        command,
6121                        &mut spec,
6122                        &mut runtime,
6123                        &registry,
6124                        &process_liveness,
6125                        &snapshot,
6126                        &mut child,
6127                        &mut commands,
6128                        &mut requeued,
6129                    ).await {
6130                        return;
6131                    }
6132                    // Reconcile the pending respawn with what the command did:
6133                    // a start may already have spawned a fresh child,
6134                    // while a disable or drain moved the snapshot out of the
6135                    // state the respawn was counting down from.
6136                    if child.is_some() || !respawn_still_pending(&snapshot) {
6137                        pending_respawn = None;
6138                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6139                    }
6140                }
6141            }
6142        } else {
6143            let Some(command) = commands.recv().await else {
6144                return;
6145            };
6146            if !handle_supervisor_command(
6147                command,
6148                &mut spec,
6149                &mut runtime,
6150                &registry,
6151                &process_liveness,
6152                &snapshot,
6153                &mut child,
6154                &mut commands,
6155                &mut requeued,
6156            )
6157            .await
6158            {
6159                return;
6160            }
6161        }
6162    }
6163}
6164
6165fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6166    info!(
6167        module_id,
6168        restart_in_window = schedule.restart_in_window,
6169        delay_ms = schedule.delay.as_millis() as u64,
6170        "respawning after crash"
6171    );
6172}
6173
6174/// Whether the respawn a backoff was counting down to is still wanted. A
6175/// disable or drain handled while the backoff elapsed moves the snapshot out
6176/// of `Restarting`, and the operator's stop must win over the pending respawn,
6177/// so every sleep-then-spawn path re-validates against the live snapshot
6178/// instead of assuming the state it left behind still holds.
6179fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6180    matches!(
6181        lock_snapshot(snapshot),
6182        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6183    )
6184}
6185
6186enum NextAction {
6187    Stop {
6188        registration_released: bool,
6189    },
6190    Restart {
6191        schedule: Option<CrashRestartSchedule>,
6192    },
6193}
6194
6195#[allow(clippy::too_many_arguments)]
6196async fn handle_supervisor_command(
6197    command: SupervisorCommand,
6198    spec: &mut ModuleSpec,
6199    runtime: &mut SupervisorRuntimeConfig,
6200    registry: &Arc<Registry>,
6201    process_liveness: &SupervisorProcessLiveness,
6202    snapshot: &SharedSnapshot,
6203    child: &mut Option<SupervisedChild>,
6204    commands: &mut mpsc::Receiver<SupervisorCommand>,
6205    requeued: &mut VecDeque<SupervisorCommand>,
6206) -> bool {
6207    match command {
6208        SupervisorCommand::Drain { reply } => {
6209            // A plain stop runs no forwarding drain, so nothing reaches the
6210            // module over its connection before the wait: ask by signal.
6211            let result = drain_optional_child(
6212                &spec.module_id,
6213                spec.protocol,
6214                StopNotice::NotSent,
6215                registry,
6216                runtime.forwarding.as_deref(),
6217                snapshot,
6218                &runtime.terminal_ring,
6219                &runtime.spawn_events,
6220                child,
6221                runtime.drain_timeout,
6222                ModuleState::Stopped,
6223                None,
6224            )
6225            .await;
6226            let registration_released = result.is_ok();
6227            let _ = reply.send(result);
6228            if registration_released {
6229                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6230            }
6231            false
6232        }
6233        SupervisorCommand::Retire { reply } => {
6234            let result = async {
6235                let stop_notice = begin_forwarding_drain_if_configured(
6236                    spec,
6237                    runtime,
6238                    registry,
6239                    snapshot,
6240                    None,
6241                    RouteCloseReason::Disable,
6242                )
6243                .await?;
6244                drain_optional_child(
6245                    &spec.module_id,
6246                    spec.protocol,
6247                    stop_notice,
6248                    registry,
6249                    runtime.forwarding.as_deref(),
6250                    snapshot,
6251                    &runtime.terminal_ring,
6252                    &runtime.spawn_events,
6253                    child,
6254                    runtime.drain_timeout,
6255                    ModuleState::Stopped,
6256                    None,
6257                )
6258                .await
6259            }
6260            .await;
6261            let registration_released = result.is_ok();
6262            let _ = reply.send(result);
6263            if registration_released {
6264                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6265            }
6266            false
6267        }
6268        SupervisorCommand::Restart {
6269            drain_timeout_ms,
6270            received_at_generation,
6271            queued_at,
6272            reply,
6273        } => {
6274            // Without this line a restart that waited in the queue (behind a
6275            // health probe cycle or another command) was invisible: the log
6276            // showed only the drain timing out, minutes after the operator's call.
6277            info!(
6278                module_id = %spec.module_id,
6279                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6280                "restart command dequeued"
6281            );
6282            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6283            // caller whose own request lane rides the module being restarted: the
6284            // caller's in-flight request keeps the drain from quiescing, the drain
6285            // keeps the restart from completing, and the completion keeps the reply
6286            // from releasing the caller — so the drain always timed out and cut the
6287            // initiator with a GOODBYE, even on a healthy module. Replying once the
6288            // restart is validated lets a self-lane caller settle, which is exactly
6289            // what makes the drain succeed. Completion is observable via
6290            // supervisor.list / module status; a post-ack failure lands the module
6291            // in a visible terminal state below rather than in a reply nobody can
6292            // receive.
6293            let validation = match lock_snapshot(snapshot) {
6294                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6295                    module_id: spec.module_id.clone(),
6296                }),
6297                Ok(_) => Ok(()),
6298                Err(err) => Err(err),
6299            };
6300            let initiated = validation.is_ok();
6301            let _ = reply.send(validation);
6302            // A restart asks for a fresh process. Commands run one at a time,
6303            // so a restart queued behind another restart (two operator calls
6304            // in quick succession) is dequeued the moment the first one has
6305            // spawned its replacement -- before that process has sent HELLO.
6306            // Running it would drain and kill the process the first restart
6307            // just produced, which is the opposite of what both callers asked
6308            // for. If a process spawned after this request was received is
6309            // still supervised, the request is already satisfied. Not when the
6310            // configuration changed since that spawn: then the newer process
6311            // predates the spec this restart may exist to apply.
6312            let satisfied_by_generation = if initiated && child.is_some() {
6313                lock_snapshot(snapshot).ok().and_then(|state| {
6314                    (state.spawn_generation > received_at_generation
6315                        && !state.configuration_updated_since_spawn)
6316                        .then_some(state.spawn_generation)
6317                })
6318            } else {
6319                None
6320            };
6321            let satisfied_by_pending = initiated
6322                && child.is_none()
6323                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6324                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6325                    if pending {
6326                        state.coalesced_restart_pending = true;
6327                    }
6328                    pending
6329                });
6330            if satisfied_by_pending {
6331                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6332            } else if let Some(generation) = satisfied_by_generation {
6333                info!(
6334                    module_id = %spec.module_id,
6335                    received_at_generation,
6336                    "restart already satisfied by generation {generation}; not restarting again"
6337                );
6338            } else if initiated {
6339                // Precedence: this restart's operator override, else the module's
6340                // configured budget (already resolved into the runtime).
6341                let drain_timeout = drain_timeout_ms
6342                    .map(Duration::from_millis)
6343                    .unwrap_or(runtime.drain_timeout);
6344                if let Err(err) = restart_child(
6345                    spec,
6346                    runtime,
6347                    registry,
6348                    process_liveness,
6349                    snapshot,
6350                    child,
6351                    drain_timeout,
6352                )
6353                .await
6354                {
6355                    warn!(
6356                        module_id = %spec.module_id,
6357                        error = %err,
6358                        "operator restart failed after initiation ack; module state carries the outcome"
6359                    );
6360                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6361                        state.state = ModuleState::Failed;
6362                        clear_current_process_facts(state);
6363                    });
6364                }
6365            }
6366            true
6367        }
6368        SupervisorCommand::Reload { reply } => {
6369            let result =
6370                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6371            if result.is_ok()
6372                && runtime
6373                    .scheduled_respawn
6374                    .lock()
6375                    .unwrap_or_else(|p| p.into_inner())
6376                    .as_ref()
6377                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6378            {
6379                *runtime
6380                    .deferred_reload_reply
6381                    .lock()
6382                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6383            } else {
6384                let _ = reply.send(result);
6385            }
6386            true
6387        }
6388        SupervisorCommand::SetEnabled { enabled, reply } => {
6389            let result = set_child_enabled(
6390                spec,
6391                runtime,
6392                registry,
6393                process_liveness,
6394                snapshot,
6395                child,
6396                enabled,
6397            )
6398            .await;
6399            let _ = reply.send(result);
6400            true
6401        }
6402        SupervisorCommand::UpdateConfiguration {
6403            spec: next_spec,
6404            health,
6405            drain_timeout_ms,
6406            reply,
6407        } => {
6408            if let Some(handle) = &runtime.supervisor_handle {
6409                handle.apply_identity_configuration(&next_spec);
6410            }
6411            *spec = next_spec;
6412            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6413                state.configuration_updated_since_spawn = true;
6414            });
6415            let health_changed = runtime.health != health;
6416            runtime.health = health;
6417            // Reset the cadence and old endpoint's failure streak on a live
6418            // health-policy change rather than waiting for its old deadline.
6419            if health_changed {
6420                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6421                    state.health = ModuleHealthStatus::default();
6422                });
6423            }
6424            runtime.drain_timeout = drain_timeout_ms
6425                .map(Duration::from_millis)
6426                .unwrap_or(runtime.default_drain_timeout);
6427            *runtime
6428                .effective_drain_timeout
6429                .lock()
6430                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6431            let _ = reply.send(());
6432            true
6433        }
6434        SupervisorCommand::Swap {
6435            ready_timeout,
6436            reply,
6437        } => {
6438            let end = swap::run_swap(
6439                spec,
6440                runtime,
6441                registry,
6442                process_liveness,
6443                snapshot,
6444                child,
6445                commands,
6446                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6447                reply,
6448            )
6449            .await;
6450            requeued.extend(end.requeue);
6451            true
6452        }
6453    }
6454}
6455
6456async fn restart_child(
6457    spec: &ModuleSpec,
6458    runtime: &SupervisorRuntimeConfig,
6459    registry: &Registry,
6460    process_liveness: &SupervisorProcessLiveness,
6461    snapshot: &SharedSnapshot,
6462    child: &mut Option<SupervisedChild>,
6463    drain_timeout: Duration,
6464) -> Result<(), SuperviseError> {
6465    // Restart cycles a running module; it must not silently start a disabled one.
6466    if !lock_snapshot(snapshot)?.enabled {
6467        return Err(SuperviseError::Disabled {
6468            module_id: spec.module_id.clone(),
6469        });
6470    }
6471    let stop_notice = begin_forwarding_drain_with_timeout(
6472        spec,
6473        runtime,
6474        registry,
6475        snapshot,
6476        None,
6477        RouteCloseReason::Restart,
6478        drain_timeout,
6479    )
6480    .await?;
6481
6482    if child.is_some() {
6483        drain_optional_child(
6484            &spec.module_id,
6485            spec.protocol,
6486            stop_notice,
6487            registry,
6488            runtime.forwarding.as_deref(),
6489            snapshot,
6490            &runtime.terminal_ring,
6491            &runtime.spawn_events,
6492            child,
6493            drain_timeout,
6494            ModuleState::Restarting,
6495            Some(true),
6496        )
6497        .await?;
6498    } else {
6499        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6500            state.enabled = true;
6501            state.state = ModuleState::Restarting;
6502            clear_current_process_facts(state);
6503        })?;
6504        release_dead_registration(
6505            registry,
6506            runtime.forwarding.as_deref(),
6507            snapshot,
6508            &spec.module_id,
6509        )
6510        .await?;
6511    }
6512
6513    reset_restart_count(snapshot, &spec.module_id)?;
6514    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6515    schedule_respawn(
6516        runtime,
6517        snapshot,
6518        &spec.module_id,
6519        runtime.restart_policy.backoff,
6520        RespawnKind::Spawn,
6521    )
6522}
6523
6524async fn reload_child(
6525    spec: &ModuleSpec,
6526    runtime: &SupervisorRuntimeConfig,
6527    registry: &Registry,
6528    process_liveness: &SupervisorProcessLiveness,
6529    snapshot: &SharedSnapshot,
6530    child: &mut Option<SupervisedChild>,
6531) -> Result<(), SuperviseError> {
6532    // Reload cycles a running module; it must not silently start a disabled one.
6533    if !lock_snapshot(snapshot)?.enabled {
6534        return Err(SuperviseError::Disabled {
6535            module_id: spec.module_id.clone(),
6536        });
6537    }
6538    let stop_notice = begin_forwarding_drain(
6539        spec,
6540        runtime,
6541        registry,
6542        snapshot,
6543        Some(true),
6544        RouteCloseReason::Reload,
6545    )
6546    .await?;
6547
6548    if child.is_some() {
6549        drain_optional_child(
6550            &spec.module_id,
6551            spec.protocol,
6552            stop_notice,
6553            registry,
6554            runtime.forwarding.as_deref(),
6555            snapshot,
6556            &runtime.terminal_ring,
6557            &runtime.spawn_events,
6558            child,
6559            runtime.drain_timeout,
6560            ModuleState::Restarting,
6561            Some(true),
6562        )
6563        .await?;
6564    } else {
6565        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6566            state.enabled = true;
6567            state.state = ModuleState::Restarting;
6568            clear_current_process_facts(state);
6569        })?;
6570        release_dead_registration(
6571            registry,
6572            runtime.forwarding.as_deref(),
6573            snapshot,
6574            &spec.module_id,
6575        )
6576        .await?;
6577    }
6578
6579    reset_restart_count(snapshot, &spec.module_id)?;
6580    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6581    schedule_respawn(
6582        runtime,
6583        snapshot,
6584        &spec.module_id,
6585        runtime.restart_policy.backoff,
6586        RespawnKind::Reload,
6587    )
6588}
6589
6590async fn finish_reload_child(
6591    spec: &ModuleSpec,
6592    runtime: &SupervisorRuntimeConfig,
6593    registry: &Registry,
6594    process_liveness: &SupervisorProcessLiveness,
6595    snapshot: &SharedSnapshot,
6596    child: &mut Option<SupervisedChild>,
6597) -> Result<(), SuperviseError> {
6598    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6599    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6600        Ok(next_child) => next_child,
6601        Err(err) => {
6602            return handle_reload_spawn_failure(
6603                spec,
6604                runtime,
6605                process_liveness,
6606                snapshot,
6607                child,
6608                format!("new child failed to spawn: {err}"),
6609            )
6610            .await;
6611        }
6612    };
6613    *child = Some(next_child);
6614
6615    let wait_outcome = {
6616        let active_child = child.as_mut().expect("new reload child was just stored");
6617        wait_for_registration_after_reload(
6618            registry,
6619            &spec.module_id,
6620            snapshot,
6621            active_child,
6622            REGISTRY_RELEASE_TIMEOUT,
6623        )
6624        .await?
6625    };
6626
6627    match wait_outcome {
6628        RegistrationWaitOutcome::Registered => {
6629            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6630            Ok(())
6631        }
6632        RegistrationWaitOutcome::Exited(exit_report) => {
6633            if let Some(active_child) = child.as_mut() {
6634                active_child.drain_stderr(&spec.module_id).await;
6635            }
6636            // Keep the reaped child's roster guard until its terminal is written.
6637            // Shutdown waits on that guard, not on the child Option used for respawn.
6638            let mut exited_child = child.take().expect("exited reload child is still stored");
6639            #[cfg(test)]
6640            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6641                gate.reached.notify_one();
6642                gate.resume.notified().await;
6643            }
6644            let result = handle_reload_child_registration_failure(
6645                spec,
6646                runtime,
6647                registry,
6648                process_liveness,
6649                snapshot,
6650                child,
6651                ReloadRegistrationFailure {
6652                    exit_report: registration_failure_exit_report(exit_report),
6653                    reason: exited_child
6654                        .spawn_failure
6655                        .clone()
6656                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6657                },
6658            )
6659            .await;
6660            exited_child.release_roster();
6661            result
6662        }
6663        RegistrationWaitOutcome::TimedOut => {
6664            let mut timed_out_child = child
6665                .take()
6666                .expect("timed-out reload child is still running");
6667            timed_out_child
6668                .start_kill()
6669                .map_err(|source| SuperviseError::Kill {
6670                    module_id: spec.module_id.clone(),
6671                    source,
6672                })?;
6673            let status = timed_out_child
6674                .wait()
6675                .await
6676                .map_err(|source| SuperviseError::Wait {
6677                    module_id: spec.module_id.clone(),
6678                    source,
6679                })?;
6680            timed_out_child.drain_stderr(&spec.module_id).await;
6681            handle_reload_child_registration_failure(
6682                spec,
6683                runtime,
6684                registry,
6685                process_liveness,
6686                snapshot,
6687                child,
6688                ReloadRegistrationFailure {
6689                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6690                        snapshot,
6691                        &timed_out_child,
6692                        &status,
6693                    )),
6694                    reason: format!(
6695                        "new child did not register within {:?}",
6696                        REGISTRY_RELEASE_TIMEOUT
6697                    ),
6698                },
6699            )
6700            .await
6701        }
6702    }
6703}
6704
6705async fn set_child_enabled(
6706    spec: &ModuleSpec,
6707    runtime: &SupervisorRuntimeConfig,
6708    registry: &Registry,
6709    process_liveness: &SupervisorProcessLiveness,
6710    snapshot: &SharedSnapshot,
6711    child: &mut Option<SupervisedChild>,
6712    enabled: bool,
6713) -> Result<bool, SuperviseError> {
6714    let (current_enabled, current_state, respawn_pending) = {
6715        let state = lock_snapshot(snapshot)?;
6716        (state.enabled, state.state, state.respawn_pending)
6717    };
6718    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6719    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6720    // clean (Stopped) has no live process and no other in-band recovery — the
6721    // operator's start is the explicit recovery act and resets the budget. Without
6722    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6723    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6724    // the one providing every agent's shell.
6725    let revive_terminal = enabled
6726        && current_enabled
6727        && child.is_none()
6728        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6729            || (current_state == ModuleState::Restarting && !respawn_pending));
6730    if current_enabled == enabled && !revive_terminal {
6731        return Ok(false);
6732    }
6733
6734    if enabled {
6735        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6736            state.enabled = true;
6737            state.state = ModuleState::Starting;
6738            clear_current_process_facts(state);
6739        })?;
6740        #[cfg(test)]
6741        if runtime.test_seed_stale_facts_before_enable_spawn {
6742            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6743                state.process_alive = true;
6744                state.pid = Some(41);
6745                state.spawned_at_ms = Some(42);
6746                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6747                state.spawned_file_identity = Some(SpawnedFileIdentity {
6748                    device: 43,
6749                    inode: 44,
6750                });
6751            })?;
6752        }
6753        release_dead_registration(
6754            registry,
6755            runtime.forwarding.as_deref(),
6756            snapshot,
6757            &spec.module_id,
6758        )
6759        .await?;
6760        reset_restart_count(snapshot, &spec.module_id)?;
6761        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6762        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6763            Ok(next_child) => next_child,
6764            Err(err) => {
6765                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6766                    state.state = ModuleState::Failed;
6767                    clear_current_process_facts(state);
6768                }) {
6769                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6770                }
6771                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6772                return Err(err);
6773            }
6774        };
6775        *child = Some(next_child);
6776        debug!(module_id = %spec.module_id, "supervised module enabled");
6777        Ok(true)
6778    } else {
6779        let stop_notice = begin_forwarding_drain_if_configured(
6780            spec,
6781            runtime,
6782            registry,
6783            snapshot,
6784            Some(false),
6785            RouteCloseReason::Disable,
6786        )
6787        .await?;
6788        drain_optional_child(
6789            &spec.module_id,
6790            spec.protocol,
6791            stop_notice,
6792            registry,
6793            runtime.forwarding.as_deref(),
6794            snapshot,
6795            &runtime.terminal_ring,
6796            &runtime.spawn_events,
6797            child,
6798            runtime.drain_timeout,
6799            ModuleState::Disabled,
6800            Some(false),
6801        )
6802        .await?;
6803        debug!(module_id = %spec.module_id, "supervised module disabled");
6804        Ok(true)
6805    }
6806}
6807
6808#[allow(clippy::too_many_arguments)]
6809async fn on_child_exit(
6810    spec: &ModuleSpec,
6811    policy: RestartPolicy,
6812    registry: &Registry,
6813    snapshot: &SharedSnapshot,
6814    terminal_ring: &Arc<Mutex<TerminalRing>>,
6815    spawn_events: &SpawnEventFeed,
6816    roster: &ChildRoster,
6817    exit_report: ExitReport,
6818) -> NextAction {
6819    // Once the daemon has begun shutting down, no exit is a crash to recover
6820    // from: the module is exiting because the daemon is going away (EOF on its
6821    // connection, or a service manager signalling the whole cgroup). Record it
6822    // as such and never schedule a respawn, which would only start a process
6823    // for the shutdown to end again.
6824    if roster.is_closed() {
6825        return on_child_exit_during_daemon_shutdown(
6826            spec,
6827            registry,
6828            snapshot,
6829            terminal_ring,
6830            spawn_events,
6831            exit_report,
6832        )
6833        .await;
6834    }
6835    // Every stop the supervisor itself asks for (operator stop, disable,
6836    // restart, reload, swap, a health restart, a drain that runs out of budget)
6837    // takes the child out of the supervise loop and reaps it in
6838    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
6839    // that reaches this point was not requested by the daemon.
6840    //
6841    // For a subc-wire module a clean exit is still a stop: those modules are
6842    // written to re-raise SIGTERM, so a stray outside signal already reads as a
6843    // crash, and exiting 0 is a deliberate choice the module made. A
6844    // `protocol: "none"` module is a stock program we cannot change, and many
6845    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
6846    // stop would leave the module down for good after any stray signal, so it
6847    // goes through the crash path instead: it spends restart budget, respawns
6848    // with the crash backoff, and ends `failed` when the budget runs out.
6849    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6850        && running_protocol(spec, snapshot) == ModuleProtocol::None;
6851    match exit_report.kind {
6852        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6853            info!(
6854                module_id = %spec.module_id,
6855                exit_code = ?exit_report.code,
6856                exit_signal = ?exit_report.signal,
6857                "supervised module exited cleanly"
6858            );
6859            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6860                state.state = ModuleState::Stopped;
6861                clear_current_process_facts(state);
6862                state.last_exit = Some(exit_report.clone());
6863            }) {
6864                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6865            }
6866            record_terminal(
6867                &spec.module_id,
6868                terminal_ring,
6869                spawn_events,
6870                &exit_report,
6871                TerminalDisposition::Stopped,
6872            );
6873            let registration_released = match wait_for_registration_release(
6874                registry,
6875                &spec.module_id,
6876                REGISTRY_RELEASE_TIMEOUT,
6877            )
6878            .await
6879            {
6880                Ok(()) => true,
6881                Err(err) => {
6882                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6883                    false
6884                }
6885            };
6886            NextAction::Stop {
6887                registration_released,
6888            }
6889        }
6890        ExitKind::Clean | ExitKind::Crash => {
6891            if unrequested_clean_exit_of_protocol_none {
6892                warn!(
6893                    module_id = %spec.module_id,
6894                    exit_code = ?exit_report.code,
6895                    exit_signal = ?exit_report.signal,
6896                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
6897                );
6898            } else {
6899                warn!(
6900                    module_id = %spec.module_id,
6901                    exit_code = ?exit_report.code,
6902                    exit_signal = ?exit_report.signal,
6903                    "supervised module exited abnormally (crash)"
6904                );
6905            }
6906            let mut restart_schedule = None;
6907            let mut disposition = TerminalDisposition::Disabled;
6908            // Set only when the budget is what stopped the module, so the
6909            // terminal record says which limit was hit rather than leaving
6910            // `failed` to be read as "crashed once, badly".
6911            let mut disposition_detail = lock_snapshot(snapshot)
6912                .ok()
6913                .and_then(|mut state| state.spawn_failure.take());
6914            let now = Instant::now();
6915            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6916                clear_current_process_facts(state);
6917                state.last_exit = Some(exit_report.clone());
6918                if state.enabled {
6919                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
6920                        state.state = ModuleState::Restarting;
6921                        restart_schedule = Some(schedule);
6922                        disposition = TerminalDisposition::Restarting;
6923                    } else {
6924                        state.state = ModuleState::Failed;
6925                        disposition = TerminalDisposition::Failed;
6926                        let budget = policy.budget_exhausted_detail();
6927                        disposition_detail =
6928                            Some(disposition_detail.take().map_or_else(
6929                                || budget.clone(),
6930                                |cause| format!("{cause}; {budget}"),
6931                            ));
6932                    }
6933                } else {
6934                    state.state = ModuleState::Disabled;
6935                    disposition = TerminalDisposition::Disabled;
6936                }
6937            }) {
6938                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
6939                return NextAction::Stop {
6940                    registration_released: false,
6941                };
6942            }
6943            if disposition == TerminalDisposition::Failed {
6944                // The window is in the message, not only in the fields: this line
6945                // is read in a scrollback where a bare `max_restarts=3` reads as a
6946                // lifetime cap and sends the operator looking for three crashes
6947                // that never happened together.
6948                error!(
6949                    module_id = %spec.module_id,
6950                    max_restarts = policy.max_restarts,
6951                    window_secs = policy.window.as_secs(),
6952                    "module stopped: {}",
6953                    policy.budget_exhausted_detail()
6954                );
6955            }
6956            record_terminal_with_detail(
6957                &spec.module_id,
6958                terminal_ring,
6959                spawn_events,
6960                &exit_report,
6961                disposition,
6962                disposition_detail,
6963            );
6964
6965            if let Some(schedule) = restart_schedule {
6966                NextAction::Restart {
6967                    schedule: Some(schedule),
6968                }
6969            } else {
6970                let registration_released = match wait_for_registration_release(
6971                    registry,
6972                    &spec.module_id,
6973                    REGISTRY_RELEASE_TIMEOUT,
6974                )
6975                .await
6976                {
6977                    Ok(()) => true,
6978                    Err(err) => {
6979                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
6980                        false
6981                    }
6982                };
6983                NextAction::Stop {
6984                    registration_released,
6985                }
6986            }
6987        }
6988        ExitKind::DeliberateSeverance => {
6989            warn!(
6990                module_id = %spec.module_id,
6991                exit_code = ?exit_report.code,
6992                exit_signal = ?exit_report.signal,
6993                "supervised module exited after deliberate connection severance"
6994            );
6995            let mut should_restart = false;
6996            let mut disposition = TerminalDisposition::Disabled;
6997            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6998                clear_current_process_facts(state);
6999                state.last_exit = Some(exit_report.clone());
7000                state.lifetime_restarts += 1;
7001                if state.enabled {
7002                    state.state = ModuleState::Restarting;
7003                    should_restart = true;
7004                    disposition = TerminalDisposition::Restarting;
7005                } else {
7006                    state.state = ModuleState::Disabled;
7007                }
7008            }) {
7009                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7010                return NextAction::Stop {
7011                    registration_released: false,
7012                };
7013            }
7014            record_terminal(
7015                &spec.module_id,
7016                terminal_ring,
7017                spawn_events,
7018                &exit_report,
7019                disposition,
7020            );
7021
7022            if should_restart {
7023                NextAction::Restart { schedule: None }
7024            } else {
7025                let registration_released = match wait_for_registration_release(
7026                    registry,
7027                    &spec.module_id,
7028                    REGISTRY_RELEASE_TIMEOUT,
7029                )
7030                .await
7031                {
7032                    Ok(()) => true,
7033                    Err(err) => {
7034                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7035                        false
7036                    }
7037                };
7038                NextAction::Stop {
7039                    registration_released,
7040                }
7041            }
7042        }
7043    }
7044}
7045
7046async fn on_child_exit_during_daemon_shutdown(
7047    spec: &ModuleSpec,
7048    registry: &Registry,
7049    snapshot: &SharedSnapshot,
7050    terminal_ring: &Arc<Mutex<TerminalRing>>,
7051    spawn_events: &SpawnEventFeed,
7052    exit_report: ExitReport,
7053) -> NextAction {
7054    info!(
7055        module_id = %spec.module_id,
7056        exit_code = ?exit_report.code,
7057        exit_signal = ?exit_report.signal,
7058        exit_kind = ?exit_report.kind,
7059        "supervised module exited during daemon shutdown; not restarting it"
7060    );
7061    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7062        state.state = ModuleState::Stopped;
7063        clear_current_process_facts(state);
7064        state.last_exit = Some(exit_report.clone());
7065    }) {
7066        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7067    }
7068    record_terminal(
7069        &spec.module_id,
7070        terminal_ring,
7071        spawn_events,
7072        &exit_report,
7073        TerminalDisposition::DaemonShutdown,
7074    );
7075    let registration_released =
7076        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7077            .await
7078            .is_ok();
7079    NextAction::Stop {
7080        registration_released,
7081    }
7082}
7083
7084fn record_wait_error_terminal(
7085    module_id: &str,
7086    terminal_ring: &Arc<Mutex<TerminalRing>>,
7087    spawn_events: &SpawnEventFeed,
7088) {
7089    record_terminal(
7090        module_id,
7091        terminal_ring,
7092        spawn_events,
7093        &wait_error_exit_report(),
7094        TerminalDisposition::Failed,
7095    );
7096}
7097
7098fn record_terminal(
7099    module_id: &str,
7100    terminal_ring: &Arc<Mutex<TerminalRing>>,
7101    spawn_events: &SpawnEventFeed,
7102    exit_report: &ExitReport,
7103    disposition: TerminalDisposition,
7104) {
7105    record_terminal_with_detail(
7106        module_id,
7107        terminal_ring,
7108        spawn_events,
7109        exit_report,
7110        disposition,
7111        None,
7112    );
7113}
7114
7115/// The ring lock is held only to capture the read (see
7116/// `TerminalJournal::capture_read`), so this module's exits keep recording
7117/// while the journal files are read. Blocking: it reads files.
7118fn durable_terminal_history_of(
7119    terminal_ring: &Mutex<TerminalRing>,
7120    module_id: &str,
7121) -> subc_control::TerminalHistory {
7122    let read = terminal_ring
7123        .lock()
7124        .unwrap_or_else(|p| p.into_inner())
7125        .capture_durable_history();
7126    read.read(module_id)
7127}
7128
7129fn record_terminal_with_detail(
7130    module_id: &str,
7131    terminal_ring: &Arc<Mutex<TerminalRing>>,
7132    spawn_events: &SpawnEventFeed,
7133    exit_report: &ExitReport,
7134    disposition: TerminalDisposition,
7135    disposition_detail: Option<String>,
7136) {
7137    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7138    let record = TerminalRecord {
7139        exit_code: exit_report.code,
7140        exit_signal: exit_report.signal,
7141        at_ms: exit_report.at_ms,
7142        disposition,
7143        exit_kind: exit_report.kind.into(),
7144        disposition_detail,
7145    };
7146    terminal_ring
7147        .lock()
7148        .unwrap_or_else(|poisoned| poisoned.into_inner())
7149        .record_exit(module_id, record);
7150}
7151
7152fn untrack_if_registration_released(
7153    process_liveness: &SupervisorProcessLiveness,
7154    registry: &Registry,
7155    module_id: &str,
7156    snapshot: &SharedSnapshot,
7157) {
7158    match registry.get_module(module_id) {
7159        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7160        Ok(Some(_)) => {}
7161        Err(err) => {
7162            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7163        }
7164    }
7165}
7166
7167/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7168/// then apply the module's configured entries minus daemon-private capture keys.
7169///
7170/// Separated from `spawn_child` only so it can be asserted without spawning a
7171/// process — a duplicate of this logic in a test would pass while the real one
7172/// drifted, which is the defect class this function exists to avoid.
7173/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7174/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7175/// either and the argument would stop a stock binary from starting at all.
7176/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7177///
7178/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7179/// through [`apply_wire_spawn_args_for_role`].
7180#[cfg(test)]
7181fn apply_wire_spawn_args(
7182    command: &mut Command,
7183    spec: &ModuleSpec,
7184    connection_file_path: Option<&std::path::Path>,
7185    handle: Option<&SupervisorHandle>,
7186) -> Result<Option<NonceHandoff>, SuperviseError> {
7187    apply_wire_spawn_args_for_role(
7188        command,
7189        spec,
7190        connection_file_path,
7191        handle,
7192        SpawnRole::Plain,
7193    )
7194}
7195
7196/// The read end of a spawn's launch-nonce pipe, prepared by
7197/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7198/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7199/// handoff and keeps only the environment copy.
7200#[cfg(unix)]
7201type NonceHandoff = subc_os::LaunchNonceHandoff;
7202#[cfg(not(unix))]
7203type NonceHandoff = std::convert::Infallible;
7204
7205/// Prepare wire identity for a plain spawn or a swap candidate.
7206///
7207/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7208/// a separate candidate token so the still-serving incumbent and its consumers
7209/// keep their nonce. Both records are installed before the process exists, so
7210/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7211///
7212/// On Unix the nonce is delivered only through a pipe. It is written into
7213/// a pipe whose read end the child gets as descriptor 3, named by
7214/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7215/// process of the same user cannot read it with `ps eww`. That handoff is
7216/// returned rather than installed here, because installing it replaces
7217/// whatever the child has at descriptor 3 and so must be the last pre-exec
7218/// step, after the Linux cgroup placement that the caller registers later.
7219/// Windows retains the environment handoff until restricted handle inheritance
7220/// can be implemented outside std's process primitives.
7221fn apply_wire_spawn_args_for_role(
7222    command: &mut Command,
7223    spec: &ModuleSpec,
7224    connection_file_path: Option<&std::path::Path>,
7225    handle: Option<&SupervisorHandle>,
7226    role: SpawnRole,
7227) -> Result<Option<NonceHandoff>, SuperviseError> {
7228    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7229    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7230    // included: a daemon started from a module's process tree inherits it,
7231    // and passing it on would point the child at a descriptor it does not
7232    // have.
7233    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7234    // Remove inherited or configured copies too: withholding must mean absent.
7235    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7236    if spec.protocol == ModuleProtocol::None {
7237        return Ok(None);
7238    }
7239    if let Some(connection_file_path) = connection_file_path {
7240        command.arg(SUBC_ARG).arg(connection_file_path);
7241    }
7242
7243    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7244    // route.open attestation. Reserved modules additionally use the same nonce
7245    // for HELLO id-squatting protection. A respawn rotates both records.
7246    let nonce = generate_launch_nonce()?;
7247    if let Some(handle) = handle {
7248        match role {
7249            SpawnRole::Plain => {
7250                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7251                if spec.reserved {
7252                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7253                }
7254            }
7255            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7256        }
7257    }
7258    #[cfg(unix)]
7259    let handoff = {
7260        let handoff =
7261            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7262                program: spec.program.clone(),
7263                source,
7264                cgroup_path: None,
7265            })?;
7266        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7267        Some(handoff)
7268    };
7269    #[cfg(not(unix))]
7270    let handoff = None;
7271    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7272    // handle to this child without leaking it to concurrently spawned processes.
7273    #[cfg(not(unix))]
7274    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7275    Ok(handoff)
7276}
7277
7278fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7279    command.env_remove(CK_LOG_ENV);
7280    // The spawn role is the supervisor's to set, and only on a swap candidate
7281    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7282    // it, is what makes it absent on a plain spawn: the daemon's own
7283    // environment could carry it, and so could a spec built outside daemon
7284    // config (config refuses it as an `env` key). A module reading it on a
7285    // plain restart would pick the long swap budget and leave callers waiting.
7286    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7287    for (key, value) in &spec.env {
7288        // cortexkit-log currently exposes retention only as a Rust struct, not
7289        // environment names. These values are daemon-private sink metadata and
7290        // must never become a public child-process contract by being inherited.
7291        if matches!(
7292            key.as_str(),
7293            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7294        ) || key == SUBC_SPAWN_ROLE_ENV
7295        {
7296            continue;
7297        }
7298        command.env(key, value);
7299    }
7300}
7301
7302/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7303/// of a blue/green swap.
7304#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7305enum SpawnRole {
7306    Plain,
7307    SwapCandidate,
7308}
7309
7310/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7311/// `apply_child_env` has already removed the variable for every spawn.
7312fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7313    if role == SpawnRole::SwapCandidate {
7314        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7315    }
7316}
7317
7318fn spawn_child(
7319    spec: &ModuleSpec,
7320    connection_file_path: Option<&std::path::Path>,
7321    handle: Option<&SupervisorHandle>,
7322    ring: &Arc<Mutex<StderrRing>>,
7323    capture_logs_dir: Option<&std::path::Path>,
7324    roster: &ChildRoster,
7325    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7326) -> Result<SupervisedChild, SuperviseError> {
7327    spawn_child_in_slot(
7328        spec,
7329        connection_file_path,
7330        handle,
7331        ring,
7332        capture_logs_dir,
7333        roster,
7334        #[cfg(target_os = "linux")]
7335        cgroup_placement,
7336        SpawnRole::Plain,
7337        false,
7338    )
7339}
7340
7341/// Spawn one process of `spec` into a slot.
7342///
7343/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7344/// A swap candidate needs a different cgroup from the process it is replacing,
7345/// which is still alive: in the same cgroup the two would be one kill domain,
7346/// and killing a failed candidate could take the incumbent with it.
7347///
7348/// The stderr capture file is `<module_id>.stderr.log` for every process of
7349/// the module, whichever slot it is in, because that is the one file
7350/// `ck module logs` reads. During a swap's overlap both processes append to it;
7351/// the daemon writes whole lines, so the two interleave by line, which is also
7352/// the merged view an operator wants while a swap runs.
7353#[allow(clippy::too_many_arguments)]
7354fn spawn_child_in_slot(
7355    spec: &ModuleSpec,
7356    connection_file_path: Option<&std::path::Path>,
7357    handle: Option<&SupervisorHandle>,
7358    ring: &Arc<Mutex<StderrRing>>,
7359    capture_logs_dir: Option<&std::path::Path>,
7360    roster: &ChildRoster,
7361    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7362    role: SpawnRole,
7363    alternate_slot: bool,
7364) -> Result<SupervisedChild, SuperviseError> {
7365    if roster.is_closed() {
7366        return Err(SuperviseError::Spawn {
7367            program: spec.program.clone(),
7368            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7369            cgroup_path: None,
7370        });
7371    }
7372    #[cfg(target_os = "linux")]
7373    let cgroup_name = {
7374        // Slot names alone are not kill domains: a retired incumbent may still
7375        // be draining when a later enable/restart spawns into the same slot.
7376        // Decimal entropy keeps the suffix unambiguous; Placement performs
7377        // the module-id escaping and constructs the filesystem path.
7378        if cgroup_placement.is_none() {
7379            swap::cgroup_name(&spec.module_id, alternate_slot)
7380        } else {
7381            let nonce = generate_launch_nonce()?;
7382            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7383            // Leave room for byte escaping and the suffix under NAME_MAX. The
7384            // label is only for humans; the nonce identifies the kill domain.
7385            let mut end = spec.module_id.len().min(64);
7386            while !spec.module_id.is_char_boundary(end) {
7387                end -= 1;
7388            }
7389            format!(
7390                "{}_{suffix}",
7391                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7392            )
7393        }
7394    };
7395    #[cfg(not(target_os = "linux"))]
7396    let _ = alternate_slot;
7397    #[cfg(target_os = "macos")]
7398    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7399    #[cfg(not(target_os = "macos"))]
7400    let mut command = Command::new(&spec.program);
7401    command.args(&spec.args);
7402    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7403    // that is the whole of the intent, so remove that one key rather than the
7404    // environment.
7405    //
7406    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7407    // and took the POSIX environment with it. Modules spawned that way had no
7408    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7409    // logging:
7410    //
7411    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7412    //     both unset it fell back to the temp dir alone and `ck` could not find
7413    //     a daemon running on the same machine from inside any module's process
7414    //     tree — reporting a path the file has never lived at, which reads as
7415    //     "the daemon did not write its file".
7416    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7417    //     the RELATIVE `.local/share`, so a module deriving its own store path
7418    //     resolved it against its own CWD. That is the store-fragmentation
7419    //     defect the daemon already refuses in config (`parse_doc` rejects a
7420    //     relative `storage.data_home`) arriving by derivation instead.
7421    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7422    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7423    //     quietly rather than erroring.
7424    //
7425    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7426    // offered one candidate under /tmp while the file sat in /run/user/1000.
7427    //
7428    // A configured module is unaffected either way: `module_spec()` puts the
7429    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7430    // wins over anything ambient.
7431    apply_child_env(&mut command, spec);
7432    apply_spawn_role(&mut command, role);
7433    let nonce_handoff =
7434        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7435
7436    #[cfg(target_os = "linux")]
7437    let cgroup_path = cgroup_placement
7438        .map(|placement| placement.module_path(&cgroup_name))
7439        .transpose()
7440        .map_err(|source| SuperviseError::Cgroup {
7441            module_id: spec.module_id.clone(),
7442            source,
7443        })?;
7444    #[cfg(not(target_os = "linux"))]
7445    let cgroup_path: Option<PathBuf> = None;
7446    #[cfg(target_os = "linux")]
7447    if let Some(path) = &cgroup_path {
7448        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7449            if let Some(placement) = cgroup_placement {
7450                remove_module_cgroup(placement, &cgroup_name);
7451            }
7452            return Err(error);
7453        }
7454    }
7455
7456    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7457        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7458        match ChildOutputSink::open(&path, capture_retention(spec)) {
7459            Ok(sink) => sink,
7460            Err(error) => {
7461                warn!(
7462                    module_id = %spec.module_id,
7463                    path = %path.display(),
7464                    error = %error,
7465                    "could not open child output capture file; forwarding to stderr"
7466                );
7467                ChildOutputSink::Stderr
7468            }
7469        }
7470    } else {
7471        ChildOutputSink::Stderr
7472    };
7473
7474    command.stdout(Stdio::piped());
7475    command.stderr(Stdio::piped());
7476    command.kill_on_drop(true);
7477    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7478    // before exec). In the daemon's group, a service manager that kills the
7479    // job's process group when the daemon exits (launchd's default) killed
7480    // every module at the same moment its control connection closed, so no
7481    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7482    // module is reached only by the daemon: the EOF it sees when its
7483    // connection closes, and the bounded stop in `child_roster` for anything
7484    // still running after that. On Linux this composes with the cgroup
7485    // placement above: that is a pre_exec write to cgroup.procs, std performs
7486    // setpgid in the child before running pre_exec callbacks, and the two
7487    // change independent process attributes.
7488    //
7489    // stdin is /dev/null because a process outside the terminal's foreground
7490    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7491    // by hand would otherwise hand down. Under a service manager stdin is
7492    // already /dev/null.
7493    #[cfg(unix)]
7494    command.process_group(0);
7495    command.stdin(Stdio::null());
7496    // The LAST pre-exec step, after the cgroup placement above: installing the
7497    // nonce at descriptor 3 replaces whatever the child had there, which could
7498    // be the descriptor an earlier step writes through.
7499    #[cfg(unix)]
7500    if let Some(handoff) = nonce_handoff {
7501        handoff.install_last(command.as_std_mut());
7502    }
7503    #[cfg(not(unix))]
7504    let _ = nonce_handoff;
7505
7506    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7507    // cannot run a single instruction -- and therefore cannot spawn a
7508    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7509    // other two steps and why the window matters.
7510    #[cfg(windows)]
7511    subc_jobobject::suspend_on_create_async(&mut command);
7512    let mut child = match command.spawn() {
7513        Ok(child) => child,
7514        Err(source) => {
7515            #[cfg(target_os = "linux")]
7516            if let Some(placement) = cgroup_placement {
7517                remove_module_cgroup(placement, &cgroup_name);
7518            }
7519            return Err(SuperviseError::Spawn {
7520                program: spec.program.clone(),
7521                source,
7522                cgroup_path,
7523            });
7524        }
7525    };
7526    // The parent must close its writer now: the acknowledgement pipe reports EOF
7527    // only when every writer is gone, and the child's copy closes when the
7528    // trampoline replaces itself with the module. Command holds only an integer
7529    // in its pre_exec callback, not another writer.
7530    #[cfg(target_os = "macos")]
7531    drop(exec_ack);
7532
7533    // Containment, steps 2 and 3: assign while suspended, then resume.
7534    #[cfg(windows)]
7535    let job = contain_spawned_child(&child, spec)?;
7536    let spawned_at_ms = unix_ms_now();
7537    let spawned_from = spec.program.clone();
7538    let spawned_file_identity = spawned_file_identity(&spawned_from);
7539    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7540        program: spec.program.clone(),
7541        source: io::Error::other("spawned child exposed no live pid"),
7542        cgroup_path: cgroup_path.clone(),
7543    })?;
7544    let process_start_time = crate::provenance::process_start_time(pid);
7545    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7546    // Unix spawn returns after exec's error pipe closes. The kernel image is
7547    // therefore the executable to compare during a future orphan sweep: PATH
7548    // lookup and shebang interpretation may select a different file from the
7549    // configured program. Keep the literal program's identity for provenance,
7550    // but never use it as proof that a recorded pid may be signalled.
7551    let recorded_image = observe_spawned_image(pid);
7552    // spawn() confirms only the first exec, into the trampoline. Never persist
7553    // the trampoline image; the asynchronous acknowledgement publishes the
7554    // module image once the trampoline has replaced itself with the module.
7555    #[cfg(target_os = "macos")]
7556    let recorded_image = if privacy_exec.is_some() {
7557        None
7558    } else {
7559        recorded_image
7560    };
7561    #[cfg(target_os = "linux")]
7562    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7563    #[cfg(not(target_os = "linux"))]
7564    let recorded_cgroup_name = None;
7565    let roster_guard = roster.admit(
7566        spec.module_id.clone(),
7567        pid,
7568        spec.protocol,
7569        process_start_time,
7570        crate::child_roster::RecordedIdentity {
7571            start_time: recorded_image.map(|image| image.start_time),
7572            executable: recorded_image
7573                .and_then(|image| image.executable)
7574                .map(crate::live_children::ExecutableIdentity::from),
7575            cgroup_name: recorded_cgroup_name,
7576            #[cfg(target_os = "linux")]
7577            cgroup_placement: cgroup_placement.cloned(),
7578        },
7579    );
7580    // The check at the top of this function can pass just before daemon
7581    // shutdown begins, and the process is only in the roster from here on.
7582    // The shutdown stop returns as soon as it finds the roster empty, so a
7583    // process admitted after that look would outlive the daemon. The roster
7584    // is closed before the stop first reads it and admission happens under
7585    // the roster's lock, so either the stop sees this process or this check
7586    // sees the roster closed: end the process now rather than start a module
7587    // the daemon is about to stop.
7588    if roster.is_closed() {
7589        // This child was never admitted, so there is no module protocol shutdown to wait for.
7590        #[cfg(target_os = "linux")]
7591        kill_module_cgroup(cgroup_placement, &cgroup_name);
7592        if let Err(error) = child.start_kill() {
7593            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7594        }
7595        #[cfg(target_os = "linux")]
7596        if let Some(placement) = cgroup_placement {
7597            // This spawn was never admitted, so shutdown has no roster entry
7598            // to await. Do not detach its cleanup: the runtime could exit
7599            // before that task reaps the rejected child and removes its group.
7600            while matches!(child.try_wait(), Ok(None)) {
7601                std::thread::yield_now();
7602            }
7603            if matches!(
7604                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7605                subc_cgroup::KillOutcome::Killed
7606            ) {
7607                if let Ok(path) = placement.module_path(&cgroup_name) {
7608                    while std::fs::read_to_string(path.join("cgroup.events"))
7609                        .ok()
7610                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7611                    {
7612                        std::thread::yield_now();
7613                    }
7614                }
7615            }
7616            remove_module_cgroup(placement, &cgroup_name);
7617        }
7618        drop(roster_guard);
7619        return Err(SuperviseError::Spawn {
7620            program: spec.program.clone(),
7621            source: io::Error::other(
7622                "the daemon began shutting down while this process was starting; ended it",
7623            ),
7624            cgroup_path,
7625        });
7626    }
7627
7628    let stdout_pump = match child.stdout.take() {
7629        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7630        None => {
7631            warn!(
7632                module_id = %spec.module_id,
7633                "spawned child exposed no stdout pipe; file capture will be incomplete"
7634            );
7635            None
7636        }
7637    };
7638    let stderr_pump = match child.stderr.take() {
7639        Some(stderr) => {
7640            let generation = ring
7641                .lock()
7642                .unwrap_or_else(|poisoned| poisoned.into_inner())
7643                .begin_process();
7644            Some(StderrPump {
7645                task: tokio::spawn(pump_stderr_to(
7646                    stderr,
7647                    Arc::clone(ring),
7648                    generation,
7649                    output_sink,
7650                )),
7651                generation,
7652            })
7653        }
7654        None => {
7655            // Spawning succeeded but the pipe did not materialise. Recording it as
7656            // uncaptured keeps the tail honest: the alternative is an empty tail
7657            // that reads as a module which printed nothing.
7658            ring.lock()
7659                .unwrap_or_else(|poisoned| poisoned.into_inner())
7660                .mark_not_captured("stderr pipe was not available on spawn");
7661            warn!(
7662                module_id = %spec.module_id,
7663                "spawned child exposed no stderr pipe; tail will be unavailable"
7664            );
7665            None
7666        }
7667    };
7668
7669    Ok(SupervisedChild {
7670        child,
7671        protocol: spec.protocol,
7672        #[cfg(target_os = "linux")]
7673        module_id: cgroup_name,
7674        #[cfg(target_os = "linux")]
7675        cgroup_placement: cgroup_placement.cloned(),
7676        #[cfg(windows)]
7677        job,
7678        stdout_pump,
7679        stderr_pump,
7680        stderr_ring: Arc::clone(ring),
7681        spawned_at_ms,
7682        spawned_from,
7683        spawned_file_identity,
7684        process_start_time,
7685        process_identity,
7686        pid,
7687        roster_guard: Some(roster_guard),
7688        #[cfg(target_os = "macos")]
7689        privacy_exec,
7690        spawn_failure: None,
7691    })
7692}
7693
7694#[cfg(target_os = "linux")]
7695pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7696    use subc_cgroup::KillOutcome;
7697    match subc_cgroup::kill_module(placement, module_id) {
7698        KillOutcome::Killed => {}
7699        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7700            debug!(
7701                module_id,
7702                "cgroup tree kill unavailable; using direct-child kill"
7703            );
7704        }
7705        KillOutcome::IoError { path, error } => {
7706            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7707        }
7708    }
7709}
7710
7711/// Contain a freshly spawned Windows child and start it.
7712///
7713/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7714/// child assigned **while it is still suspended** (step 1 is
7715/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7716///
7717/// A child that is never resumed hangs forever holding a pid, so a resume
7718/// failure kills the child and fails the spawn rather than returning a
7719/// `SupervisedChild` that can never run.
7720///
7721/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7722/// it did before this existed, whereas refusing to start one would be a new
7723/// outage. It is logged at warn because it means a helper process could leak.
7724#[cfg(windows)]
7725fn contain_spawned_child(
7726    child: &Child,
7727    spec: &ModuleSpec,
7728) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7729    let module_id = spec.module_id.as_str();
7730    let Some(pid) = child.id() else {
7731        // The child exited between spawn and here. Its tree, if it made one,
7732        // needs no containment: nothing is left to contain.
7733        warn!(
7734            module_id,
7735            "spawned child had already exited before containment; no job object attached"
7736        );
7737        return Ok(None);
7738    };
7739
7740    let job = match subc_jobobject::JobObject::new() {
7741        Ok(job) => job,
7742        Err(source) => {
7743            warn!(
7744                module_id,
7745                error = %source,
7746                "could not create a job object; this module's helper processes will not be \
7747                 reaped on teardown"
7748            );
7749            // Resume regardless: leaving the child suspended would turn a
7750            // containment gap into a hung module.
7751            resume_suspended_child(pid, spec)?;
7752            return Ok(None);
7753        }
7754    };
7755
7756    if let Err(source) = job.assign(child) {
7757        warn!(
7758            module_id,
7759            error = %source,
7760            "could not assign the child to its job object; this module's helper processes \
7761             will not be reaped on teardown"
7762        );
7763        resume_suspended_child(pid, spec)?;
7764        return Ok(None);
7765    }
7766
7767    resume_suspended_child(pid, spec)?;
7768    Ok(Some(job))
7769}
7770
7771/// Resume a suspended child, killing it if it cannot be started.
7772///
7773/// A suspended process holds a pid and does nothing, so there is no useful
7774/// state to return: the caller gets an error and the spawn fails.
7775#[cfg(windows)]
7776fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7777    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7778        // Kill it here rather than leaving a suspended process for the caller
7779        // to notice; `kill_on_drop` would eventually do this, but the module
7780        // would have been reported as running in between.
7781        let _ = std::process::Command::new("taskkill.exe")
7782            .args(["/PID", &pid.to_string(), "/T", "/F"])
7783            .stdin(Stdio::null())
7784            .stdout(Stdio::null())
7785            .stderr(Stdio::null())
7786            .status();
7787        return Err(SuperviseError::Spawn {
7788            program: spec.program.clone(),
7789            source,
7790            cgroup_path: None,
7791        });
7792    }
7793    Ok(())
7794}
7795
7796#[cfg(target_os = "linux")]
7797fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7798    match placement.remove_module(module_id) {
7799        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7800        Err(error) => warn!(
7801            module_id,
7802            error = %error,
7803            "could not remove module cgroup after process exit; continuing teardown"
7804        ),
7805    }
7806}
7807
7808#[cfg(target_os = "linux")]
7809async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7810    // Reaping the direct child is not proof its descendants exited. End the
7811    // residual tree and wait for the kernel's population fact before rmdir;
7812    // otherwise a successful parent wait leaks a directory on each restart.
7813    if matches!(
7814        subc_cgroup::kill_module(Some(placement), module_id),
7815        subc_cgroup::KillOutcome::Killed
7816    ) {
7817        if let Ok(path) = placement.module_path(module_id) {
7818            while std::fs::read_to_string(path.join("cgroup.events"))
7819                .ok()
7820                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7821            {
7822                sleep(Duration::from_millis(1)).await;
7823            }
7824        }
7825    }
7826    remove_module_cgroup(placement, module_id);
7827}
7828
7829#[cfg(target_os = "linux")]
7830fn apply_cgroup_placement(
7831    command: &mut Command,
7832    spec: &ModuleSpec,
7833    path: &std::path::Path,
7834) -> Result<(), SuperviseError> {
7835    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7836        module_id: spec.module_id.clone(),
7837        source,
7838    })
7839}
7840
7841fn capture_retention(spec: &ModuleSpec) -> Retention {
7842    let defaults = Retention::default();
7843    let value = |name: &str| {
7844        spec.env
7845            .iter()
7846            .rev()
7847            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7848    };
7849    Retention {
7850        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7851            .and_then(|value| value.parse().ok())
7852            .unwrap_or(defaults.max_file_mb),
7853        keep: value(CAPTURE_KEEP_ENV)
7854            .and_then(|value| value.parse().ok())
7855            .unwrap_or(defaults.keep),
7856        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7857            .and_then(|value| value.parse().ok())
7858            .unwrap_or(defaults.max_age_days),
7859    }
7860}
7861
7862/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
7863/// module's registration to the exact process the supervisor spawned.
7864fn generate_launch_nonce() -> Result<String, SuperviseError> {
7865    let mut bytes = [0u8; 32];
7866    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7867        reason: source.to_string(),
7868    })?;
7869    let mut hex = String::with_capacity(64);
7870    for b in bytes {
7871        use std::fmt::Write;
7872        let _ = write!(hex, "{b:02x}");
7873    }
7874    Ok(hex)
7875}
7876
7877/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
7878/// signal about how many leading bytes matched.
7879fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7880    if a.len() != b.len() {
7881        return false;
7882    }
7883    let mut diff = 0u8;
7884    for (x, y) in a.iter().zip(b.iter()) {
7885        diff |= x ^ y;
7886    }
7887    diff == 0
7888}
7889
7890/// The kernel's image after an acknowledged exec, shared by ordinary launches
7891/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
7892/// not identities inferred from a configured pathname.
7893fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
7894    subc_os::Process::open(pid)
7895        .ok()
7896        .flatten()
7897        .and_then(|process| process.observe())
7898}
7899
7900fn spawn_and_mark_running(
7901    spec: &ModuleSpec,
7902    runtime: &SupervisorRuntimeConfig,
7903    snapshot: &SharedSnapshot,
7904) -> Result<SupervisedChild, SuperviseError> {
7905    let child = spawn_child(
7906        spec,
7907        runtime.connection_file_path.as_deref(),
7908        runtime.supervisor_handle.as_ref(),
7909        &runtime.stderr_ring,
7910        runtime.capture_logs_dir.as_deref(),
7911        &runtime.child_roster,
7912        #[cfg(target_os = "linux")]
7913        runtime.cgroup_placement.as_ref(),
7914    )?;
7915    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
7916    Ok(child)
7917}
7918
7919enum RegistrationWaitOutcome {
7920    Registered,
7921    Exited(ExitReport),
7922    TimedOut,
7923}
7924
7925struct ReloadRegistrationFailure {
7926    exit_report: ExitReport,
7927    reason: String,
7928}
7929
7930#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7931enum BusyGaugeObservation {
7932    Quiescent,
7933    Busy,
7934    Omitted,
7935}
7936
7937fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
7938    let Some(metrics) = metrics.and_then(Value::as_object) else {
7939        return BusyGaugeObservation::Omitted;
7940    };
7941    let mut sum = 0u128;
7942    for gauge in gauges {
7943        let Some(value) = metrics.get(gauge) else {
7944            return BusyGaugeObservation::Omitted;
7945        };
7946        let Some(value) = value.as_u64() else {
7947            return BusyGaugeObservation::Busy;
7948        };
7949        sum = sum.saturating_add(u128::from(value));
7950    }
7951    if sum == 0 {
7952        BusyGaugeObservation::Quiescent
7953    } else {
7954        BusyGaugeObservation::Busy
7955    }
7956}
7957
7958fn declared_busy_gauges(
7959    registry: &Registry,
7960    module_id: &str,
7961) -> Result<Vec<String>, SuperviseError> {
7962    busy_gauges_of(
7963        registry
7964            .get_module(module_id)
7965            .map_err(SuperviseError::Registry)?,
7966    )
7967}
7968
7969/// [`declared_busy_gauges`] for the registration a connection holds, in any
7970/// slot: after cutover the incumbent is no longer the id's active
7971/// registration, and its own manifest is the one that names its gauges.
7972fn declared_busy_gauges_for_connection(
7973    registry: &Registry,
7974    connection_id: ConnectionId,
7975) -> Result<Vec<String>, SuperviseError> {
7976    busy_gauges_of(
7977        registry
7978            .get_module_by_connection(connection_id)
7979            .map_err(SuperviseError::Registry)?,
7980    )
7981}
7982
7983fn busy_gauges_of(
7984    registration: Option<crate::registry::ModuleRegistration>,
7985) -> Result<Vec<String>, SuperviseError> {
7986    let Some(registration) = registration else {
7987        return Ok(Vec::new());
7988    };
7989    let Some(self_signals) = registration.manifest.self_signals else {
7990        return Ok(Vec::new());
7991    };
7992
7993    let mut gauges = Vec::new();
7994    for declaration in self_signals {
7995        if declaration.kind != SelfSignalKind::Busy {
7996            continue;
7997        }
7998        match declaration.anchored_to {
7999            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8000                gauges.extend(declared)
8001            }
8002            _ => {
8003                // An invalid Busy anchor is fail-safe: the empty name cannot be
8004                // present in a conforming health report, so this drain stays busy.
8005                gauges.push(String::new());
8006            }
8007        }
8008    }
8009    Ok(gauges)
8010}
8011
8012/// Wait for `endpoint` to have nothing in flight and, when the module declares
8013/// busy gauges, for a health probe to report them quiet. The probe is addressed
8014/// by `scope`: a swap's superseded incumbent must be asked about its own
8015/// gauges, and by module id the probe would reach the promoted candidate.
8016async fn wait_for_forwarding_quiescence(
8017    forwarding: &ForwardingTable,
8018    module_id: &str,
8019    runtime: &SupervisorRuntimeConfig,
8020    endpoint: crate::ModuleEndpointId,
8021    deadline: Instant,
8022    busy_gauges: &[String],
8023    scope: DrainScope,
8024) -> Result<bool, SuperviseError> {
8025    let mut gauges_quiescent = busy_gauges.is_empty();
8026    let mut next_probe_at = Instant::now();
8027    let mut omission_counted = false;
8028
8029    loop {
8030        let now = Instant::now();
8031        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8032            let report = match scope {
8033                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8034                DrainScope::Endpoint(endpoint) => {
8035                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8036                }
8037            };
8038            gauges_quiescent = match report {
8039                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8040                    BusyGaugeObservation::Quiescent => true,
8041                    BusyGaugeObservation::Busy => false,
8042                    BusyGaugeObservation::Omitted => {
8043                        if !omission_counted {
8044                            forwarding
8045                                .counters()
8046                                .increment_drains_with_undeclared_gauge();
8047                            omission_counted = true;
8048                        }
8049                        false
8050                    }
8051                },
8052                Err(err) => {
8053                    warn!(
8054                        module_id,
8055                        error = %err,
8056                        "drain health.check did not produce declared busy gauges; treating module as busy"
8057                    );
8058                    false
8059                }
8060            };
8061            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8062        }
8063
8064        let in_flight = forwarding
8065            .endpoint_in_flight_count(endpoint)
8066            .map_err(SuperviseError::Forwarding)?;
8067        if in_flight == 0 && gauges_quiescent {
8068            return Ok(true);
8069        }
8070
8071        let now = Instant::now();
8072        if now >= deadline {
8073            return Ok(false);
8074        }
8075        let mut wait = deadline
8076            .saturating_duration_since(now)
8077            .min(REGISTRY_RELEASE_POLL);
8078        if !busy_gauges.is_empty() {
8079            wait = wait.min(next_probe_at.saturating_duration_since(now));
8080        }
8081        sleep(wait).await;
8082    }
8083}
8084
8085/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8086///
8087/// `Ok` is always honest and passed straight through -- the wait actually measured
8088/// in-flight state. `Err` means the wait produced no measurement at all (the
8089/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8090/// constant: the drain did not complete. Never recomputed from route state, never a
8091/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8092fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8093    match wait_result {
8094        Ok(drained) => *drained,
8095        Err(_) => false,
8096    }
8097}
8098
8099fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8100    for released in released_routes {
8101        let frame = match Frame::build_with_version(
8102            released.negotiated_ver,
8103            FrameType::Goodbye,
8104            control_flags(),
8105            released.channel,
8106            released.epoch,
8107            0,
8108            Vec::new(),
8109        ) {
8110            Ok(frame) => frame,
8111            Err(err) => {
8112                warn!(
8113                    route_channel = released.channel,
8114                    error = %err,
8115                    "failed to build supervisor drain route GOODBYE frame"
8116                );
8117                continue;
8118            }
8119        };
8120        if !released.close_on_delivery_failure() {
8121            crate::forwarding::send_module_route_goodbye(
8122                &forwarding.counters(),
8123                &released.sink,
8124                frame,
8125                released.module_id.as_deref(),
8126                "supervisor drain",
8127            );
8128            continue;
8129        }
8130        if let Err(err) = released.sink.try_send(frame) {
8131            warn!(
8132                target_connection_id = released.connection_id.get(),
8133                route_channel = released.channel,
8134                error = %err,
8135                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8136            );
8137            let _ = forwarding.escalate_client_delivery_failure(
8138                released.connection_id,
8139                released.channel,
8140                released.epoch,
8141                CloseReason::new(
8142                    "route_goodbye_delivery_failed",
8143                    format!(
8144                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8145                        released.channel
8146                    ),
8147                ),
8148                crate::forwarding::UndeliveredFrame {
8149                    module_id: released.module_id.as_deref(),
8150                    sink: &released.sink,
8151                },
8152            );
8153        }
8154    }
8155}
8156
8157fn send_module_draining(
8158    module_id: &str,
8159    reason: RouteCloseReason,
8160    deadline_ms: u64,
8161    target: &ModuleDrainTarget,
8162) {
8163    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8164        reason,
8165        deadline_ms,
8166    }) {
8167        Ok(body) => body,
8168        Err(err) => {
8169            warn!(
8170                module_id,
8171                error = %err,
8172                "failed to encode module draining command"
8173            );
8174            return;
8175        }
8176    };
8177    let frame = match Frame::build_with_version(
8178        target.negotiated_ver,
8179        FrameType::Push,
8180        control_flags(),
8181        0,
8182        0,
8183        0,
8184        body,
8185    ) {
8186        Ok(frame) => frame,
8187        Err(err) => {
8188            warn!(
8189                module_id,
8190                error = %err,
8191                "failed to build module draining command frame"
8192            );
8193            return;
8194        }
8195    };
8196    if let Err(err) = target.sink.try_send(frame) {
8197        warn!(
8198            module_id,
8199            target_connection_id = target.endpoint.connection_id.get(),
8200            error = %err,
8201            "module draining command was not delivered to peer"
8202        );
8203    }
8204}
8205
8206/// The channel-0 GOODBYE that tells a module its stop is planned.
8207fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8208    match Frame::build_with_version(
8209        negotiated_ver,
8210        FrameType::Goodbye,
8211        control_flags(),
8212        0,
8213        0,
8214        0,
8215        Vec::new(),
8216    ) {
8217        Ok(frame) => Some(frame),
8218        Err(err) => {
8219            warn!(
8220                module_id,
8221                error = %err,
8222                "failed to build module GOODBYE frame"
8223            );
8224            None
8225        }
8226    }
8227}
8228
8229/// Send every registered module connection its module GOODBYE at daemon
8230/// shutdown, then request that connection's close.
8231///
8232/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8233/// before EOF, so the GOODBYE must reach the socket before the close. A close
8234/// request does not wait for the connection's queued frames: its writer gets a
8235/// bounded grace after the close, is aborted if it overruns it, and the daemon
8236/// process may exit before that grace ends. So with `wait_for_flush`, each
8237/// connection is closed only after its writer has acknowledged writing the
8238/// GOODBYE, or once a short shared budget runs out, so one module that is not
8239/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8240/// are only queued, for a shutdown the operator has told to stop waiting.
8241/// A connection that is already gone is skipped.
8242#[cfg(unix)]
8243async fn send_module_goodbyes_for_daemon_shutdown(
8244    forwarding: &Arc<ForwardingTable>,
8245    reason: &CloseReason,
8246    wait_for_flush: bool,
8247) {
8248    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8249    let targets = match forwarding.module_connections() {
8250        Ok(targets) => targets,
8251        Err(err) => {
8252            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8253            return;
8254        }
8255    };
8256    let deadline = Instant::now() + GOODBYE_BUDGET;
8257    let mut sends = tokio::task::JoinSet::new();
8258    for target in targets {
8259        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8260            continue;
8261        };
8262        if !wait_for_flush {
8263            if let Err(err) = target.sink.try_send(frame) {
8264                debug!(
8265                    module_id = %target.module_id,
8266                    error = %err,
8267                    "shutdown module GOODBYE was not queued"
8268                );
8269            }
8270            continue;
8271        }
8272        let forwarding = Arc::clone(forwarding);
8273        let reason = reason.clone();
8274        sends.spawn(async move {
8275            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8276                Ok(Ok(())) => {}
8277                Ok(Err(err)) => debug!(
8278                    module_id = %target.module_id,
8279                    error = %err,
8280                    "module connection closed before its shutdown GOODBYE was written"
8281                ),
8282                Err(_) => warn!(
8283                    module_id = %target.module_id,
8284                    budget = ?GOODBYE_BUDGET,
8285                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8286                ),
8287            }
8288            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8289        });
8290    }
8291    // Every task ends by the shared deadline, so this wait is bounded too.
8292    while sends.join_next().await.is_some() {}
8293}
8294
8295fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8296    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8297        return;
8298    };
8299    if let Err(err) = target.sink.try_send(frame) {
8300        warn!(
8301            module_id,
8302            target_connection_id = target.endpoint.connection_id.get(),
8303            error = %err,
8304            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8305        );
8306        forwarding.request_connection_close(
8307            target.endpoint.connection_id,
8308            CloseReason::new(
8309                "module_goodbye_delivery_failed",
8310                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8311            ),
8312        );
8313    }
8314}
8315
8316#[derive(Clone, Copy)]
8317struct ForwardingDrainContext<'a> {
8318    spec: &'a ModuleSpec,
8319    runtime: &'a SupervisorRuntimeConfig,
8320    registry: &'a Registry,
8321    scope: DrainScope,
8322}
8323
8324/// Which process a forwarding drain addresses.
8325#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8326enum DrainScope {
8327    /// Whatever endpoint is active for the module id: every plain stop,
8328    /// restart and reload. Also moves the module's state to `Draining`.
8329    Active,
8330    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8331    /// module id would resolve to the promoted candidate and leave neither
8332    /// process routable. The module's state is left alone, since the promoted
8333    /// candidate is what it describes and that process is running.
8334    Endpoint(crate::ModuleEndpointId),
8335}
8336
8337/// Whether a child being drained has already been asked to stop by the time
8338/// its drain wait starts.
8339///
8340/// The drain wait is the same budget whatever this says. What it decides is
8341/// whether the supervisor must ask by signal before that wait begins: a child
8342/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8343/// healthy or not.
8344#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8345enum StopNotice {
8346    /// The module was sent `module.draining` and a module GOODBYE over its own
8347    /// registered connection, and stops itself.
8348    SentOverConnection,
8349    /// The forwarding drain found no registered connection for the module: a
8350    /// subc child spawned moments ago that has not sent HELLO yet, or a
8351    /// `protocol: "none"` child, which never registers.
8352    NoConnection,
8353    /// This path sends nothing over the module's connection: the supervisor has
8354    /// no forwarding table, or the caller stops the child without a forwarding
8355    /// drain.
8356    NotSent,
8357}
8358
8359async fn begin_forwarding_drain(
8360    spec: &ModuleSpec,
8361    runtime: &SupervisorRuntimeConfig,
8362    registry: &Registry,
8363    snapshot: &SharedSnapshot,
8364    enabled: Option<bool>,
8365    reason: RouteCloseReason,
8366) -> Result<StopNotice, SuperviseError> {
8367    let Some(forwarding) = runtime.forwarding.as_ref() else {
8368        return Err(SuperviseError::ReloadUnavailable {
8369            module_id: spec.module_id.clone(),
8370            reason: "supervisor was not configured with a forwarding table".to_string(),
8371        });
8372    };
8373
8374    begin_forwarding_drain_with(
8375        forwarding,
8376        ForwardingDrainContext {
8377            spec,
8378            runtime,
8379            registry,
8380            scope: DrainScope::Active,
8381        },
8382        snapshot,
8383        enabled,
8384        reason,
8385        runtime.drain_timeout,
8386    )
8387    .await
8388}
8389
8390async fn begin_forwarding_drain_if_configured(
8391    spec: &ModuleSpec,
8392    runtime: &SupervisorRuntimeConfig,
8393    registry: &Registry,
8394    snapshot: &SharedSnapshot,
8395    enabled: Option<bool>,
8396    reason: RouteCloseReason,
8397) -> Result<StopNotice, SuperviseError> {
8398    begin_forwarding_drain_with_timeout(
8399        spec,
8400        runtime,
8401        registry,
8402        snapshot,
8403        enabled,
8404        reason,
8405        runtime.drain_timeout,
8406    )
8407    .await
8408}
8409
8410/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8411/// budget, for paths where the operator overrides the module's configured one
8412/// (`supervisor.restart{drain_timeout_ms}`).
8413async fn begin_forwarding_drain_with_timeout(
8414    spec: &ModuleSpec,
8415    runtime: &SupervisorRuntimeConfig,
8416    registry: &Registry,
8417    snapshot: &SharedSnapshot,
8418    enabled: Option<bool>,
8419    reason: RouteCloseReason,
8420    drain_timeout: Duration,
8421) -> Result<StopNotice, SuperviseError> {
8422    let Some(forwarding) = runtime.forwarding.as_ref() else {
8423        return Ok(StopNotice::NotSent);
8424    };
8425
8426    begin_forwarding_drain_with(
8427        forwarding,
8428        ForwardingDrainContext {
8429            spec,
8430            runtime,
8431            registry,
8432            scope: DrainScope::Active,
8433        },
8434        snapshot,
8435        enabled,
8436        reason,
8437        drain_timeout,
8438    )
8439    .await
8440}
8441
8442async fn begin_forwarding_drain_with(
8443    forwarding: &ForwardingTable,
8444    context: ForwardingDrainContext<'_>,
8445    snapshot: &SharedSnapshot,
8446    enabled: Option<bool>,
8447    reason: RouteCloseReason,
8448    drain_timeout: Duration,
8449) -> Result<StopNotice, SuperviseError> {
8450    let ForwardingDrainContext {
8451        spec,
8452        runtime,
8453        registry,
8454        scope,
8455    } = context;
8456    debug_assert_ne!(reason, RouteCloseReason::Crash);
8457    let terminal = matches!(reason, RouteCloseReason::Disable);
8458    let drain_started_at = Instant::now();
8459    let drain_deadline = drain_started_at + drain_timeout;
8460    let deadline_ms =
8461        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8462    let busy_gauges = match scope {
8463        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8464        DrainScope::Endpoint(endpoint) => {
8465            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8466        }
8467    };
8468
8469    // Admission gate first: route.open/commit and route REQUEST admission are closed
8470    // before the first quiescence check, so the outstanding count can only fall.
8471    let gate_started = Instant::now();
8472    let drain_target = match scope {
8473        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8474        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8475    }
8476    .map_err(SuperviseError::Forwarding)?;
8477    // The instant admission closed, and how long taking the forwarding write
8478    // lock to close it took. The timeout line reports only the quiescence
8479    // wait, so without this a drain that started late looked like one that
8480    // started on time.
8481    info!(
8482        module_id = %spec.module_id,
8483        ?reason,
8484        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8485        connected = drain_target.is_some(),
8486        "module drain began; route admission closed"
8487    );
8488    if scope == DrainScope::Active {
8489        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8490            state.state = ModuleState::Draining;
8491            state.draining_to_replace =
8492                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8493            if let Some(enabled) = enabled {
8494                state.enabled = enabled;
8495            }
8496        })?;
8497    }
8498
8499    let Some(target) = drain_target.as_ref() else {
8500        // Nothing was sent: the module has no registered connection to carry
8501        // `module.draining` or a GOODBYE. The caller must not assume the child
8502        // was asked to stop.
8503        return Ok(StopNotice::NoConnection);
8504    };
8505    {
8506        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8507        let routes = forwarding
8508            .endpoint_routes(target.endpoint)
8509            .map_err(SuperviseError::Forwarding)?;
8510        let routes_notified = routes.len();
8511        crate::control::send_route_control_pushes(
8512            forwarding,
8513            routes.clone(),
8514            ClientControlPush::RouteClosing {
8515                module_id: spec.module_id.clone(),
8516                channels: Vec::new(),
8517                reason,
8518            },
8519        );
8520        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8521
8522        // `route.closing` was just sent above: from here on every return path,
8523        // including an early one, MUST send `route.closed` before propagating
8524        // anything else. A client holds `closing` as a promise that a verdict is
8525        // coming; leaving early without `closed` strands it waiting forever, since
8526        // `closing` carries no timeout of its own.
8527        let wait_result = wait_for_forwarding_quiescence(
8528            forwarding,
8529            &spec.module_id,
8530            runtime,
8531            target.endpoint,
8532            drain_deadline,
8533            &busy_gauges,
8534            scope,
8535        )
8536        .await;
8537        let drained = drained_after_quiescence_wait(&wait_result);
8538        if let Err(err) = &wait_result {
8539            error!(
8540                module_id = %spec.module_id,
8541                ?reason,
8542                error = %err,
8543                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8544            );
8545        } else if !drained {
8546            // Name what the drain waited on. Without it the line says only that
8547            // something did not settle, and "one wedged call" and "every
8548            // session's held stream" read the same; the first is a module bug,
8549            // the second is a module that should end its streams on
8550            // module.draining. Read before teardown releases the routes.
8551            let holdouts = forwarding
8552                .endpoint_drain_holdouts(target.endpoint)
8553                .unwrap_or_default();
8554            warn!(
8555                module_id = %spec.module_id,
8556                waited = ?drain_timeout,
8557                ?reason,
8558                held_requests = holdouts.requests,
8559                held_routes = holdouts.routes,
8560                total_routes = holdouts.total_routes,
8561                top_connections = ?holdouts.top_connections,
8562                // `module_channel:corr`, so the module can find each held request
8563                // in its own log; capped, so `held_requests` is the full count.
8564                held = %holdouts
8565                    .held
8566                    .iter()
8567                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8568                    .collect::<Vec<_>>()
8569                    .join(","),
8570                "route drain timed out before request quiescence; forcing teardown"
8571            );
8572        }
8573        crate::control::send_route_control_pushes(
8574            forwarding,
8575            routes,
8576            ClientControlPush::RouteClosed {
8577                module_id: spec.module_id.clone(),
8578                channels: Vec::new(),
8579                reason,
8580                drained,
8581                abandoned: target.abandoned_bindings.len() as u32,
8582                excluded_subscriptions: target.excluded_subscriptions,
8583                terminal: Some(terminal),
8584            },
8585        );
8586        wait_result?;
8587
8588        // `route.closed` has now been sent unconditionally above. From here the
8589        // remaining steps are cleanup (route + module GOODBYE) rather than a
8590        // promise the client is waiting on, but a lock-poisoned
8591        // `release_module_endpoint_routes` would otherwise skip the module
8592        // GOODBYE silently too -- send it before propagating the error.
8593        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8594            Ok(routes) => routes,
8595            Err(err) => {
8596                warn!(
8597                    module_id = %spec.module_id,
8598                    ?reason,
8599                    error = %err,
8600                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8601                );
8602                send_module_goodbye(&spec.module_id, forwarding, target);
8603                return Err(SuperviseError::Forwarding(err));
8604            }
8605        };
8606        let route_goodbye_count = released_routes.len();
8607        send_route_goodbyes(forwarding, released_routes);
8608        send_module_goodbye(&spec.module_id, forwarding, target);
8609
8610        // The drain's happy path was previously silent: every emission above is
8611        // best-effort with only its failure arm logged, so "were consumers told"
8612        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8613        // hang where the open question was exactly whether teardown notice went
8614        // out). One summary line makes that class decidable in one grep.
8615        info!(
8616            module_id = %spec.module_id,
8617            ?reason,
8618            routes_notified,
8619            route_goodbyes = route_goodbye_count,
8620            abandoned_reservations = target.abandoned_bindings.len(),
8621            excluded_subscriptions = target.excluded_subscriptions,
8622            drained,
8623            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8624        );
8625    }
8626
8627    Ok(StopNotice::SentOverConnection)
8628}
8629
8630/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8631/// the only slot a plain (non-swap) spawn can register into.
8632async fn wait_for_registration_after_reload(
8633    registry: &Registry,
8634    module_id: &str,
8635    snapshot: &SharedSnapshot,
8636    child: &mut SupervisedChild,
8637    wait: Duration,
8638) -> Result<RegistrationWaitOutcome, SuperviseError> {
8639    wait_for_slot_registration(
8640        registry,
8641        crate::registry::RegistrationSlot::Active(module_id),
8642        module_id,
8643        snapshot,
8644        child,
8645        wait,
8646    )
8647    .await
8648}
8649
8650/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8651///
8652/// Keyed on the slot rather than the bare module id because during a swap the
8653/// id's active slot is already held by the incumbent: an id-keyed wait would
8654/// report the incumbent's registration as the candidate's and a candidate that
8655/// never registers would look registered. A swap candidate waits on
8656/// `crate::registry::RegistrationSlot::Candidate`.
8657async fn wait_for_slot_registration(
8658    registry: &Registry,
8659    slot: crate::registry::RegistrationSlot<'_>,
8660    module_id: &str,
8661    snapshot: &SharedSnapshot,
8662    child: &mut SupervisedChild,
8663    wait: Duration,
8664) -> Result<RegistrationWaitOutcome, SuperviseError> {
8665    let deadline = Instant::now() + wait;
8666    loop {
8667        if registry
8668            .registration(slot)
8669            .map_err(SuperviseError::Registry)?
8670            .is_some()
8671        {
8672            return Ok(RegistrationWaitOutcome::Registered);
8673        }
8674
8675        let now = Instant::now();
8676        if now >= deadline {
8677            return Ok(RegistrationWaitOutcome::TimedOut);
8678        }
8679        let remaining = deadline.saturating_duration_since(now);
8680        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8681
8682        tokio::select! {
8683            wait_result = child.wait() => {
8684                let status = wait_result.map_err(|source| SuperviseError::Wait {
8685                    module_id: module_id.to_string(),
8686                    source,
8687                })?;
8688                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8689                    snapshot,
8690                    child,
8691                    &status,
8692                )));
8693            }
8694            _ = sleep(poll) => {}
8695        }
8696    }
8697}
8698
8699fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8700    // A replacement process that exits before HELLO did not provide service, even
8701    // if it used status 0. Count it against the restart cap as a new-binary failure.
8702    if exit_report.kind != ExitKind::DeliberateSeverance {
8703        exit_report.kind = ExitKind::Crash;
8704    }
8705    exit_report
8706}
8707
8708async fn handle_reload_child_registration_failure(
8709    spec: &ModuleSpec,
8710    runtime: &SupervisorRuntimeConfig,
8711    registry: &Registry,
8712    process_liveness: &SupervisorProcessLiveness,
8713    snapshot: &SharedSnapshot,
8714    _child: &mut Option<SupervisedChild>,
8715    failure: ReloadRegistrationFailure,
8716) -> Result<(), SuperviseError> {
8717    let ReloadRegistrationFailure {
8718        exit_report,
8719        reason,
8720    } = failure;
8721    match on_child_exit(
8722        spec,
8723        runtime.restart_policy,
8724        registry,
8725        snapshot,
8726        &runtime.terminal_ring,
8727        &runtime.spawn_events,
8728        &runtime.child_roster,
8729        exit_report,
8730    )
8731    .await
8732    {
8733        NextAction::Stop {
8734            registration_released,
8735        } => {
8736            if registration_released {
8737                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8738            }
8739        }
8740        NextAction::Restart { schedule } => {
8741            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8742                schedule.delay
8743            });
8744            if let Some(schedule) = schedule {
8745                log_crash_respawn(&spec.module_id, schedule);
8746            }
8747            schedule_respawn(
8748                runtime,
8749                snapshot,
8750                &spec.module_id,
8751                delay,
8752                RespawnKind::Spawn,
8753            )?;
8754        }
8755    }
8756    Err(SuperviseError::ReloadFailed {
8757        module_id: spec.module_id.clone(),
8758        reason,
8759    })
8760}
8761
8762async fn handle_reload_spawn_failure(
8763    spec: &ModuleSpec,
8764    runtime: &SupervisorRuntimeConfig,
8765    process_liveness: &SupervisorProcessLiveness,
8766    snapshot: &SharedSnapshot,
8767    _child: &mut Option<SupervisedChild>,
8768    reason: String,
8769) -> Result<(), SuperviseError> {
8770    let now = Instant::now();
8771    let mut schedule = None;
8772    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8773        clear_current_process_facts(state);
8774        if state.enabled {
8775            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8776            state.state = if schedule.is_some() {
8777                ModuleState::Restarting
8778            } else {
8779                ModuleState::Failed
8780            };
8781        } else {
8782            state.state = ModuleState::Disabled;
8783        }
8784    })?;
8785    if let Some(schedule) = schedule {
8786        schedule_respawn(
8787            runtime,
8788            snapshot,
8789            &spec.module_id,
8790            schedule.delay,
8791            RespawnKind::Spawn,
8792        )?;
8793    } else {
8794        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8795    }
8796    Err(SuperviseError::ReloadFailed {
8797        module_id: spec.module_id.clone(),
8798        reason,
8799    })
8800}
8801
8802fn control_flags() -> Flags {
8803    Flags::new(false, Priority::Passive, false)
8804}
8805
8806#[allow(clippy::too_many_arguments)]
8807async fn drain_optional_child(
8808    module_id: &str,
8809    protocol: ModuleProtocol,
8810    stop_notice: StopNotice,
8811    registry: &Registry,
8812    forwarding: Option<&ForwardingTable>,
8813    snapshot: &SharedSnapshot,
8814    terminal_ring: &Arc<Mutex<TerminalRing>>,
8815    spawn_events: &SpawnEventFeed,
8816    child: &mut Option<SupervisedChild>,
8817    drain_timeout: Duration,
8818    final_state: ModuleState,
8819    enabled: Option<bool>,
8820) -> Result<(), SuperviseError> {
8821    if let Some(child) = child.take() {
8822        drain_child_to_state(
8823            module_id,
8824            protocol,
8825            stop_notice,
8826            registry,
8827            forwarding,
8828            snapshot,
8829            terminal_ring,
8830            spawn_events,
8831            child,
8832            drain_timeout,
8833            final_state,
8834            enabled,
8835        )
8836        .await
8837    } else {
8838        update_snapshot(snapshot, Some(module_id), |state| {
8839            state.state = final_state;
8840            if let Some(enabled) = enabled {
8841                state.enabled = enabled;
8842            }
8843            clear_current_process_facts(state);
8844        })?;
8845        release_dead_registration(registry, forwarding, snapshot, module_id).await
8846    }
8847}
8848
8849#[allow(clippy::too_many_arguments)]
8850async fn drain_child_to_state(
8851    module_id: &str,
8852    _protocol: ModuleProtocol,
8853    stop_notice: StopNotice,
8854    registry: &Registry,
8855    forwarding: Option<&ForwardingTable>,
8856    snapshot: &SharedSnapshot,
8857    terminal_ring: &Arc<Mutex<TerminalRing>>,
8858    spawn_events: &SpawnEventFeed,
8859    mut child: SupervisedChild,
8860    drain_timeout: Duration,
8861    final_state: ModuleState,
8862    enabled: Option<bool>,
8863) -> Result<(), SuperviseError> {
8864    let protocol = child.protocol;
8865    update_snapshot(snapshot, Some(module_id), |state| {
8866        state.state = ModuleState::Draining;
8867        state.draining_to_replace = final_state == ModuleState::Restarting;
8868        if let Some(enabled) = enabled {
8869            state.enabled = enabled;
8870        }
8871    })?;
8872
8873    // The wait below is the same budget in every case; what differs is
8874    // whether anything has ASKED the child to stop before it starts. Only a
8875    // forwarding drain that reached the module's registered connection has
8876    // (`module.draining`, then a module GOODBYE). Every other child was told
8877    // nothing: a `protocol: "none"` module, which never registers; a subc
8878    // module spawned moments ago that has not sent HELLO yet; or a stop that
8879    // runs no forwarding drain. Without a signal the budget is only a delay
8880    // in front of SIGKILL -- and the not-yet-registered child is the worst
8881    // case, because it registers into a module that is already draining,
8882    // is never told, and is killed while healthy.
8883    if stop_notice != StopNotice::SentOverConnection {
8884        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8885            info!(
8886                module_id,
8887                pid = child.pid,
8888                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8889                "module has no connection yet; requesting stop by signal"
8890            );
8891        }
8892        request_graceful_stop(module_id, &child);
8893    }
8894
8895    let exit_report = match timeout(drain_timeout, child.wait()).await {
8896        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
8897        Ok(Err(source)) => {
8898            fail_snapshot(snapshot, Some(module_id), None);
8899            return Err(SuperviseError::Wait {
8900                module_id: module_id.to_string(),
8901                source,
8902            });
8903        }
8904        Err(_) => {
8905            // Mirror the sibling arm above: state is already `Draining`, and an
8906            // error propagated from here would strand it there -- a state
8907            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
8908            // `Failed | Stopped`), leaving an operator Restart as the only exit.
8909            // `Failed` before `?` keeps the module operator-visible and
8910            // revivable. Trigger is an ESRCH race (process exits between the
8911            // drain timeout firing and the kill) or a post-kill wait failure
8912            // (issue #34).
8913            //
8914            // Logged because the kill is otherwise visible only as signal 9 in
8915            // the terminal ring, and the budget it follows can be long enough
8916            // that consumers see a stretch of refusals with no stated cause.
8917            warn!(
8918                module_id,
8919                pid = child.pid,
8920                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8921                reason = ?final_state,
8922                ?stop_notice,
8923                "drain budget expired before the module exited; killing it"
8924            );
8925            child.start_kill().map_err(|source| {
8926                fail_snapshot(snapshot, Some(module_id), None);
8927                SuperviseError::Kill {
8928                    module_id: module_id.to_string(),
8929                    source,
8930                }
8931            })?;
8932            let status = child.wait().await.map_err(|source| {
8933                fail_snapshot(snapshot, Some(module_id), None);
8934                SuperviseError::Wait {
8935                    module_id: module_id.to_string(),
8936                    source,
8937                }
8938            })?;
8939            classify_reaped_child_exit(snapshot, &child, &status)
8940        }
8941    };
8942
8943    update_snapshot(snapshot, Some(module_id), |state| {
8944        state.state = final_state;
8945        if let Some(enabled) = enabled {
8946            state.enabled = enabled;
8947        }
8948        clear_current_process_facts(state);
8949        state.last_exit = Some(exit_report.clone());
8950        if exit_report.kind == ExitKind::DeliberateSeverance {
8951            state.lifetime_restarts += 1;
8952        }
8953    })?;
8954    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
8955    record_terminal_with_detail(
8956        module_id,
8957        terminal_ring,
8958        spawn_events,
8959        &exit_report,
8960        terminal_disposition(final_state),
8961        detail,
8962    );
8963    child.drain_stderr(module_id).await;
8964
8965    release_dead_registration(registry, forwarding, snapshot, module_id).await
8966}
8967
8968/// Ask a child that nothing else has asked to stop, by signal.
8969///
8970/// A registered subc module is asked over its own connection: the drain sends
8971/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
8972/// module GOODBYE, and the module stops itself. A module that speaks no subc
8973/// wire receives none of that, and neither does a subc module that has not
8974/// registered yet, so for them the drain budget would be pure delay in front of
8975/// a SIGKILL -- and for a process with a store to flush (JetStream is the
8976/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
8977/// into a recovery on the next start.
8978///
8979/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
8980/// rule rather than an optimisation: that module's graceful stop is already
8981/// running by the time its child is drained, and a signal would race it.
8982///
8983/// Best-effort by construction. A child that has already exited is the ordinary
8984/// case rather than an error (the kill lands on a reaped or exiting pid), so a
8985/// failure is logged at debug and the wait-then-kill below still decides the
8986/// outcome.
8987#[cfg(unix)]
8988fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
8989    let Some(pid) = child
8990        .id()
8991        .and_then(|pid| i32::try_from(pid).ok())
8992        .and_then(rustix::process::Pid::from_raw)
8993    else {
8994        debug!(
8995            module_id,
8996            "no pid to signal for teardown; falling through to the drain wait"
8997        );
8998        return;
8999    };
9000    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9001        Ok(()) => debug!(
9002            module_id,
9003            "sent SIGTERM to a module nothing else asked to stop"
9004        ),
9005        Err(err) => debug!(
9006            module_id,
9007            error = %err,
9008            "SIGTERM to module failed; the drain wait and kill still apply"
9009        ),
9010    }
9011}
9012
9013/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9014/// Windows does offer need cooperation this supervisor cannot assume: a console
9015/// control event requires sharing a console with the child, and `WM_CLOSE`
9016/// requires the child to pump a message loop. A supervised server process does
9017/// neither, so there is nothing to send and teardown is the wait followed by the
9018/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9019/// the thing `protocol: "none"` exists to avoid.
9020#[cfg(not(unix))]
9021fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9022    debug!(
9023        module_id,
9024        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9025    );
9026}
9027
9028fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9029    match final_state {
9030        ModuleState::Stopped => TerminalDisposition::Stopped,
9031        ModuleState::Disabled => TerminalDisposition::Disabled,
9032        ModuleState::Restarting => TerminalDisposition::Restarting,
9033        ModuleState::Failed => TerminalDisposition::Failed,
9034        ModuleState::Starting
9035        | ModuleState::Running
9036        | ModuleState::Unresponsive
9037        | ModuleState::Draining => {
9038            unreachable!("terminal exits only finish in terminal or restarting states")
9039        }
9040    }
9041}
9042
9043/// Release a reaped child's registration before allowing another spawn.
9044///
9045/// EOF is not a process-lifetime signal: an inherited socket can stay open
9046/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9047/// reading EOF. After the normal release grace, request connection close (which
9048/// cancels both reads and dispatch), then allow one more release grace for the
9049/// connection guard's forwarding cleanup. Never evict a different connection.
9050async fn release_dead_registration(
9051    registry: &Registry,
9052    forwarding: Option<&ForwardingTable>,
9053    snapshot: &SharedSnapshot,
9054    module_id: &str,
9055) -> Result<(), SuperviseError> {
9056    let result = async {
9057        let registration = registry
9058            .get_module(module_id)
9059            .map_err(SuperviseError::Registry)?;
9060        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9061            Ok(()) => return Ok(()),
9062            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9063            Err(err) => return Err(err),
9064        }
9065        let pid = lock_snapshot(snapshot)?.reaped_pid;
9066        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9067            warn!(
9068                module_id,
9069                pid,
9070                connection_id = registration.connection_id.get(),
9071                "reaped module registration outlived release grace; closing dead connection"
9072            );
9073            forwarding.request_connection_close(
9074                registration.connection_id,
9075                CloseReason::new(
9076                    "supervised_process_reaped",
9077                    format!("module '{module_id}' pid {pid} exited"),
9078                ),
9079            );
9080            wait_for_slot_registration_release(
9081                registry,
9082                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9083                REGISTRY_RELEASE_TIMEOUT,
9084            )
9085            .await?;
9086        }
9087        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9088    }
9089    .await;
9090    if let Err(err) = &result {
9091        fail_snapshot(snapshot, Some(module_id), None);
9092        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9093    }
9094    result
9095}
9096
9097/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9098/// plain stop or restart waits for before it spawns a replacement.
9099async fn wait_for_registration_release(
9100    registry: &Registry,
9101    module_id: &str,
9102    wait: Duration,
9103) -> Result<(), SuperviseError> {
9104    wait_for_slot_registration_release(
9105        registry,
9106        crate::registry::RegistrationSlot::Active(module_id),
9107        wait,
9108    )
9109    .await
9110}
9111
9112/// Wait for the registration in `slot` to go away.
9113///
9114/// Keyed on the slot rather than the bare module id because a successful swap
9115/// never empties the id's active slot (the promoted candidate is in it), so an
9116/// id-keyed wait for the incumbent's release would always time out. Draining a
9117/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9118/// incumbent's connection instead.
9119async fn wait_for_slot_registration_release(
9120    registry: &Registry,
9121    slot: crate::registry::RegistrationSlot<'_>,
9122    wait: Duration,
9123) -> Result<(), SuperviseError> {
9124    let deadline = Instant::now() + wait;
9125    let mut release_events = registration_release_events().subscribe();
9126    let still_active = |registration: &crate::registry::ModuleRegistration| {
9127        SuperviseError::RegistrationStillActive {
9128            module_id: registration.manifest.module_id.clone(),
9129            waited: wait,
9130        }
9131    };
9132    loop {
9133        let _observed_generation = *release_events.borrow_and_update();
9134        let Some(registration) = registry
9135            .registration(slot)
9136            .map_err(SuperviseError::Registry)?
9137        else {
9138            return Ok(());
9139        };
9140
9141        let now = Instant::now();
9142        if now >= deadline {
9143            return Err(still_active(&registration));
9144        }
9145
9146        let remaining = deadline.saturating_duration_since(now);
9147        match timeout(remaining, release_events.changed()).await {
9148            Ok(Ok(())) | Ok(Err(_)) => {}
9149            Err(_) => return Err(still_active(&registration)),
9150        }
9151    }
9152}
9153
9154#[cfg(test)]
9155mod slot_registration_wait_tests {
9156    use super::*;
9157    use crate::registry::{ConnectionId, RegistrationSlot};
9158    use subc_protocol::manifest::ModuleManifest;
9159
9160    #[tokio::test]
9161    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9162        let registry = Arc::new(Registry::default());
9163        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9164        let runtime = supervisor.runtime_config();
9165        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9166        let spec = ModuleSpec {
9167            module_id: "enable-stale-registration".to_string(),
9168            program: PathBuf::from("/missing/enable-retry-test"),
9169            args: Vec::new(),
9170            env: Vec::new(),
9171            reserved: false,
9172            reserved_prefixes: Vec::new(),
9173            protocol: ModuleProtocol::Subc,
9174            overlap: Default::default(),
9175        };
9176        let connection = ConnectionId::new(90);
9177        registry
9178            .register_with_control_ops(
9179                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9180                1,
9181                connection,
9182                Vec::new(),
9183            )
9184            .unwrap();
9185        let mut child = None;
9186        let err = set_child_enabled(
9187            &spec,
9188            &runtime,
9189            &registry,
9190            &supervisor.process_liveness,
9191            &snapshot,
9192            &mut child,
9193            true,
9194        )
9195        .await
9196        .unwrap_err();
9197        assert!(matches!(
9198            err,
9199            SuperviseError::RegistrationStillActive { .. }
9200        ));
9201        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9202        assert!(child.is_none());
9203        registry.deregister_connection(connection).unwrap();
9204        let err = set_child_enabled(
9205            &spec,
9206            &runtime,
9207            &registry,
9208            &supervisor.process_liveness,
9209            &snapshot,
9210            &mut child,
9211            true,
9212        )
9213        .await
9214        .unwrap_err();
9215        assert!(
9216            matches!(err, SuperviseError::Spawn { .. }),
9217            "second enable must attempt a spawn: {err}"
9218        );
9219        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9220    }
9221
9222    const INCUMBENT: u64 = 1;
9223    const CANDIDATE: u64 = 2;
9224
9225    fn swapped_registry() -> Arc<Registry> {
9226        let registry = Arc::new(Registry::default());
9227        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9228        registry
9229            .register_with_control_ops(
9230                manifest.clone(),
9231                1,
9232                ConnectionId::new(INCUMBENT),
9233                Vec::new(),
9234            )
9235            .unwrap();
9236        registry
9237            .register_candidate_with_control_ops(
9238                manifest,
9239                1,
9240                ConnectionId::new(CANDIDATE),
9241                Vec::new(),
9242            )
9243            .unwrap();
9244        registry
9245    }
9246
9247    /// After a promotion the id's active slot is held by the new process, so an
9248    /// id-keyed wait for the incumbent's release can never succeed; the
9249    /// connection-keyed wait completes as soon as the incumbent deregisters.
9250    #[tokio::test]
9251    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9252        let registry = swapped_registry();
9253        registry.promote_candidate("m").unwrap().unwrap();
9254
9255        assert!(matches!(
9256            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9257            Err(SuperviseError::RegistrationStillActive { .. })
9258        ));
9259
9260        // Still held while the incumbent's connection has not deregistered.
9261        assert!(matches!(
9262            wait_for_slot_registration_release(
9263                &registry,
9264                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9265                Duration::from_millis(50),
9266            )
9267            .await,
9268            Err(SuperviseError::RegistrationStillActive { .. })
9269        ));
9270
9271        let releaser = Arc::clone(&registry);
9272        let release = tokio::spawn(async move {
9273            sleep(Duration::from_millis(20)).await;
9274            releaser
9275                .deregister_connection(ConnectionId::new(INCUMBENT))
9276                .unwrap();
9277            notify_registration_release();
9278        });
9279        wait_for_slot_registration_release(
9280            &registry,
9281            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9282            Duration::from_secs(5),
9283        )
9284        .await
9285        .expect("the incumbent's own registration is released");
9286        release.await.unwrap();
9287        assert!(registry.get_module("m").unwrap().is_some());
9288    }
9289
9290    /// The candidate slot is waited on separately from the active slot: the
9291    /// incumbent's registration neither holds up nor stands in for it.
9292    #[tokio::test]
9293    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9294        let registry = swapped_registry();
9295        assert!(matches!(
9296            wait_for_slot_registration_release(
9297                &registry,
9298                RegistrationSlot::Candidate("m"),
9299                Duration::from_millis(50),
9300            )
9301            .await,
9302            Err(SuperviseError::RegistrationStillActive { .. })
9303        ));
9304        registry
9305            .deregister_connection(ConnectionId::new(CANDIDATE))
9306            .unwrap();
9307        wait_for_slot_registration_release(
9308            &registry,
9309            RegistrationSlot::Candidate("m"),
9310            Duration::from_millis(50),
9311        )
9312        .await
9313        .expect("a candidate slot with no candidate is released");
9314        assert!(registry
9315            .registration(RegistrationSlot::Active("m"))
9316            .unwrap()
9317            .is_some());
9318    }
9319}
9320
9321fn classify_exit(status: &ExitStatus) -> ExitReport {
9322    ExitReport {
9323        kind: if status.success() {
9324            ExitKind::Clean
9325        } else {
9326            ExitKind::Crash
9327        },
9328        code: status.code(),
9329        signal: exit_signal(status),
9330        at_ms: unix_ms_now(),
9331    }
9332}
9333
9334/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9335/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9336/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9337/// disposition still must be `Failed` so the terminal ring is not silently missing
9338/// an entry, matching what `fail_snapshot` records for this same arm.
9339fn wait_error_exit_report() -> ExitReport {
9340    ExitReport {
9341        kind: ExitKind::Crash,
9342        code: None,
9343        signal: None,
9344        at_ms: unix_ms_now(),
9345    }
9346}
9347
9348#[cfg(unix)]
9349fn exit_signal(status: &ExitStatus) -> Option<i32> {
9350    use std::os::unix::process::ExitStatusExt;
9351
9352    status.signal()
9353}
9354
9355#[cfg(not(unix))]
9356fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9357    None
9358}
9359
9360/// Give an operator-touched module its full crash budget back.
9361///
9362/// Named for the counter it used to zero; it now empties the in-window ring,
9363/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9364/// ledger of what happened survives every operator action.
9365fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9366    update_snapshot(snapshot, Some(module_id), |state| {
9367        state.clear_crash_restarts();
9368    })
9369}
9370
9371fn set_running(
9372    snapshot: &SharedSnapshot,
9373    child: &SupervisedChild,
9374    module_id: &str,
9375    spawn_events: &SpawnEventFeed,
9376) -> Result<(), SuperviseError> {
9377    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9378        module_id: Some(module_id.to_string()),
9379    })?;
9380    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9381    if std::mem::take(&mut state.coalesced_restart_pending) {
9382        let generation = state.spawn_generation;
9383        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9384    }
9385    state.drain_disposition_detail = None;
9386    state.spawn_failure = None;
9387    // Every caller of this is a plain spawn, which always uses the primary key;
9388    // a promoted swap candidate sets the flag itself after this returns.
9389    state.in_alternate_slot = false;
9390    state.configuration_updated_since_spawn = false;
9391    state.spawned_protocol = Some(child.protocol);
9392    state.state = ModuleState::Running;
9393    state.enabled = true;
9394    state.process_alive = true;
9395    state.pid = child.id();
9396    state.spawned_at_ms = Some(child.spawned_at_ms);
9397    state.spawned_from = Some(child.spawned_from.clone());
9398    state.spawned_file_identity = child.spawned_file_identity;
9399    state.process_start_time = child.process_start_time;
9400    Ok(())
9401}
9402
9403fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9404    state.process_alive = false;
9405    state.spawned_protocol = None;
9406    state.pid = None;
9407    state.spawned_at_ms = None;
9408    state.spawned_from = None;
9409    state.spawned_file_identity = None;
9410    state.process_start_time = None;
9411    state.deliberate_severance = None;
9412}
9413
9414#[cfg(test)]
9415fn record_deliberate_severance(
9416    snapshot: &SharedSnapshot,
9417    identity: ProcessIdentity,
9418) -> Result<(), SuperviseError> {
9419    update_snapshot(snapshot, None, |state| {
9420        state.deliberate_severance = Some(identity);
9421    })
9422}
9423
9424fn apply_deliberate_severance_marker(
9425    snapshot: &SharedSnapshot,
9426    exited_identity: Option<ProcessIdentity>,
9427    mut exit_report: ExitReport,
9428) -> ExitReport {
9429    let marker = lock_snapshot(snapshot)
9430        .ok()
9431        .and_then(|mut state| state.deliberate_severance.take());
9432    if marker.is_some() && marker == exited_identity {
9433        exit_report.kind = ExitKind::DeliberateSeverance;
9434    }
9435    exit_report
9436}
9437
9438fn classify_reaped_child_exit(
9439    snapshot: &SharedSnapshot,
9440    child: &SupervisedChild,
9441    status: &ExitStatus,
9442) -> ExitReport {
9443    let _ = update_snapshot(snapshot, None, |state| {
9444        state.reaped_pid = Some(child.pid);
9445        state.spawn_failure = child.spawn_failure.clone();
9446    });
9447    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9448}
9449
9450fn fail_snapshot(
9451    snapshot: &SharedSnapshot,
9452    module_id: Option<&str>,
9453    last_exit: Option<ExitReport>,
9454) {
9455    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9456        state.state = ModuleState::Failed;
9457        clear_current_process_facts(state);
9458        if let Some(last_exit) = last_exit {
9459            state.last_exit = Some(last_exit);
9460        }
9461    }) {
9462        error!(error = %err, "failed to mark supervisor state failed");
9463    }
9464}
9465
9466fn update_snapshot(
9467    snapshot: &SharedSnapshot,
9468    module_id: Option<&str>,
9469    update: impl FnOnce(&mut SupervisorSnapshot),
9470) -> Result<(), SuperviseError> {
9471    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9472        module_id: module_id.map(ToOwned::to_owned),
9473    })?;
9474    update(&mut state);
9475    Ok(())
9476}
9477
9478const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9479
9480fn lock_snapshot_for_control<'a>(
9481    snapshot: &'a SharedSnapshot,
9482    module_id: &str,
9483    caller: &'static str,
9484) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9485    let started_at = Instant::now();
9486    let guard = lock_snapshot(snapshot)?;
9487    let waited = started_at.elapsed();
9488    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9489        warn!(
9490            module_id = %module_id,
9491            waited_ms = waited.as_millis() as u64,
9492            caller = %caller,
9493            "slow snapshot lock"
9494        );
9495    }
9496    Ok(guard)
9497}
9498
9499fn lock_snapshot(
9500    snapshot: &SharedSnapshot,
9501) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9502    snapshot
9503        .lock()
9504        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9505}
9506
9507#[cfg(test)]
9508mod terminal_history_tests {
9509    use std::{
9510        path::PathBuf,
9511        sync::Arc,
9512        time::{Duration, Instant},
9513    };
9514
9515    use tokio::time::sleep;
9516
9517    use super::{
9518        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9519        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9520        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9521        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9522        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9523        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
9524        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
9525    };
9526    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9527    // use for their own wall-clock deadlines: crash-restart instants must be on
9528    // the same clock the production code stamps them with, which is tokio's (and
9529    // is what `start_paused` tests can move).
9530    use super::Instant as ClockInstant;
9531    use crate::{
9532        registry::Registry,
9533        terminal_ring::{TerminalRing, TerminalRingConfig},
9534    };
9535    use std::sync::Mutex;
9536    use subc_control::TerminalDisposition;
9537
9538    /// See the twin in `control.rs` for why this derives the path from
9539    /// `current_exe()` and why the existence check is here: `--lib` alone does
9540    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9541    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9542    pub(super) fn fake_aft_stub_path() -> PathBuf {
9543        let mut path = std::env::current_exe().expect("current_exe available in tests");
9544        path.pop();
9545        path.pop();
9546        path.push(if cfg!(windows) {
9547            "fake-aft-stub.exe"
9548        } else {
9549            "fake-aft-stub"
9550        });
9551        assert!(
9552            path.exists(),
9553            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9554             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9555            path.display()
9556        );
9557        path
9558    }
9559
9560    #[test]
9561    fn reserved_never_spawned_refuses_every_hello() {
9562        // The canary hole: a reserved id whose module has never spawned had NO
9563        // gate entry and admitted anyone -- the reservation protected the nonce
9564        // holder, not the NAME. Now the entry is present with no legitimate
9565        // holder and refuses all comers.
9566        let supervisor = SupervisorHandle::default();
9567        supervisor.apply_identity_configuration(&ModuleSpec {
9568            module_id: "never-spawned".to_string(),
9569            program: PathBuf::from("/usr/bin/false"),
9570            args: Vec::new(),
9571            env: Vec::new(),
9572            reserved: true,
9573            reserved_prefixes: Vec::new(),
9574            protocol: ModuleProtocol::Subc,
9575            overlap: Default::default(),
9576        });
9577        assert!(
9578            supervisor
9579                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9580                .is_some(),
9581            "forged nonce must refuse on a reserved never-spawned id"
9582        );
9583        assert!(
9584            supervisor
9585                .reserved_hello_rejection("never-spawned", None)
9586                .is_some(),
9587            "absent nonce must refuse on a reserved never-spawned id"
9588        );
9589        // And a real spawn nonce minted later admits exactly that nonce.
9590        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9591        supervisor.apply_identity_configuration(&ModuleSpec {
9592            module_id: "never-spawned".to_string(),
9593            program: PathBuf::from("/usr/bin/false"),
9594            args: Vec::new(),
9595            env: Vec::new(),
9596            reserved: true,
9597            reserved_prefixes: Vec::new(),
9598            protocol: ModuleProtocol::Subc,
9599            overlap: Default::default(),
9600        });
9601        assert!(supervisor
9602            .reserved_hello_rejection("never-spawned", Some("minted"))
9603            .is_none());
9604        assert!(supervisor
9605            .reserved_hello_rejection("never-spawned", Some("forged"))
9606            .is_some());
9607    }
9608
9609    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9610    /// happened, which is what "spent budget" looks like to every reader.
9611    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9612        let now = ClockInstant::now();
9613        for _ in 0..count {
9614            state.crash_restarts.push_back(now);
9615        }
9616    }
9617
9618    /// Age the oldest recorded restart out of `window`, standing in for the hours
9619    /// that would otherwise have to pass. Injecting the instant is the point: a
9620    /// test that slept a real window would take ten minutes and still prove less.
9621    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9622        let aged = state
9623            .crash_restarts
9624            .front()
9625            .expect("a crash restart must be recorded before it can be aged")
9626            .checked_sub(window + Duration::from_secs(1))
9627            .expect("the test clock is far enough from its origin to age an instant");
9628        state.crash_restarts[0] = aged;
9629    }
9630
9631    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9632        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9633        seed_crash_restarts(&mut state, count);
9634        state
9635    }
9636
9637    #[test]
9638    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9639        let policy = RestartPolicy::new(3, Duration::ZERO);
9640        let now = ClockInstant::now();
9641        assert!(daemon_will_restart(
9642            &mut snapshot_with_restarts(true, 2),
9643            &policy,
9644            now
9645        ));
9646        assert!(!daemon_will_restart(
9647            &mut snapshot_with_restarts(true, 3),
9648            &policy,
9649            now
9650        ));
9651        assert!(!daemon_will_restart(
9652            &mut snapshot_with_restarts(false, 0),
9653            &policy,
9654            now
9655        ));
9656    }
9657
9658    #[test]
9659    fn crash_restart_backoff_escalates_with_in_window_count() {
9660        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9661            .with_max_backoff(Duration::from_secs(30));
9662        let now = ClockInstant::now();
9663        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9664        let schedules = (0..4)
9665            .map(|_| {
9666                state
9667                    .next_crash_restart(&policy, now)
9668                    .expect("the test policy allows four crash restarts")
9669            })
9670            .collect::<Vec<_>>();
9671
9672        assert_eq!(
9673            schedules
9674                .iter()
9675                .map(|schedule| schedule.restart_in_window)
9676                .collect::<Vec<_>>(),
9677            vec![0, 1, 2, 3]
9678        );
9679        assert_eq!(
9680            schedules
9681                .iter()
9682                .map(|schedule| schedule.delay)
9683                .collect::<Vec<_>>(),
9684            vec![
9685                Duration::from_millis(100),
9686                Duration::from_secs(1),
9687                Duration::from_secs(10),
9688                Duration::from_secs(30),
9689            ]
9690        );
9691    }
9692
9693    #[test]
9694    fn crash_restart_backoff_resets_after_ring_clear() {
9695        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9696        let now = ClockInstant::now();
9697        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9698        assert_eq!(
9699            state.next_crash_restart(&policy, now).unwrap().delay,
9700            Duration::from_millis(100)
9701        );
9702        assert_eq!(
9703            state.next_crash_restart(&policy, now).unwrap().delay,
9704            Duration::from_secs(1)
9705        );
9706
9707        state.clear_crash_restarts();
9708        let schedule = state
9709            .next_crash_restart(&policy, now)
9710            .expect("a cleared ring must allow another restart");
9711        assert_eq!(schedule.restart_in_window, 0);
9712        assert_eq!(schedule.delay, Duration::from_millis(100));
9713    }
9714
9715    #[test]
9716    fn crash_restart_backoff_ignores_aged_restarts() {
9717        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9718        let now = ClockInstant::now();
9719        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9720        state
9721            .next_crash_restart(&policy, now)
9722            .expect("the first restart is allowed");
9723        state
9724            .next_crash_restart(&policy, now)
9725            .expect("the second restart is allowed");
9726        state.crash_restarts[0] = now
9727            .checked_sub(policy.window + Duration::from_secs(1))
9728            .expect("the fake clock can age a restart past the window");
9729
9730        let schedule = state
9731            .next_crash_restart(&policy, now)
9732            .expect("an aged restart must release its slot");
9733        assert_eq!(schedule.restart_in_window, 1);
9734        assert_eq!(schedule.delay, Duration::from_secs(1));
9735        assert_eq!(state.crash_restarts.len(), 2);
9736    }
9737
9738    /// The budget is a rate: the same three spent restarts refuse a respawn
9739    /// while they are recent and allow one once they have aged past the window.
9740    /// Nothing about the module changed in between, which is the whole point.
9741    #[test]
9742    fn a_budget_spent_before_the_window_no_longer_refuses() {
9743        let policy = RestartPolicy::new(3, Duration::ZERO);
9744        let mut state = snapshot_with_restarts(true, 3);
9745        let now = ClockInstant::now();
9746        assert!(!daemon_will_restart(&mut state, &policy, now));
9747
9748        assert!(daemon_will_restart(
9749            &mut state,
9750            &policy,
9751            now + policy.window + Duration::from_secs(1)
9752        ));
9753        assert!(
9754            state.crash_restarts.is_empty(),
9755            "reading the budget must drop the instants that left the window"
9756        );
9757    }
9758
9759    fn module_with_recovery_snapshot(
9760        state: ModuleState,
9761        enabled: bool,
9762        restart_count: u32,
9763    ) -> SupervisedModule {
9764        let registry = Arc::new(Registry::default());
9765        let supervisor =
9766            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9767        let module = supervisor
9768            .spawn(ModuleSpec {
9769                module_id: "recovery-snapshot".to_string(),
9770                program: fake_aft_stub_path(),
9771                args: Vec::new(),
9772                env: Vec::new(),
9773                reserved: false,
9774                reserved_prefixes: Vec::new(),
9775                protocol: ModuleProtocol::Subc,
9776                overlap: Default::default(),
9777            })
9778            .unwrap();
9779        update_snapshot(
9780            &module.inner.snapshot,
9781            Some("recovery-snapshot"),
9782            |snapshot| {
9783                snapshot.state = state;
9784                snapshot.enabled = enabled;
9785                seed_crash_restarts(snapshot, restart_count);
9786            },
9787        )
9788        .unwrap();
9789        module
9790    }
9791
9792    #[cfg(target_os = "linux")]
9793    #[tokio::test]
9794    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9795        let supervisor =
9796            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
9797                .with_cgroup_placement(None);
9798        let result = supervisor.spawn(ModuleSpec {
9799            module_id: "no-cgroup-placement".to_string(),
9800            program: fake_aft_stub_path(),
9801            args: Vec::new(),
9802            env: Vec::new(),
9803            reserved: false,
9804            reserved_prefixes: Vec::new(),
9805            protocol: ModuleProtocol::Subc,
9806            overlap: Default::default(),
9807        });
9808
9809        assert!(
9810            result.is_ok(),
9811            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9812        );
9813    }
9814
9815    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9816    async fn undecided_snapshot_uses_shared_restart_predicate() {
9817        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9818            .will_recover_after_connection_loss()
9819            .unwrap());
9820        assert!(
9821            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9822                .will_recover_after_connection_loss()
9823                .unwrap()
9824        );
9825    }
9826
9827    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9828    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9829        assert!(
9830            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9831                .will_recover_after_connection_loss()
9832                .unwrap()
9833        );
9834    }
9835
9836    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9837    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9838        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9839            .will_recover_after_connection_loss()
9840            .unwrap());
9841        assert!(
9842            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9843                .will_recover_after_connection_loss()
9844                .unwrap()
9845        );
9846    }
9847
9848    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9849    async fn warming_snapshot_is_limited_to_startup_phases() {
9850        for state in [
9851            ModuleState::Starting,
9852            ModuleState::Running,
9853            ModuleState::Restarting,
9854        ] {
9855            assert!(
9856                module_with_recovery_snapshot(state, true, 0)
9857                    .is_warming()
9858                    .unwrap(),
9859                "{state:?} should be warming"
9860            );
9861        }
9862        for state in [
9863            ModuleState::Unresponsive,
9864            ModuleState::Draining,
9865            ModuleState::Stopped,
9866            ModuleState::Failed,
9867            ModuleState::Disabled,
9868        ] {
9869            assert!(
9870                !module_with_recovery_snapshot(state, true, 0)
9871                    .is_warming()
9872                    .unwrap(),
9873                "{state:?} should not be warming"
9874            );
9875        }
9876    }
9877
9878    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9879    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
9880        let registry = Arc::new(Registry::default());
9881        let supervisor =
9882            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
9883        let module = supervisor
9884            .spawn(ModuleSpec {
9885                module_id: "terminal-history".to_string(),
9886                program: fake_aft_stub_path(),
9887                args: Vec::new(),
9888                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9889                reserved: false,
9890                reserved_prefixes: Vec::new(),
9891                protocol: ModuleProtocol::Subc,
9892                overlap: Default::default(),
9893            })
9894            .unwrap();
9895
9896        let deadline = Instant::now() + Duration::from_secs(5);
9897        loop {
9898            let history = module.terminal_history();
9899            if history.entries.len() == 2 {
9900                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
9901                assert_eq!(history.dropped, 0);
9902                assert_eq!(
9903                    history
9904                        .entries
9905                        .iter()
9906                        .map(|entry| entry.exit_code)
9907                        .collect::<Vec<_>>(),
9908                    vec![Some(23), Some(23)]
9909                );
9910                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
9911                return;
9912            }
9913            assert!(
9914                Instant::now() < deadline,
9915                "module did not retain two terminal exits: {history:?}"
9916            );
9917            sleep(Duration::from_millis(10)).await;
9918        }
9919    }
9920
9921    /// A disable issued while a crash respawn is still backing off must preempt
9922    /// that respawn: the operator's stop wins, the disable must not queue behind
9923    /// the backoff, and the module must never come back up afterwards.
9924    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9925    async fn disable_during_crash_backoff_cancels_pending_respawn() {
9926        let backoff = Duration::from_secs(2);
9927        let supervisor = Supervisor::new_for_test(
9928            Arc::new(Registry::default()),
9929            RestartPolicy::new(10, backoff),
9930        );
9931        let module = supervisor
9932            .spawn(ModuleSpec {
9933                module_id: "disable-during-backoff".to_string(),
9934                program: fake_aft_stub_path(),
9935                args: Vec::new(),
9936                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9937                reserved: false,
9938                reserved_prefixes: Vec::new(),
9939                protocol: ModuleProtocol::Subc,
9940                overlap: Default::default(),
9941            })
9942            .unwrap();
9943
9944        // Wait for the first crash to put the module into its backoff window.
9945        let deadline = Instant::now() + Duration::from_secs(5);
9946        loop {
9947            if module.status().unwrap().state == ModuleState::Restarting {
9948                break;
9949            }
9950            assert!(
9951                Instant::now() < deadline,
9952                "module never entered the crash backoff"
9953            );
9954            sleep(Duration::from_millis(10)).await;
9955        }
9956
9957        let started = Instant::now();
9958        module.set_enabled(false).await.unwrap();
9959        let waited = started.elapsed();
9960
9961        assert!(
9962            waited < backoff / 2,
9963            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
9964        );
9965        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
9966
9967        // Outlast the backoff: the respawn it was counting down to must never run.
9968        sleep(backoff + Duration::from_millis(500)).await;
9969        let status = module.status().unwrap();
9970        assert_eq!(status.state, ModuleState::Disabled);
9971        assert_eq!(
9972            status.spawn_generation, 1,
9973            "module respawned after the operator disabled it"
9974        );
9975    }
9976
9977    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
9978    /// the shape of nats-server, the program this rule exists for.
9979    #[cfg(unix)]
9980    fn protocol_none_sigterm_exits_clean_spec(
9981        module_id: &str,
9982        dir: &std::path::Path,
9983    ) -> (ModuleSpec, PathBuf, PathBuf) {
9984        let ready = dir.join("ready");
9985        let marker = dir.join("sigterm");
9986        let spec = ModuleSpec {
9987            module_id: module_id.to_string(),
9988            program: fake_aft_stub_path(),
9989            args: Vec::new(),
9990            env: vec![
9991                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
9992                (
9993                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
9994                    marker.display().to_string(),
9995                ),
9996                (
9997                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
9998                    ready.display().to_string(),
9999                ),
10000            ],
10001            reserved: false,
10002            reserved_prefixes: Vec::new(),
10003            protocol: ModuleProtocol::None,
10004            overlap: Default::default(),
10005        };
10006        (spec, ready, marker)
10007    }
10008
10009    /// Wait for a file the child writes, so a signal is never sent before the
10010    /// child's SIGTERM handler is installed (the default disposition would
10011    /// kill it by signal and the exit would not be clean).
10012    #[cfg(unix)]
10013    async fn wait_for_file(path: &std::path::Path) {
10014        let deadline = Instant::now() + Duration::from_secs(10);
10015        while !path.exists() {
10016            assert!(
10017                Instant::now() < deadline,
10018                "{} never appeared",
10019                path.display()
10020            );
10021            sleep(Duration::from_millis(10)).await;
10022        }
10023    }
10024
10025    /// A protocol-none module that exits 0 because something OUTSIDE the
10026    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10027    /// the crash-path disposition rather than `stopped`.
10028    #[cfg(unix)]
10029    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10030    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10031        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10032        let (spec, ready, marker) =
10033            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10034        let supervisor = Supervisor::new_for_test(
10035            Arc::new(Registry::default()),
10036            RestartPolicy::new(3, Duration::ZERO),
10037        );
10038        let module = supervisor.spawn(spec).unwrap();
10039        wait_for_file(&ready).await;
10040        let first_pid = module
10041            .status()
10042            .unwrap()
10043            .pid
10044            .expect("a running module reports its pid");
10045
10046        rustix::process::kill_process(
10047            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10048            rustix::process::Signal::TERM,
10049        )
10050        .unwrap();
10051
10052        let deadline = Instant::now() + Duration::from_secs(10);
10053        let respawned = loop {
10054            let status = module.status().unwrap();
10055            if status.state == ModuleState::Running
10056                && status.pid.is_some_and(|pid| pid != first_pid)
10057            {
10058                break status;
10059            }
10060            assert!(
10061                Instant::now() < deadline,
10062                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10063            );
10064            sleep(Duration::from_millis(10)).await;
10065        };
10066        assert_eq!(respawned.spawn_generation, 2);
10067        assert!(
10068            marker.exists(),
10069            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10070        );
10071
10072        let history = module.terminal_history();
10073        assert_eq!(history.entries.len(), 1, "{history:?}");
10074        let entry = &history.entries[0];
10075        assert_eq!(entry.exit_code, Some(0));
10076        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10077        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10078
10079        module.stop().await.unwrap();
10080    }
10081
10082    /// Repeated unrequested clean exits of a protocol-none module spend the
10083    /// restart budget exactly as crashes do, and the module ends `failed` with
10084    /// the budget named.
10085    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10086    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10087        let supervisor = Supervisor::new_for_test(
10088            Arc::new(Registry::default()),
10089            RestartPolicy::new(1, Duration::ZERO),
10090        );
10091        let module = supervisor
10092            .spawn(ModuleSpec {
10093                module_id: "none-clean-exit-budget".to_string(),
10094                program: fake_aft_stub_path(),
10095                args: Vec::new(),
10096                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10097                reserved: false,
10098                reserved_prefixes: Vec::new(),
10099                protocol: ModuleProtocol::None,
10100                overlap: Default::default(),
10101            })
10102            .unwrap();
10103
10104        let deadline = Instant::now() + Duration::from_secs(10);
10105        loop {
10106            let status = module.status().unwrap();
10107            if status.state == ModuleState::Failed {
10108                break;
10109            }
10110            assert!(
10111                Instant::now() < deadline,
10112                "module never exhausted its budget: {status:?} {:?}",
10113                module.terminal_history()
10114            );
10115            sleep(Duration::from_millis(10)).await;
10116        }
10117        let history = module.terminal_history();
10118        assert_eq!(
10119            history
10120                .entries
10121                .iter()
10122                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10123                .collect::<Vec<_>>(),
10124            vec![
10125                (Some(0), TerminalDisposition::Restarting),
10126                (Some(0), TerminalDisposition::Failed),
10127            ]
10128        );
10129        let detail = history.entries[1]
10130            .disposition_detail
10131            .as_deref()
10132            .expect("a budget failure names the budget");
10133        assert!(detail.contains("max_restarts=1"), "{detail}");
10134        assert_eq!(module.status().unwrap().spawn_generation, 2);
10135    }
10136
10137    /// A stop the supervisor itself requests still stops a protocol-none
10138    /// module, even though the child answers the SIGTERM with exit 0.
10139    #[cfg(unix)]
10140    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10141    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10142        for disable in [false, true] {
10143            let label = if disable {
10144                "none-requested-disable"
10145            } else {
10146                "none-requested-stop"
10147            };
10148            let dir = subc_test_support::TestTempDir::new(label);
10149            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10150            let supervisor = Supervisor::new_for_test(
10151                Arc::new(Registry::default()),
10152                RestartPolicy::new(3, Duration::ZERO),
10153            );
10154            let module = supervisor.spawn(spec).unwrap();
10155            wait_for_file(&ready).await;
10156
10157            if disable {
10158                module.set_enabled(false).await.unwrap();
10159            } else {
10160                module.stop().await.unwrap();
10161            }
10162            assert!(
10163                marker.exists(),
10164                "{label}: the child must have left through its SIGTERM handler with exit 0"
10165            );
10166
10167            // Long enough for a zero-backoff respawn to have happened if the
10168            // exit had been treated as a crash.
10169            sleep(Duration::from_millis(500)).await;
10170            let status = module.status().unwrap();
10171            let expected = if disable {
10172                ModuleState::Disabled
10173            } else {
10174                ModuleState::Stopped
10175            };
10176            assert_eq!(status.state, expected, "{label}");
10177            assert_eq!(
10178                status.spawn_generation, 1,
10179                "{label}: respawned after a requested stop"
10180            );
10181            let history = module.terminal_history();
10182            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10183            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10184            assert_ne!(
10185                history.entries[0].disposition,
10186                TerminalDisposition::Restarting,
10187                "{label}"
10188            );
10189        }
10190    }
10191
10192    /// A subc-wire module that exits 0 on its own is still a stop: the
10193    /// protocol-none rule must not reach it.
10194    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10195    async fn subc_wire_clean_exit_is_still_a_stop() {
10196        let supervisor = Supervisor::new_for_test(
10197            Arc::new(Registry::default()),
10198            RestartPolicy::new(3, Duration::ZERO),
10199        );
10200        let module = supervisor
10201            .spawn(ModuleSpec {
10202                module_id: "wire-clean-exit".to_string(),
10203                program: fake_aft_stub_path(),
10204                args: Vec::new(),
10205                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10206                reserved: false,
10207                reserved_prefixes: Vec::new(),
10208                protocol: ModuleProtocol::Subc,
10209                overlap: Default::default(),
10210            })
10211            .unwrap();
10212
10213        let deadline = Instant::now() + Duration::from_secs(10);
10214        while module.terminal_history().entries.is_empty() {
10215            assert!(Instant::now() < deadline, "module never exited");
10216            sleep(Duration::from_millis(10)).await;
10217        }
10218        // Long enough for a zero-backoff respawn to have happened.
10219        sleep(Duration::from_millis(500)).await;
10220        let status = module.status().unwrap();
10221        assert_eq!(status.state, ModuleState::Stopped);
10222        assert_eq!(status.spawn_generation, 1);
10223        let history = module.terminal_history();
10224        assert_eq!(history.entries.len(), 1, "{history:?}");
10225        assert_eq!(history.entries[0].exit_code, Some(0));
10226        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10227    }
10228
10229    #[cfg(unix)]
10230    #[tokio::test]
10231    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10232        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10233        let record = dir.join("live-children.json");
10234        let supervisor = Supervisor::new_for_test(
10235            Arc::new(Registry::default()),
10236            RestartPolicy::new(0, Duration::ZERO),
10237        );
10238        let mut runtime = supervisor.runtime_config();
10239        runtime.child_roster.record_to(record.clone());
10240        let gate = Arc::new(super::ReloadExitRecordGate::default());
10241        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10242        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10243        let spec = ModuleSpec {
10244            module_id: "reload-exit-roster".into(),
10245            program: fake_aft_stub_path(),
10246            args: Vec::new(),
10247            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10248            reserved: false,
10249            reserved_prefixes: Vec::new(),
10250            protocol: ModuleProtocol::Subc,
10251            overlap: Default::default(),
10252        };
10253        let mut child = None;
10254        let reload = super::finish_reload_child(
10255            &spec,
10256            &runtime,
10257            &supervisor.registry,
10258            &supervisor.process_liveness,
10259            &snapshot,
10260            &mut child,
10261        );
10262        tokio::pin!(reload);
10263        tokio::select! {
10264            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10265            _ = gate.reached.notified() => {}
10266        }
10267        assert!(runtime
10268            .terminal_ring
10269            .lock()
10270            .unwrap()
10271            .snapshot()
10272            .entries
10273            .is_empty());
10274        assert_eq!(
10275            crate::live_children::read_record(&record).unwrap().len(),
10276            1,
10277            "shutdown must still wait for the reaped child until its terminal record exists"
10278        );
10279        runtime.child_roster.close();
10280        gate.resume.notify_one();
10281        assert!(reload.await.is_err());
10282        assert!(crate::live_children::read_record(&record)
10283            .unwrap()
10284            .is_empty());
10285        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10286        assert_eq!(history.entries.len(), 1);
10287        assert_eq!(
10288            history.entries[0].disposition,
10289            TerminalDisposition::DaemonShutdown
10290        );
10291    }
10292
10293    /// Each restart-producing arm has its own state transition. Keeping their
10294    /// lifetime count assertions adjacent prevents a later new arm from silently
10295    /// spending budget without recording the historical restart.
10296    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10297    async fn every_restart_increment_path_advances_lifetime_count() {
10298        let supervisor = Supervisor::new_for_test(
10299            Arc::new(Registry::default()),
10300            RestartPolicy::new(1, Duration::ZERO),
10301        );
10302        let runtime = supervisor.runtime_config();
10303        let spec = ModuleSpec {
10304            module_id: "lifetime-increment-path".to_string(),
10305            program: PathBuf::from("/unused/lifetime-increment-path"),
10306            args: Vec::new(),
10307            env: Vec::new(),
10308            reserved: false,
10309            reserved_prefixes: Vec::new(),
10310            protocol: ModuleProtocol::Subc,
10311            overlap: Default::default(),
10312        };
10313
10314        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10315        assert!(matches!(
10316            on_child_exit(
10317                &spec,
10318                runtime.restart_policy,
10319                &supervisor.registry,
10320                &crash_snapshot,
10321                &runtime.terminal_ring,
10322                &runtime.spawn_events,
10323                &runtime.child_roster,
10324                ExitReport {
10325                    kind: ExitKind::Crash,
10326                    code: Some(1),
10327                    signal: None,
10328                    at_ms: 1,
10329                },
10330            )
10331            .await,
10332            NextAction::Restart { schedule: _ }
10333        ));
10334        let (crash_restarts, crash_lifetime) = {
10335            let state = lock_snapshot(&crash_snapshot).unwrap();
10336            (state.crash_restarts.len(), state.lifetime_restarts)
10337        };
10338        assert_eq!(crash_restarts, 1);
10339        assert_eq!(crash_lifetime, 1);
10340
10341        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10342        let mut health_child = None;
10343        assert!(matches!(
10344            health_restart_child(
10345                &spec,
10346                &runtime,
10347                &supervisor.registry,
10348                &supervisor.process_liveness,
10349                &health_snapshot,
10350                &mut health_child,
10351                SupervisorHealthStatus::Failing,
10352                None,
10353                2,
10354            )
10355            .await,
10356            Ok(())
10357        ));
10358        assert!(health_child.is_none());
10359        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10360        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10361        let (health_restarts, health_lifetime) = {
10362            let state = lock_snapshot(&health_snapshot).unwrap();
10363            (state.crash_restarts.len(), state.lifetime_restarts)
10364        };
10365        assert_eq!(health_restarts, 1);
10366        assert_eq!(health_lifetime, 1);
10367
10368        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10369        let mut reload_child = None;
10370        assert!(matches!(
10371            handle_reload_spawn_failure(
10372                &spec,
10373                &runtime,
10374                &supervisor.process_liveness,
10375                &reload_snapshot,
10376                &mut reload_child,
10377                "forced reload spawn failure".to_string(),
10378            )
10379            .await,
10380            Err(SuperviseError::ReloadFailed { .. })
10381        ));
10382        let (reload_restarts, reload_lifetime) = {
10383            let state = lock_snapshot(&reload_snapshot).unwrap();
10384            (state.crash_restarts.len(), state.lifetime_restarts)
10385        };
10386        assert_eq!(reload_restarts, 1);
10387        assert_eq!(reload_lifetime, 1);
10388    }
10389
10390    #[tokio::test]
10391    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10392        let supervisor = Supervisor::new_for_test(
10393            Arc::new(Registry::default()),
10394            RestartPolicy::new(3, Duration::ZERO),
10395        );
10396        let runtime = supervisor.runtime_config();
10397        let spec = ModuleSpec {
10398            module_id: "deliberately-severed".to_string(),
10399            program: PathBuf::from("/unused/deliberately-severed"),
10400            args: Vec::new(),
10401            env: Vec::new(),
10402            reserved: false,
10403            reserved_prefixes: Vec::new(),
10404            protocol: ModuleProtocol::Subc,
10405            overlap: Default::default(),
10406        };
10407        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10408        let process = ProcessIdentity {
10409            pid: 41,
10410            start_time: 101,
10411        };
10412        record_deliberate_severance(&snapshot, process).unwrap();
10413        let exit_report = apply_deliberate_severance_marker(
10414            &snapshot,
10415            Some(process),
10416            ExitReport {
10417                kind: ExitKind::Crash,
10418                code: Some(1),
10419                signal: None,
10420                at_ms: 1,
10421            },
10422        );
10423        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10424
10425        assert!(matches!(
10426            on_child_exit(
10427                &spec,
10428                runtime.restart_policy,
10429                &supervisor.registry,
10430                &snapshot,
10431                &runtime.terminal_ring,
10432                &runtime.spawn_events,
10433                &runtime.child_roster,
10434                exit_report,
10435            )
10436            .await,
10437            NextAction::Restart { schedule: _ }
10438        ));
10439        let state = lock_snapshot(&snapshot).unwrap();
10440        assert_eq!(state.lifetime_restarts, 1);
10441        assert_eq!(state.crash_restarts.len(), 0);
10442    }
10443
10444    #[tokio::test]
10445    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10446        let supervisor = Supervisor::new_for_test(
10447            Arc::new(Registry::default()),
10448            RestartPolicy::new(3, Duration::ZERO),
10449        );
10450        let runtime = supervisor.runtime_config();
10451        let spec = ModuleSpec {
10452            module_id: "genuine-crash".to_string(),
10453            program: PathBuf::from("/unused/genuine-crash"),
10454            args: Vec::new(),
10455            env: Vec::new(),
10456            reserved: false,
10457            reserved_prefixes: Vec::new(),
10458            protocol: ModuleProtocol::Subc,
10459            overlap: Default::default(),
10460        };
10461        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10462
10463        assert!(matches!(
10464            on_child_exit(
10465                &spec,
10466                runtime.restart_policy,
10467                &supervisor.registry,
10468                &snapshot,
10469                &runtime.terminal_ring,
10470                &runtime.spawn_events,
10471                &runtime.child_roster,
10472                ExitReport {
10473                    kind: ExitKind::Crash,
10474                    code: Some(1),
10475                    signal: None,
10476                    at_ms: 1,
10477                },
10478            )
10479            .await,
10480            NextAction::Restart { schedule: _ }
10481        ));
10482        let state = lock_snapshot(&snapshot).unwrap();
10483        assert_eq!(state.lifetime_restarts, 1);
10484        assert_eq!(state.crash_restarts.len(), 1);
10485    }
10486
10487    fn crash_exit_report(at_ms: u64) -> ExitReport {
10488        ExitReport {
10489            kind: ExitKind::Crash,
10490            code: Some(1),
10491            signal: None,
10492            at_ms,
10493        }
10494    }
10495
10496    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10497        ModuleSpec {
10498            module_id: module_id.to_string(),
10499            program: PathBuf::from("/unused").join(module_id),
10500            args: Vec::new(),
10501            env: Vec::new(),
10502            reserved: false,
10503            reserved_prefixes: Vec::new(),
10504            protocol: ModuleProtocol::Subc,
10505            overlap: Default::default(),
10506        }
10507    }
10508
10509    /// A real crash loop still stops. Three crashes with nothing aging out spend
10510    /// a budget of two and the third respawn is refused, and both surfaces an
10511    /// operator has -- the log line and the retained terminal record -- name the
10512    /// window rather than only the cap, because `max_restarts=2` alone is what
10513    /// this budget used to mean.
10514    #[tokio::test]
10515    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10516        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10517        let supervisor = Supervisor::new_for_test(
10518            Arc::new(Registry::default()),
10519            RestartPolicy::new(2, Duration::ZERO),
10520        );
10521        let runtime = supervisor.runtime_config();
10522        let spec = windowed_crash_spec("crash-loop-in-window");
10523        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10524
10525        for attempt in 1..=2 {
10526            assert!(
10527                matches!(
10528                    on_child_exit(
10529                        &spec,
10530                        runtime.restart_policy,
10531                        &supervisor.registry,
10532                        &snapshot,
10533                        &runtime.terminal_ring,
10534                        &runtime.spawn_events,
10535                        &runtime.child_roster,
10536                        crash_exit_report(attempt),
10537                    )
10538                    .await,
10539                    NextAction::Restart { schedule: _ }
10540                ),
10541                "crash {attempt} is inside the budget and must respawn"
10542            );
10543        }
10544
10545        assert!(matches!(
10546            on_child_exit(
10547                &spec,
10548                runtime.restart_policy,
10549                &supervisor.registry,
10550                &snapshot,
10551                &runtime.terminal_ring,
10552                &runtime.spawn_events,
10553                &runtime.child_roster,
10554                crash_exit_report(3),
10555            )
10556            .await,
10557            NextAction::Stop { .. }
10558        ));
10559
10560        {
10561            let state = lock_snapshot(&snapshot).unwrap();
10562            assert_eq!(state.state, ModuleState::Failed);
10563            assert_eq!(state.crash_restarts.len(), 2);
10564            assert_eq!(state.lifetime_restarts, 2);
10565        }
10566
10567        let history = runtime
10568            .terminal_ring
10569            .lock()
10570            .expect("terminal ring is not poisoned")
10571            .snapshot();
10572        let last = history
10573            .entries
10574            .last()
10575            .expect("the refused crash is retained");
10576        assert_eq!(last.disposition, TerminalDisposition::Failed);
10577        assert_eq!(
10578            last.disposition_detail.as_deref(),
10579            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10580        );
10581
10582        let captured = crate::router::test_log::captured_logs(&logs);
10583        assert!(
10584            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10585            "the stop must be logged with its window: {captured}"
10586        );
10587    }
10588
10589    /// The rate, stated as a test: three crashes where the first has aged past
10590    /// the window are two crashes as far as the budget is concerned, so the
10591    /// third respawn is allowed and the ring holds only the two recent ones.
10592    ///
10593    /// This is the case a lifetime counter got wrong -- and the case the daemon
10594    /// now hits routinely, since a module exits non-zero every time its
10595    /// connection to the daemon drops.
10596    #[tokio::test]
10597    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10598        let supervisor = Supervisor::new_for_test(
10599            Arc::new(Registry::default()),
10600            RestartPolicy::new(2, Duration::ZERO),
10601        );
10602        let runtime = supervisor.runtime_config();
10603        let spec = windowed_crash_spec("crash-across-windows");
10604        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10605
10606        for attempt in 1..=2 {
10607            assert!(matches!(
10608                on_child_exit(
10609                    &spec,
10610                    runtime.restart_policy,
10611                    &supervisor.registry,
10612                    &snapshot,
10613                    &runtime.terminal_ring,
10614                    &runtime.spawn_events,
10615                    &runtime.child_roster,
10616                    crash_exit_report(attempt),
10617                )
10618                .await,
10619                NextAction::Restart { schedule: _ }
10620            ));
10621        }
10622
10623        // The oldest crash moves out of the window; nothing else about the
10624        // module changes.
10625        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10626            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10627        })
10628        .unwrap();
10629
10630        assert!(
10631            matches!(
10632                on_child_exit(
10633                    &spec,
10634                    runtime.restart_policy,
10635                    &supervisor.registry,
10636                    &snapshot,
10637                    &runtime.terminal_ring,
10638                    &runtime.spawn_events,
10639                    &runtime.child_roster,
10640                    crash_exit_report(3),
10641                )
10642                .await,
10643                NextAction::Restart { schedule: _ }
10644            ),
10645            "a crash older than the window must not hold a budget slot"
10646        );
10647
10648        let state = lock_snapshot(&snapshot).unwrap();
10649        assert_eq!(state.state, ModuleState::Restarting);
10650        assert_eq!(
10651            state.crash_restarts.len(),
10652            2,
10653            "the aged instant is dropped and the new one takes its place"
10654        );
10655        assert_eq!(
10656            state.lifetime_restarts, 3,
10657            "the ledger counts every restart, including the ones the window forgot"
10658        );
10659    }
10660
10661    /// An operator restart hands the budget back whole, and the ledger keeps
10662    /// counting. Those are different questions -- "how close is this module to
10663    /// being stopped" and "how many times has it been replaced" -- and the
10664    /// operator action answers only the first.
10665    #[tokio::test]
10666    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10667        let supervisor = Supervisor::new_for_test(
10668            Arc::new(Registry::default()),
10669            RestartPolicy::new(2, Duration::ZERO),
10670        );
10671        let runtime = supervisor.runtime_config();
10672        let spec = windowed_crash_spec("operator-cleared-budget");
10673        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10674
10675        for attempt in 1..=2 {
10676            assert!(matches!(
10677                on_child_exit(
10678                    &spec,
10679                    runtime.restart_policy,
10680                    &supervisor.registry,
10681                    &snapshot,
10682                    &runtime.terminal_ring,
10683                    &runtime.spawn_events,
10684                    &runtime.child_roster,
10685                    crash_exit_report(attempt),
10686                )
10687                .await,
10688                NextAction::Restart { schedule: _ }
10689            ));
10690        }
10691
10692        reset_restart_count(&snapshot, &spec.module_id).unwrap();
10693        {
10694            let state = lock_snapshot(&snapshot).unwrap();
10695            assert!(
10696                state.crash_restarts.is_empty(),
10697                "an operator restart returns the full budget"
10698            );
10699            assert_eq!(
10700                state.lifetime_restarts, 2,
10701                "clearing the budget must not unmake the crashes"
10702            );
10703        }
10704
10705        assert!(
10706            matches!(
10707                on_child_exit(
10708                    &spec,
10709                    runtime.restart_policy,
10710                    &supervisor.registry,
10711                    &snapshot,
10712                    &runtime.terminal_ring,
10713                    &runtime.spawn_events,
10714                    &runtime.child_roster,
10715                    crash_exit_report(3),
10716                )
10717                .await,
10718                NextAction::Restart { schedule: _ }
10719            ),
10720            "the cleared budget must be spendable again"
10721        );
10722        let state = lock_snapshot(&snapshot).unwrap();
10723        assert_eq!(state.crash_restarts.len(), 1);
10724        assert_eq!(state.lifetime_restarts, 3);
10725    }
10726
10727    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10728    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
10729        let severed = ProcessIdentity {
10730            pid: 41,
10731            start_time: 101,
10732        };
10733        let successor = ProcessIdentity {
10734            pid: 41,
10735            start_time: 202,
10736        };
10737        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
10738        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
10739            state.pid = Some(successor.pid);
10740            state.process_start_time = Some(successor.start_time);
10741        })
10742        .unwrap();
10743        assert!(!module.record_deliberate_severance(severed).unwrap());
10744
10745        let exit_report = apply_deliberate_severance_marker(
10746            &module.inner.snapshot,
10747            Some(successor),
10748            ExitReport {
10749                kind: ExitKind::Crash,
10750                code: Some(1),
10751                signal: None,
10752                at_ms: 1,
10753            },
10754        );
10755
10756        assert_eq!(exit_report.kind, ExitKind::Crash);
10757    }
10758
10759    #[tokio::test]
10760    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
10761        let registry = Registry::default();
10762        let supervisor = Supervisor::new_for_test(
10763            Arc::new(Registry::default()),
10764            RestartPolicy::new(3, Duration::ZERO),
10765        );
10766        let runtime = supervisor.runtime_config();
10767        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10768        let spec = ModuleSpec {
10769            module_id: "drain-deliberate-severance".to_string(),
10770            program: fake_aft_stub_path(),
10771            args: Vec::new(),
10772            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10773            reserved: false,
10774            reserved_prefixes: Vec::new(),
10775            protocol: ModuleProtocol::Subc,
10776            overlap: Default::default(),
10777        };
10778        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10779        let process = ProcessIdentity {
10780            pid: 41,
10781            start_time: 101,
10782        };
10783        child.process_identity = Some(process);
10784        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10785            state.pid = Some(process.pid);
10786            state.process_start_time = Some(process.start_time);
10787        })
10788        .unwrap();
10789        record_deliberate_severance(&snapshot, process).unwrap();
10790
10791        drain_child_to_state(
10792            &spec.module_id,
10793            spec.protocol,
10794            // The child exits on its own; no signal may change the exit this
10795            // test classifies.
10796            StopNotice::SentOverConnection,
10797            &registry,
10798            None,
10799            &snapshot,
10800            &runtime.terminal_ring,
10801            &runtime.spawn_events,
10802            child,
10803            Duration::from_secs(1),
10804            ModuleState::Stopped,
10805            Some(false),
10806        )
10807        .await
10808        .unwrap();
10809
10810        let state = lock_snapshot(&snapshot).unwrap();
10811        assert_eq!(
10812            state.last_exit.as_ref().map(|exit| exit.kind),
10813            Some(ExitKind::DeliberateSeverance)
10814        );
10815        assert_eq!(state.lifetime_restarts, 1);
10816        assert_eq!(state.crash_restarts.len(), 0);
10817        drop(state);
10818        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10819        assert_eq!(
10820            history.entries[0].exit_kind,
10821            subc_control::TerminalExitKind::DeliberateSeverance
10822        );
10823    }
10824
10825    #[tokio::test]
10826    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
10827        let registry = Registry::default();
10828        let supervisor = Supervisor::new_for_test(
10829            Arc::new(Registry::default()),
10830            RestartPolicy::new(3, Duration::ZERO),
10831        );
10832        let runtime = supervisor.runtime_config();
10833        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10834        let spec = ModuleSpec {
10835            module_id: "ordinary-drain".to_string(),
10836            program: fake_aft_stub_path(),
10837            args: Vec::new(),
10838            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10839            reserved: false,
10840            reserved_prefixes: Vec::new(),
10841            protocol: ModuleProtocol::Subc,
10842            overlap: Default::default(),
10843        };
10844        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10845
10846        drain_child_to_state(
10847            &spec.module_id,
10848            spec.protocol,
10849            // The child exits on its own; no signal may change the exit this
10850            // test classifies.
10851            StopNotice::SentOverConnection,
10852            &registry,
10853            None,
10854            &snapshot,
10855            &runtime.terminal_ring,
10856            &runtime.spawn_events,
10857            child,
10858            Duration::from_secs(1),
10859            ModuleState::Stopped,
10860            Some(false),
10861        )
10862        .await
10863        .unwrap();
10864
10865        let state = lock_snapshot(&snapshot).unwrap();
10866        assert_eq!(
10867            state.last_exit.as_ref().map(|exit| exit.kind),
10868            Some(ExitKind::Crash)
10869        );
10870        assert_eq!(state.lifetime_restarts, 0);
10871        assert_eq!(state.crash_restarts.len(), 0);
10872    }
10873
10874    #[test]
10875    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
10876        // The server's generic fatal-routing branch only knows that the
10877        // connection failed; it does not know that the daemon deliberately
10878        // initiated a process-killing severance. Keep this seam explicit so a
10879        // future connection error path cannot silently reintroduce the stale
10880        // exemption that mislabels a later genuine crash.
10881        assert!(!include_str!("server.rs")
10882            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
10883    }
10884
10885    /// The `route.closed` `drained` value must be the quiescence wait's own
10886    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
10887    /// measurement at all and `false` is the one honest constant. This is the exact
10888    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
10889    /// on every return path, including the one that used to return early via `?`
10890    /// with `route.closing` already sent and no `route.closed` ever following.
10891    #[test]
10892    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
10893        assert!(drained_after_quiescence_wait(&Ok(true)));
10894        assert!(!drained_after_quiescence_wait(&Ok(false)));
10895        assert!(!drained_after_quiescence_wait(&Err(
10896            SuperviseError::StatePoisoned { module_id: None }
10897        )));
10898    }
10899
10900    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
10901    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
10902    /// already reaped out-of-band) still leaves a terminal record rather than none
10903    /// at all. Triggering the real `wait()` I/O error from an integration test would
10904    /// need a genuine already-reaped-child race, which is OS-specific and not
10905    /// something this suite attempts elsewhere; this test instead verifies the
10906    /// record produced for that arm end-to-end through the real `TerminalRing`, and
10907    /// the call site itself is verified by inspection to sit in that exact arm.
10908    #[test]
10909    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
10910        let ring = Arc::new(Mutex::new(TerminalRing::new(
10911            TerminalRingConfig::default(),
10912            0,
10913        )));
10914        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
10915
10916        let snapshot = ring.lock().unwrap().snapshot();
10917        assert_eq!(snapshot.entries.len(), 1);
10918        let entry = &snapshot.entries[0];
10919        assert_eq!(entry.exit_code, None);
10920        assert_eq!(entry.exit_signal, None);
10921        assert_eq!(entry.disposition, TerminalDisposition::Failed);
10922    }
10923
10924    #[test]
10925    fn wait_error_exit_path_preserves_spawn_event_density() {
10926        let feed = super::SpawnEventFeed::default();
10927        feed.configure_incarnation("wait-error-density".to_string());
10928        feed.emit_spawned("wait-error", 41, 1);
10929        let ring = Arc::new(Mutex::new(TerminalRing::new(
10930            TerminalRingConfig::default(),
10931            0,
10932        )));
10933
10934        record_wait_error_terminal("wait-error", &ring, &feed);
10935        feed.emit_spawned("after-wait-error", 42, 2);
10936
10937        let state = feed.0.lock().unwrap();
10938        let sequences = state
10939            .events
10940            .iter()
10941            .map(|event| event.cursor.seq)
10942            .collect::<Vec<_>>();
10943        assert_eq!(sequences, vec![1, 2, 3]);
10944        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
10945        assert_eq!(state.events[1].exit_code, None);
10946        assert_eq!(state.events[1].exit_signal, None);
10947    }
10948
10949    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
10950    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
10951    /// not a clean exit it never actually observed.
10952    #[test]
10953    fn wait_error_exit_report_is_classified_as_a_crash() {
10954        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
10955    }
10956}
10957
10958#[cfg(test)]
10959mod health_evidence_tests {
10960    use super::{HealthProbeError, HealthProbeEvidence};
10961    use std::collections::HashSet;
10962
10963    /// The evidential asymmetry, asserted rather than described.
10964    ///
10965    /// Exactly ONE observation is proof a module cannot serve, and the one that
10966    /// fires under CPU starvation is not it. Before the split, all fifteen
10967    /// construction sites collapsed into a single String, so a timeout carried the
10968    /// same weight as a dead lane -- which is how a healthy module was restarted
10969    /// three times in one day.
10970    #[test]
10971    fn only_a_dead_lane_is_proof_of_death() {
10972        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
10973        // Three non-proof classes, each for a different reason: silence is
10974        // consistent with health, a bad answer proves the module ALIVE, and a
10975        // daemon-side fault never reached the module at all.
10976        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
10977        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
10978        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
10979    }
10980
10981    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
10982    ///
10983    /// A shared label renders two different observations identically in the line an
10984    /// operator reads after an unexplained restart -- the exact confusion this
10985    /// change removes.
10986    #[test]
10987    fn every_evidence_class_has_a_distinct_label() {
10988        let labels = [
10989            HealthProbeError::lane_dead("").label(),
10990            HealthProbeError::no_answer("").label(),
10991            HealthProbeError::bad_answer("").label(),
10992            HealthProbeError::misconfigured("").label(),
10993        ];
10994        let unique: HashSet<_> = labels.iter().collect();
10995        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
10996    }
10997
10998    /// The class is additional information, not a replacement.
10999    ///
11000    /// An operator needs both "this was silence" and the specific text saying how
11001    /// long we waited; a classification that swallowed the message would trade one
11002    /// missing distinction for another.
11003    #[test]
11004    fn classification_preserves_the_original_message() {
11005        let err = HealthProbeError::no_answer("module did not answer within 5s");
11006        assert_eq!(err.to_string(), "module did not answer within 5s");
11007        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11008    }
11009}
11010
11011#[cfg(test)]
11012mod health_tombstone_tests {
11013    use std::{path::PathBuf, sync::Arc, time::Duration};
11014
11015    use subc_protocol::{
11016        manifest::Concurrency,
11017        session::{HealthStatus, ModuleControlResponse},
11018    };
11019    use tokio::sync::mpsc;
11020
11021    use super::{
11022        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11023        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11024    };
11025    use crate::{
11026        control::ControlHandler,
11027        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11028        registry::{ConnectionId, Registry},
11029        router::FrameSink,
11030    };
11031
11032    struct ProbeHarness {
11033        spec: ModuleSpec,
11034        runtime: SupervisorRuntimeConfig,
11035        forwarding: Arc<ForwardingTable>,
11036        module_connection: ConnectionId,
11037        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11038        handler: ControlHandler,
11039        module: super::SupervisedModule,
11040    }
11041
11042    fn probe_harness() -> ProbeHarness {
11043        let registry = Arc::new(Registry::default());
11044        let forwarding = Arc::new(ForwardingTable::default());
11045        let supervisor_handle = super::SupervisorHandle::new();
11046        let health = HealthConfig {
11047            http: None,
11048            cadence: Duration::from_secs(30),
11049            deadline: Duration::from_secs(5),
11050            failure_threshold: 3,
11051            on_degraded: HealthAction::Report,
11052            on_failing: HealthAction::Report,
11053            critical: false,
11054        };
11055        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11056            .with_forwarding(Arc::clone(&forwarding))
11057            .with_handle(supervisor_handle.clone())
11058            .with_health_config(health);
11059        let spec = ModuleSpec {
11060            module_id: "late-health-module".to_string(),
11061            program: PathBuf::from("disabled-module"),
11062            args: Vec::new(),
11063            env: Vec::new(),
11064            reserved: false,
11065            reserved_prefixes: Vec::new(),
11066            protocol: ModuleProtocol::Subc,
11067            overlap: Default::default(),
11068        };
11069        let module = supervisor
11070            .supervise_configured(spec.clone(), false)
11071            .unwrap();
11072        let runtime = supervisor.runtime_config();
11073        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11074            .with_supervisor(supervisor_handle);
11075        let module_connection = ConnectionId::new(700);
11076        let (module_tx, module_rx) = mpsc::channel(8);
11077        forwarding
11078            .register_module_connection(
11079                module_connection,
11080                spec.module_id.clone(),
11081                subc_protocol::PROTOCOL_VERSION,
11082                Concurrency::ModuleManaged,
11083                FrameSink::new(module_tx),
11084            )
11085            .unwrap();
11086
11087        ProbeHarness {
11088            spec,
11089            runtime,
11090            forwarding,
11091            module_connection,
11092            module_rx,
11093            handler,
11094            module,
11095        }
11096    }
11097
11098    async fn finish_after(
11099        harness: &mut ProbeHarness,
11100        stall: Duration,
11101    ) -> ModuleControlRpcCompletion {
11102        assert!(stall > harness.runtime.health.deadline);
11103        let deadline = harness.runtime.health.deadline;
11104        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11105        let answer = async {
11106            let frame = harness.module_rx.recv().await.expect("health.check frame");
11107            tokio::time::advance(deadline).await;
11108            tokio::task::yield_now().await;
11109            tokio::time::advance(stall - deadline).await;
11110            harness
11111                .forwarding
11112                .complete_module_control_rpc(
11113                    harness.module_connection,
11114                    frame.header.corr,
11115                    Some("health.check"),
11116                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11117                        status: HealthStatus::Ok,
11118                        detail: None,
11119                        metrics: None,
11120                    }),
11121                )
11122                .unwrap()
11123        };
11124        let (probe_result, completion) = tokio::join!(probe, answer);
11125        let err = probe_result.expect_err("probe must miss its deadline");
11126        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11127        completion
11128    }
11129
11130    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11131        let deadline = harness.runtime.health.deadline;
11132        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11133        let exhaust_deadline = async {
11134            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11135            tokio::time::advance(deadline).await;
11136            tokio::task::yield_now().await;
11137        };
11138        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11139        let err = probe_result.expect_err("probe must miss its deadline");
11140        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11141    }
11142
11143    #[tokio::test(start_paused = true)]
11144    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
11145        let mut harness = probe_harness();
11146
11147        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
11148        let first_latency = match &first {
11149            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11150            other => panic!("late answer was not retained: {other:?}"),
11151        };
11152        assert!(harness.handler.observe_module_control_completion(first));
11153
11154        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
11155        let second_latency = match &second {
11156            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11157            other => panic!("late answer was not retained: {other:?}"),
11158        };
11159        assert!(harness.handler.observe_module_control_completion(second));
11160
11161        assert_eq!(first_latency, Duration::from_secs(8));
11162        assert_eq!(
11163            second_latency - first_latency,
11164            Duration::from_secs(3),
11165            "latency must grow linearly with the additional stall"
11166        );
11167        let health = harness.module.status().unwrap().health;
11168        assert_eq!(health.late_answer_count, 2);
11169        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
11170    }
11171
11172    /// A module that answers every probe late must never march to the kill
11173    /// threshold: the late answer proves it is alive, so it must clear the miss
11174    /// streak the timeout recorded. Without the reset, a CPU-starved module
11175    /// that serves every probe seconds past the deadline accumulates
11176    /// `consecutive_failures` to the threshold and is killed — the exact
11177    /// sequence from the 2026-08-14 aft disable, where the daemon logged
11178    /// "proves the module is alive" five times while counting five misses.
11179    #[tokio::test(start_paused = true)]
11180    async fn late_answer_clears_the_consecutive_failure_streak() {
11181        let mut harness = probe_harness();
11182
11183        // Timeout recorded first: the probe path saw no answer in time.
11184        time_out_without_answer(&mut harness).await;
11185        harness
11186            .module
11187            .record_health_probe_failure_for_test("[no-answer] test miss")
11188            .unwrap();
11189        assert_eq!(
11190            harness.module.status().unwrap().health.consecutive_failures,
11191            1,
11192            "precondition: the miss must be on the streak before the late answer"
11193        );
11194
11195        // The stalled reply then lands: proof of life.
11196        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
11197        assert!(matches!(
11198            late,
11199            ModuleControlRpcCompletion::LateHealthAnswer { .. }
11200        ));
11201        assert!(harness.handler.observe_module_control_completion(late));
11202
11203        let health = harness.module.status().unwrap().health;
11204        assert_eq!(
11205            health.consecutive_failures, 0,
11206            "a late answer is an answer: the streak must reset"
11207        );
11208        assert_eq!(health.late_answer_count, 1);
11209    }
11210
11211    #[tokio::test(start_paused = true)]
11212    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
11213        let mut harness = probe_harness();
11214
11215        for _ in 0..20 {
11216            time_out_without_answer(&mut harness).await;
11217            assert_eq!(
11218                harness.forwarding.health_probe_tombstone_count().unwrap(),
11219                1
11220            );
11221        }
11222    }
11223}
11224
11225#[cfg(test)]
11226mod child_env_tests {
11227    use super::{
11228        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
11229        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
11230        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
11231    };
11232    use std::{ffi::OsStr, path::PathBuf};
11233    use tokio::process::Command;
11234
11235    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
11236        ModuleSpec {
11237            module_id: "env-plan".to_string(),
11238            program: PathBuf::from("/nonexistent"),
11239            args: Vec::new(),
11240            env,
11241            reserved: false,
11242            reserved_prefixes: Vec::new(),
11243            protocol: ModuleProtocol::Subc,
11244            overlap: Default::default(),
11245        }
11246    }
11247
11248    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
11249    /// one still gets its own.
11250    ///
11251    /// This is the narrow goal `env_clear()` was reached for, and the reason the
11252    /// fix is `env_remove` rather than deleting the line: an operator's ambient
11253    /// filter silently becoming an unconfigured module's log level is a real
11254    /// defect, just a much smaller one than clearing the environment.
11255    ///
11256    /// Asserted on the command plan rather than a spawned child because proving
11257    /// the ABSENCE of an inherited variable needs the parent's environment
11258    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
11259    /// removal as `(key, None)`, which is exactly the distinction wanted: not
11260    /// "absent because nobody set it" but "explicitly unset for the child".
11261    #[test]
11262    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
11263        let mut command = Command::new("/nonexistent");
11264        apply_child_env(&mut command, &spec(Vec::new()));
11265        let removed = command
11266            .as_std()
11267            .get_envs()
11268            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
11269        assert!(
11270            removed,
11271            "ambient CK_LOG must be explicitly removed for an unconfigured module"
11272        );
11273
11274        let mut configured = Command::new("/nonexistent");
11275        apply_child_env(
11276            &mut configured,
11277            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
11278        );
11279        let effective = configured
11280            .as_std()
11281            .get_envs()
11282            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
11283            .last()
11284            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11285        assert_eq!(
11286            effective,
11287            Some(Some("debug".to_string())),
11288            "a module's configured CK_LOG must survive the ambient removal"
11289        );
11290    }
11291
11292    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
11293    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
11294    /// the same reason as the CK_LOG test above.
11295    ///
11296    /// The argument is the load-bearing half: a stock binary exits on an
11297    /// unknown flag before it listens, so with `--subc` appended the mode
11298    /// could not supervise the one process it exists for. Found by the first
11299    /// conformance run (nats-server: `flag provided but not defined: -subc`).
11300    #[test]
11301    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
11302        let connection_file = std::path::Path::new("/run/subc-connection.json");
11303        let handle = SupervisorHandle::new();
11304
11305        let mut none_spec = spec(Vec::new());
11306        none_spec.protocol = ModuleProtocol::None;
11307        let mut none = Command::new("/nonexistent");
11308        let none_handoff =
11309            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
11310                .expect("protocol-none spawn args apply");
11311        assert!(
11312            none_handoff.is_none(),
11313            "protocol:none spawn must not receive a nonce descriptor"
11314        );
11315        assert!(
11316            !none.as_std().get_envs().any(|(key, value)| key
11317                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
11318                && value.is_some()),
11319            "protocol:none spawn must not name a nonce descriptor"
11320        );
11321        let none_args: Vec<String> = none
11322            .as_std()
11323            .get_args()
11324            .map(|a| a.to_string_lossy().into_owned())
11325            .collect();
11326        assert!(
11327            !none_args.iter().any(|a| a == SUBC_ARG),
11328            "protocol:none argv must not carry --subc; got {none_args:?}"
11329        );
11330        let none_has_nonce = none
11331            .as_std()
11332            .get_envs()
11333            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
11334        assert!(
11335            !none_has_nonce,
11336            "protocol:none spawn must not receive a launch nonce"
11337        );
11338        let none_has_module_id = none
11339            .as_std()
11340            .get_envs()
11341            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
11342        assert!(
11343            none_has_module_id,
11344            "SUBC_MODULE_ID is inert and stays on every path"
11345        );
11346        assert!(
11347            handle.spawn_nonce(&none_spec.module_id).is_none(),
11348            "no nonce record for a process that will never present one"
11349        );
11350
11351        // Control: the subc-wire path is unchanged by the branch above.
11352        let wire_spec = spec(Vec::new());
11353        let mut wire = Command::new("/nonexistent");
11354        let wire_handoff =
11355            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
11356                .expect("subc-wire spawn args apply");
11357        let wire_fd_env = wire
11358            .as_std()
11359            .get_envs()
11360            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
11361            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11362        #[cfg(unix)]
11363        assert_eq!(
11364            wire_fd_env,
11365            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11366            "a subc-wire spawn names the pipe it will receive at descriptor 3"
11367        );
11368        #[cfg(not(unix))]
11369        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11370        let wire_args: Vec<String> = wire
11371            .as_std()
11372            .get_args()
11373            .map(|a| a.to_string_lossy().into_owned())
11374            .collect();
11375        assert_eq!(
11376            wire_args,
11377            vec![
11378                SUBC_ARG.to_string(),
11379                connection_file.to_string_lossy().into_owned()
11380            ],
11381            "a subc-wire spawn still carries --subc <path>"
11382        );
11383        assert_eq!(
11384            wire.as_std()
11385                .get_envs()
11386                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11387            !cfg!(unix),
11388            "only Windows supplies the environment nonce"
11389        );
11390        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11391    }
11392
11393    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
11394    /// spec tries to set it; only a swap candidate carries it.
11395    ///
11396    /// "Set it only on candidates" is not enough, because spawn applies the
11397    /// spec's env verbatim and the daemon's own environment is inherited: either
11398    /// could hand a plain restart the swap role, and a module reading it would
11399    /// warm on its long swap budget while callers wait. Asserted as an explicit
11400    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
11401    /// test above gives.
11402    #[test]
11403    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11404        let role = |command: &Command| {
11405            command
11406                .as_std()
11407                .get_envs()
11408                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11409                .last()
11410                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11411        };
11412        let forged = spec(vec![(
11413            SUBC_SPAWN_ROLE_ENV.to_string(),
11414            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11415        )]);
11416
11417        let mut plain = Command::new("/nonexistent");
11418        apply_child_env(&mut plain, &forged);
11419        apply_spawn_role(&mut plain, SpawnRole::Plain);
11420        assert_eq!(
11421            role(&plain),
11422            Some(None),
11423            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11424        );
11425
11426        let mut candidate = Command::new("/nonexistent");
11427        apply_child_env(&mut candidate, &spec(Vec::new()));
11428        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11429        assert_eq!(
11430            role(&candidate),
11431            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11432        );
11433    }
11434
11435    /// Daemon-private capture retention keys never reach the child.
11436    ///
11437    /// cortexkit-log exposes retention as a Rust struct with no environment
11438    /// names, so these entries are supervisor metadata. Passing them through
11439    /// would invent a public child-process contract by accident.
11440    #[test]
11441    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11442        let mut command = Command::new("/nonexistent");
11443        apply_child_env(
11444            &mut command,
11445            &spec(vec![
11446                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11447                ("KEPT".to_string(), "yes".to_string()),
11448            ]),
11449        );
11450        let keys: Vec<String> = command
11451            .as_std()
11452            .get_envs()
11453            .filter(|(_, value)| value.is_some())
11454            .map(|(key, _)| key.to_string_lossy().into_owned())
11455            .collect();
11456        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11457        assert!(
11458            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11459            "daemon-private capture key leaked to the child: {keys:?}"
11460        );
11461    }
11462}
11463
11464#[cfg(test)]
11465mod jitter_tests {
11466    use super::jittered_health_delay;
11467    use std::{collections::HashSet, time::Duration};
11468
11469    /// Module ids drawn from a real fleet, so the dispersal claim is about names
11470    /// that actually occur rather than invented ones.
11471    ///
11472    /// This is a SAMPLE, not a registry: the property under test is that distinct
11473    /// ids disperse, which holds for any set of distinct strings. Several entries
11474    /// are already historical (modules get renamed), and that costs nothing here --
11475    /// but it means a reader must not mistake this for the live module set, and a
11476    /// rename sweep will match it without there being anything to change.
11477    const FLEET: [&str; 14] = [
11478        "aft",
11479        "alfonso-core",
11480        "magic-context",
11481        "broca",
11482        "thalamus",
11483        "quota",
11484        "engram",
11485        "plexus",
11486        "cerebellum",
11487        "astrocyte",
11488        "synapse",
11489        "subc-mcp",
11490        "cortexkit-credentials",
11491        "subc-federation",
11492    ];
11493
11494    /// Probes must not converge after a fleet-wide restart.
11495    ///
11496    /// This is the property the jitter exists for: every module reconnects at
11497    /// once, and without dispersal all fourteen would then probe on the same
11498    /// tick forever. Nothing failed visibly when this went untested -- a
11499    /// convergent fleet still probes correctly, just in a burst, so the symptom
11500    /// is a periodic load spike that looks like whatever else is running.
11501    #[test]
11502    fn probe_delays_disperse_across_the_fleet() {
11503        let cadence = Duration::from_secs(30);
11504        let delays: HashSet<Duration> = FLEET
11505            .iter()
11506            .map(|id| jittered_health_delay(id, 0, cadence))
11507            .collect();
11508        assert_eq!(
11509            delays.len(),
11510            FLEET.len(),
11511            "every supervised module must land on its own probe offset"
11512        );
11513    }
11514
11515    /// The offset may only ever DELAY a probe, never bring it forward.
11516    ///
11517    /// A delay below the cadence would probe a module more often than
11518    /// configured, which is the opposite of what an operator asked for and
11519    /// would tighten the failure budget without anyone changing it.
11520    #[test]
11521    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11522        let cadence = Duration::from_secs(30);
11523        let span = cadence / 10;
11524        for id in FLEET {
11525            for probe_index in 0..8 {
11526                let delay = jittered_health_delay(id, probe_index, cadence);
11527                assert!(
11528                    delay >= cadence,
11529                    "{id}#{probe_index}: jitter must not shorten the cadence"
11530                );
11531                assert!(
11532                    delay < cadence + span,
11533                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11534                );
11535            }
11536        }
11537    }
11538
11539    /// A module keeps its offset across daemon restarts.
11540    ///
11541    /// The delay is derived rather than randomised precisely so a restart does
11542    /// not re-roll every module into a fresh chance of collision. A random
11543    /// source would satisfy the dispersal test above and quietly lose this.
11544    #[test]
11545    fn a_module_offset_is_stable_across_restarts() {
11546        let cadence = Duration::from_secs(30);
11547        for id in FLEET {
11548            assert_eq!(
11549                jittered_health_delay(id, 0, cadence),
11550                jittered_health_delay(id, 0, cadence),
11551                "{id}: the same module and probe index must produce the same offset"
11552            );
11553        }
11554    }
11555
11556    /// A zero cadence disables probing rather than producing a busy loop.
11557    #[test]
11558    fn zero_cadence_yields_zero_delay() {
11559        assert_eq!(
11560            jittered_health_delay("aft", 0, Duration::ZERO),
11561            Duration::ZERO
11562        );
11563    }
11564}
11565
11566#[cfg(all(test, target_os = "linux"))]
11567mod cgroup_placement_tests {
11568    use super::{
11569        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11570        SupervisedChild,
11571    };
11572    use crate::stderr_tail::{StderrRing, StderrTailConfig};
11573    use std::{
11574        fs, io,
11575        path::{Path, PathBuf},
11576        sync::{Arc, Mutex},
11577    };
11578    use subc_test_support::TestTempDir;
11579    use tokio::process::Command;
11580
11581    #[tokio::test]
11582    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11583        use super::*;
11584        let dir = TestTempDir::new("unique-spawn-cgroups");
11585        let root = PathBuf::from(format!(
11586            "/sys/fs/cgroup/subc-unique-test-{}-{}",
11587            std::process::id(),
11588            unix_ms_now()
11589        ));
11590        if let Err(error) = fs::create_dir(&root) {
11591            assert!(
11592                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11593                "required cgroup test cannot execute: {error}"
11594            );
11595            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11596            return;
11597        }
11598        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11599        let group_count = || {
11600            fs::read_dir(root.join("subc-modules"))
11601                .unwrap()
11602                .map(|entry| entry.unwrap().file_type().unwrap())
11603                .filter(|kind| kind.is_dir())
11604                .count()
11605        };
11606        let supervisor =
11607            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
11608                .with_cgroup_placement(Some(placement.clone()));
11609        let runtime = supervisor.runtime_config();
11610        let mut spec = ModuleSpec {
11611            module_id: "unique-spawn".into(),
11612            program: PathBuf::from("/bin/sleep"),
11613            args: vec!["60".into()],
11614            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11615                .into_iter()
11616                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
11617                .collect(),
11618            reserved: false,
11619            reserved_prefixes: vec![],
11620            protocol: ModuleProtocol::None,
11621            overlap: Default::default(),
11622        };
11623        let spawn = |spec: &ModuleSpec| {
11624            spawn_child(
11625                spec,
11626                None,
11627                None,
11628                &runtime.stderr_ring,
11629                None,
11630                &runtime.child_roster,
11631                Some(&placement),
11632            )
11633            .unwrap()
11634        };
11635        let mut live = spawn(&spec);
11636        for _ in 0..3 {
11637            // A new process can enter the old slot while retirement is pending.
11638            let next = spawn(&spec);
11639            assert_ne!(live.module_id, next.module_id);
11640            live.start_kill().unwrap();
11641            live.wait().await.unwrap();
11642            live = next;
11643            assert!(
11644                live.child.try_wait().unwrap().is_none(),
11645                "retiring the old slot must not kill the replacement"
11646            );
11647            assert_eq!(
11648                group_count(),
11649                1,
11650                "only the live spawn's cgroup should remain"
11651            );
11652        }
11653        supervisor.begin_daemon_shutdown();
11654        let reap = tokio::spawn(async move {
11655            live.wait().await.unwrap();
11656        });
11657        supervisor
11658            .end_children_for_daemon_shutdown(false, std::future::pending())
11659            .await;
11660        reap.await.unwrap();
11661        assert_eq!(group_count(), 0);
11662        // A normal exit uses the same tree-cleanup path as a killed spawn.
11663        spec.program = PathBuf::from("/bin/true");
11664        spec.args.clear();
11665        let fresh_roster = ChildRoster::default();
11666        let mut short = spawn_child(
11667            &spec,
11668            None,
11669            None,
11670            &runtime.stderr_ring,
11671            None,
11672            &fresh_roster,
11673            Some(&placement),
11674        )
11675        .unwrap();
11676        short.wait().await.unwrap();
11677        assert_eq!(group_count(), 0);
11678        spec.module_id = "_".repeat(255);
11679        let mut long_id = spawn_child(
11680            &spec,
11681            None,
11682            None,
11683            &runtime.stderr_ring,
11684            None,
11685            &fresh_roster,
11686            Some(&placement),
11687        )
11688        .unwrap();
11689        long_id.wait().await.unwrap();
11690        assert_eq!(
11691            group_count(),
11692            0,
11693            "valid long module IDs must not exceed cgroup NAME_MAX"
11694        );
11695        fs::remove_dir(root.join("subc-modules")).unwrap();
11696        fs::remove_dir(root).unwrap();
11697        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
11698    }
11699
11700    #[test]
11701    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
11702        let path = Path::new("/definitely-missing-subc-cgroup");
11703        let mut command = Command::new("true");
11704        let error = apply_cgroup_placement(
11705            &mut command,
11706            &ModuleSpec {
11707                module_id: "broken-cgroup".to_string(),
11708                program: PathBuf::from("true"),
11709                args: Vec::new(),
11710                env: Vec::new(),
11711                reserved: false,
11712                reserved_prefixes: Vec::new(),
11713                protocol: ModuleProtocol::Subc,
11714                overlap: Default::default(),
11715            },
11716            path,
11717        )
11718        .expect_err("a parent cgroup open failure must reject the supervised spawn");
11719        let reason = error.to_string();
11720
11721        assert!(
11722            matches!(error, SuperviseError::Cgroup { .. }),
11723            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
11724        );
11725        assert!(
11726            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
11727            "parent cgroup open failure must name cgroup.procs: {reason}"
11728        );
11729    }
11730
11731    #[tokio::test]
11732    async fn reaping_a_child_removes_its_empty_module_cgroup() {
11733        let root = TestTempDir::new("supervisor-reap-cgroup");
11734        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11735        let placement = subc_cgroup::prepare_at(&root)
11736            .expect("prepare scratch cgroup root")
11737            .expect("scratch root has a cgroup.procs marker");
11738        let module_id = "reaped-module";
11739        let module = placement
11740            .module_path(module_id)
11741            .expect("create scratch module cgroup");
11742        let child = Command::new("true")
11743            .env("XDG_DATA_HOME", root.path())
11744            .env("XDG_RUNTIME_DIR", root.path())
11745            .env("XDG_CONFIG_HOME", root.path())
11746            .spawn()
11747            .expect("spawn short-lived child");
11748        let pid = child.id().expect("spawned child has pid");
11749        let mut child = SupervisedChild {
11750            child,
11751            protocol: ModuleProtocol::Subc,
11752            module_id: module_id.to_string(),
11753            cgroup_placement: Some(placement),
11754            stdout_pump: None,
11755            stderr_pump: None,
11756            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
11757            spawned_at_ms: 0,
11758            spawned_from: PathBuf::from("true"),
11759            spawned_file_identity: None,
11760            process_start_time: None,
11761            process_identity: None,
11762            pid,
11763            roster_guard: None,
11764            #[cfg(target_os = "macos")]
11765            privacy_exec: None,
11766            spawn_failure: None,
11767        };
11768
11769        child.wait().await.expect("reap short-lived child");
11770
11771        assert!(
11772            !module.exists(),
11773            "reaping the supervised child must remove its empty cgroup"
11774        );
11775    }
11776
11777    #[test]
11778    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
11779        let root = TestTempDir::new("supervisor-non-empty-cgroup");
11780        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11781        let placement = subc_cgroup::prepare_at(&root)
11782            .expect("prepare scratch cgroup root")
11783            .expect("scratch root has a cgroup.procs marker");
11784        let module = placement
11785            .module_path("surviving-module")
11786            .expect("create scratch module cgroup");
11787        fs::write(module.join("surviving-process"), b"still present")
11788            .expect("make scratch cgroup non-empty");
11789        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
11790
11791        remove_module_cgroup(&placement, "surviving-module");
11792
11793        let logs = crate::router::test_log::captured_logs(&logs);
11794        assert!(
11795            module.exists(),
11796            "failed removal must leave the cgroup intact"
11797        );
11798        assert!(
11799            logs.contains("could not remove module cgroup after process exit; continuing teardown")
11800                && logs.contains("surviving-module"),
11801            "best-effort removal must report the failure without returning it: {logs}"
11802        );
11803    }
11804
11805    #[test]
11806    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
11807        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
11808        let reason = SuperviseError::Spawn {
11809            program: PathBuf::from("/bin/true"),
11810            source: io::Error::from_raw_os_error(13),
11811            cgroup_path: Some(cgroup_path.clone()),
11812        }
11813        .to_string();
11814
11815        assert!(
11816            reason.contains(&cgroup_path.display().to_string()),
11817            "a pre_exec spawn failure must name the cgroup path: {reason}"
11818        );
11819    }
11820}
11821
11822#[cfg(test)]
11823mod spawn_subscriber_lag_tests {
11824    use super::*;
11825
11826    /// A subscriber whose connection stops draining is dropped once its frame
11827    /// channel fills. The client must learn that from a terminal Error frame
11828    /// after the frames already queued for it, not from a stream that simply
11829    /// goes quiet.
11830    #[tokio::test]
11831    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
11832        let feed = SpawnEventFeed::default();
11833        feed.configure_incarnation("lag-incarnation".to_string());
11834        // A one-slot connection queue that nobody reads until the emits are
11835        // done: the forwarder parks on it and the subscriber channel fills.
11836        let (tx, mut rx) = mpsc::channel(1);
11837        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
11838            .expect("subscribe");
11839        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
11840        for index in 0..emitted {
11841            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
11842            // Let the forwarder take what it can so the fill point is the
11843            // subscriber channel, not a scheduling accident.
11844            tokio::task::yield_now().await;
11845        }
11846        assert_eq!(
11847            feed.subscriber_count(),
11848            0,
11849            "the lagged subscriber must be removed"
11850        );
11851
11852        let mut data = Vec::new();
11853        let mut last = None;
11854        loop {
11855            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
11856                .await
11857                .expect("the forwarder must finish once the subscriber is dropped");
11858            let Some(outbound) = next else { break };
11859            let frame = outbound.frame;
11860            if frame.header.ty == FrameType::StreamData {
11861                assert!(last.is_none(), "no data may follow the terminal frame");
11862                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
11863                data.push(event.cursor.seq);
11864            } else {
11865                assert!(last.is_none(), "exactly one terminal frame");
11866                last = Some(frame);
11867            }
11868        }
11869        assert!(!data.is_empty(), "queued frames drain before the terminal");
11870        for pair in data.windows(2) {
11871            assert_eq!(
11872                pair[1],
11873                pair[0] + 1,
11874                "queued frames arrive dense and in order"
11875            );
11876        }
11877        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
11878        assert_eq!(terminal.header.ty, FrameType::Error);
11879        assert_eq!(terminal.header.corr, 7);
11880        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
11881        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
11882        let detail = body.detail.expect("lagged error carries detail");
11883        assert_eq!(
11884            detail["first_undelivered_cursor"]["seq"],
11885            data.last().unwrap() + 1,
11886            "the named cursor is the first event the subscriber did not receive"
11887        );
11888        assert_eq!(
11889            detail["first_undelivered_cursor"]["daemon_incarnation"],
11890            "lag-incarnation"
11891        );
11892    }
11893}
11894
11895#[cfg(test)]
11896mod terminal_history_read_concurrency_tests {
11897    use super::*;
11898    use crate::terminal_journal::read_pause;
11899    use std::sync::mpsc as std_mpsc;
11900    use subc_test_support::TestTempDir;
11901
11902    fn journaled_ring(
11903        journal: &Arc<crate::terminal_journal::TerminalJournal>,
11904    ) -> Arc<Mutex<TerminalRing>> {
11905        Arc::new(Mutex::new(
11906            TerminalRing::new(TerminalRingConfig::default(), 1)
11907                .with_journal(Some(Arc::clone(journal))),
11908        ))
11909    }
11910
11911    fn crash(at_ms: u64) -> ExitReport {
11912        ExitReport {
11913            kind: ExitKind::Crash,
11914            code: Some(1),
11915            signal: None,
11916            at_ms,
11917        }
11918    }
11919
11920    /// Record an exit on another thread and report whether it finished within
11921    /// `bound`. The recorder thread is left running if it did not.
11922    fn record_within(
11923        module_id: &'static str,
11924        ring: &Arc<Mutex<TerminalRing>>,
11925        at_ms: u64,
11926        bound: Duration,
11927    ) -> bool {
11928        let ring = Arc::clone(ring);
11929        let (done, done_rx) = std_mpsc::channel();
11930        std::thread::spawn(move || {
11931            record_terminal(
11932                module_id,
11933                &ring,
11934                &SpawnEventFeed::default(),
11935                &crash(at_ms),
11936                TerminalDisposition::Restarting,
11937            );
11938            let _ = done.send(());
11939        });
11940        done_rx.recv_timeout(bound).is_ok()
11941    }
11942
11943    /// A history read in progress must not hold the journal writer (which every
11944    /// module's exit recording needs) or the module's own ring. Exits recorded
11945    /// while the read is paused complete promptly; the paused read answers as of
11946    /// the moment it started, and the next read has each exit exactly once.
11947    #[test]
11948    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
11949        let dir = TestTempDir::new("terminal-history-concurrent-read");
11950        let path = dir.join("terminals.jsonl");
11951        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
11952            path.clone(),
11953            "daemon".into(),
11954        ));
11955        let reader_ring = journaled_ring(&journal);
11956        let other_ring = journaled_ring(&journal);
11957        assert!(record_within(
11958            "reader-module",
11959            &reader_ring,
11960            10,
11961            Duration::from_secs(5)
11962        ));
11963
11964        let (started, release) = read_pause::install(&path);
11965        let reading = {
11966            let ring = Arc::clone(&reader_ring);
11967            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
11968        };
11969        started
11970            .recv_timeout(Duration::from_secs(5))
11971            .expect("the history read reached its pause");
11972
11973        let bound = Duration::from_secs(1);
11974        assert!(
11975            record_within("other-module", &other_ring, 20, bound),
11976            "another module's exit waited on a history read (journal writer held)"
11977        );
11978        assert!(
11979            record_within("reader-module", &reader_ring, 30, bound),
11980            "the read module's own exit waited on its history read (ring held)"
11981        );
11982
11983        drop(release);
11984        let paused = reading.join().unwrap();
11985        assert_eq!(
11986            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
11987            vec![10],
11988            "an exit recorded after the read began lands in neither half of it"
11989        );
11990        assert_eq!(paused.journal_skipped_lines, 0);
11991        assert_eq!(paused.journal_read_errors, 0);
11992
11993        let after = durable_terminal_history_of(&reader_ring, "reader-module");
11994        assert_eq!(
11995            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
11996            vec![10, 30],
11997            "the next read merges ring and journal with no duplicate"
11998        );
11999        assert_eq!(after.journal_skipped_lines, 0);
12000    }
12001}
12002
12003/// What a restart does with the exited process's stderr reader. These drive
12004/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12005/// holds, so a reader that has not been scheduled by the bound is a controlled
12006/// input rather than something only a loaded machine produces.
12007#[cfg(test)]
12008mod stderr_settle_tests {
12009    use std::{
12010        future::Future,
12011        io,
12012        pin::Pin,
12013        sync::{Arc, Mutex},
12014        task::{Context, Poll},
12015        time::Duration,
12016    };
12017
12018    use tokio::{
12019        io::{AsyncRead, ReadBuf},
12020        sync::oneshot,
12021        time::Instant,
12022    };
12023
12024    use super::{settle_stderr_pump, StderrPump};
12025    use crate::stderr_tail::{
12026        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12027    };
12028
12029    const BOUND: Duration = Duration::from_millis(250);
12030
12031    /// Yields `before`, then stays pending until the gate is released, then
12032    /// yields `after` and reaches EOF. The bytes after the gate were written
12033    /// by a process that has already exited; only the reader is behind.
12034    struct HeldReader {
12035        before: Option<Vec<u8>>,
12036        gate: Option<oneshot::Receiver<()>>,
12037        after: io::Cursor<Vec<u8>>,
12038    }
12039
12040    impl AsyncRead for HeldReader {
12041        fn poll_read(
12042            mut self: Pin<&mut Self>,
12043            cx: &mut Context<'_>,
12044            buf: &mut ReadBuf<'_>,
12045        ) -> Poll<io::Result<()>> {
12046            if let Some(bytes) = self.before.take() {
12047                buf.put_slice(&bytes);
12048                return Poll::Ready(Ok(()));
12049            }
12050            if let Some(gate) = self.gate.as_mut() {
12051                match Pin::new(gate).poll(cx) {
12052                    Poll::Pending => return Poll::Pending,
12053                    Poll::Ready(_) => self.gate = None,
12054                }
12055            }
12056            Pin::new(&mut self.after).poll_read(cx, buf)
12057        }
12058    }
12059
12060    struct DiscardSink;
12061
12062    impl OutputSink for DiscardSink {
12063        fn write_line(&mut self, _line: &[u8]) {}
12064    }
12065
12066    fn line(text: &str) -> TailEntry {
12067        TailEntry::Line {
12068            text: text.to_string(),
12069            truncated: false,
12070            at_ms: None,
12071        }
12072    }
12073
12074    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
12075        ring.lock().unwrap()
12076    }
12077
12078    /// Start a reader for a new process generation that delivers `before`
12079    /// immediately and `after` only once the returned sender fires (or is
12080    /// dropped).
12081    fn held_pump(
12082        ring: &Arc<Mutex<StderrRing>>,
12083        before: &str,
12084        after: &str,
12085    ) -> (StderrPump, oneshot::Sender<()>) {
12086        let generation = lock(ring).begin_process();
12087        let (release, gate) = oneshot::channel();
12088        let reader = HeldReader {
12089            before: Some(before.as_bytes().to_vec()),
12090            gate: Some(gate),
12091            after: io::Cursor::new(after.as_bytes().to_vec()),
12092        };
12093        let task = tokio::spawn(pump_stderr_to(
12094            reader,
12095            Arc::clone(ring),
12096            generation,
12097            DiscardSink,
12098        ));
12099        (StderrPump { task, generation }, release)
12100    }
12101
12102    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
12103        for _ in 0..1000 {
12104            if done(&lock(ring)) {
12105                return;
12106            }
12107            tokio::time::sleep(Duration::from_millis(1)).await;
12108        }
12109        panic!(
12110            "ring never reached the expected state: {:?}",
12111            lock(ring).snapshot(None, None)
12112        );
12113    }
12114
12115    #[tokio::test(start_paused = true)]
12116    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
12117        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12118        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
12119
12120        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
12121        let before_release = lock(&ring).snapshot(None, None);
12122        assert!(
12123            matches!(before_release.capture, CaptureState::Incomplete { .. }),
12124            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
12125        );
12126
12127        // The restart: the next process starts and writes before the old
12128        // reader catches up.
12129        let next = lock(&ring).begin_process();
12130        lock(&ring).push_line_from(next, "next process booting");
12131        release.send(()).unwrap();
12132        wait_until(&ring, |ring| {
12133            ring.snapshot(None, None).capture == CaptureState::Captured
12134        })
12135        .await;
12136
12137        assert_eq!(
12138            untimed(lock(&ring).snapshot(None, None).entries),
12139            vec![
12140                line("booting"),
12141                line("config error: missing storage"),
12142                TailEntry::ProcessStart,
12143                line("next process booting"),
12144            ],
12145            "the crash's last line must survive a slow reader and stay in the crashed process's section"
12146        );
12147    }
12148
12149    #[tokio::test(start_paused = true)]
12150    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
12151    ) {
12152        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12153        // `_held` is never fired: a descendant keeps the pipe open for the
12154        // whole test.
12155        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
12156
12157        let started = Instant::now();
12158        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
12159        assert_eq!(
12160            started.elapsed(),
12161            BOUND,
12162            "the restart must wait exactly the bound for a pipe that stays open, no longer"
12163        );
12164
12165        let next = lock(&ring).begin_process();
12166        lock(&ring).push_line_from(next, "next process booting");
12167        tokio::time::sleep(Duration::from_secs(60)).await;
12168
12169        let snapshot = lock(&ring).snapshot(None, None);
12170        match &snapshot.capture {
12171            CaptureState::Incomplete { reason } => assert!(
12172                reason.contains("had not reached EOF") && reason.contains("250ms"),
12173                "the reason must say what is missing and after how long: {reason}"
12174            ),
12175            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
12176        }
12177        assert_eq!(
12178            untimed(snapshot.entries),
12179            vec![
12180                line("parent exiting"),
12181                TailEntry::ProcessStart,
12182                line("next process booting"),
12183            ]
12184        );
12185    }
12186
12187    #[tokio::test(start_paused = true)]
12188    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
12189        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12190        let (pump, release) = held_pump(&ring, "one\n", "two\n");
12191        release.send(()).unwrap();
12192
12193        settle_stderr_pump("clean", &ring, pump, BOUND).await;
12194
12195        let snapshot = lock(&ring).snapshot(None, None);
12196        assert_eq!(snapshot.capture, CaptureState::Captured);
12197        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
12198    }
12199}
12200
12201/// Containment of a module's process tree (issue #109).
12202///
12203/// The behaviour these defend against is a module helper surviving its module:
12204/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
12205/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
12206/// compounds it.
12207///
12208/// They run against the SUPERVISOR rather than the job-object crate because the
12209/// claim is about teardown: a crate-level test proves a job can reap a tree, not
12210/// that the daemon's drain path reaches it.
12211///
12212/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
12213/// lane there is a separate containment path with its own tests.
12214#[cfg(all(test, windows))]
12215mod job_containment_tests {
12216    use super::*;
12217    use std::{
12218        path::{Path, PathBuf},
12219        sync::{Arc, Mutex},
12220        time::{Duration, Instant},
12221    };
12222    use subc_test_support::TestTempDir;
12223
12224    /// The stub, expected beside this test executable.
12225    ///
12226    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
12227    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
12228    /// failure then reads as a broken test rather than an unbuilt dependency.
12229    fn stub_path() -> PathBuf {
12230        let mut path = std::env::current_exe().expect("current_exe available in tests");
12231        path.pop();
12232        path.pop();
12233        path.push("fake-aft-stub.exe");
12234        assert!(
12235            path.exists(),
12236            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
12237             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
12238            path.display()
12239        );
12240        path
12241    }
12242
12243    /// Poll for the grandchild pid the stub records, and parse it.
12244    fn read_grandchild_pid(path: &Path) -> u32 {
12245        let deadline = Instant::now() + Duration::from_secs(10);
12246        loop {
12247            if let Ok(contents) = std::fs::read_to_string(path) {
12248                if let Ok(pid) = contents.trim().parse() {
12249                    return pid;
12250                }
12251            }
12252            assert!(
12253                Instant::now() < deadline,
12254                "the stub never recorded a grandchild pid at {}",
12255                path.display()
12256            );
12257            std::thread::sleep(Duration::from_millis(10));
12258        }
12259    }
12260
12261    /// Everything one fixture run needs, so the two tests below differ in exactly
12262    /// one place: whether the child is contained.
12263    struct Fixture {
12264        _dir: TestTempDir,
12265        module_id: String,
12266        grandchild: u32,
12267        child: Option<SupervisedChild>,
12268        registry: Arc<Registry>,
12269        snapshot: Arc<Mutex<SupervisorSnapshot>>,
12270        terminal_ring: Arc<Mutex<TerminalRing>>,
12271        spawn_events: SpawnEventFeed,
12272    }
12273
12274    fn fixture(label: &str, module_id: &str) -> Fixture {
12275        let dir = TestTempDir::new(label);
12276        let pid_file = dir.join("grandchild.pid");
12277        let supervisor = Supervisor::new_for_test(
12278            Arc::new(Registry::default()),
12279            RestartPolicy::new(3, Duration::ZERO),
12280        );
12281        let runtime = supervisor.runtime_config();
12282        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12283        let spec = ModuleSpec {
12284            module_id: module_id.to_string(),
12285            program: stub_path(),
12286            // Zero args deliberately: a `--subc` argument would make the stub dial
12287            // a daemon that is not there, and the failure would land in the same
12288            // stderr ring this fixture exists to keep quiet.
12289            args: Vec::new(),
12290            env: vec![
12291                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
12292                (
12293                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
12294                    pid_file.display().to_string(),
12295                ),
12296            ],
12297            reserved: false,
12298            reserved_prefixes: Vec::new(),
12299            protocol: ModuleProtocol::Subc,
12300            overlap: Default::default(),
12301        };
12302        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
12303            .expect("spawn the supervised fixture");
12304        let grandchild = read_grandchild_pid(&pid_file);
12305        Fixture {
12306            _dir: dir,
12307            module_id: module_id.to_string(),
12308            grandchild,
12309            child: Some(child),
12310            registry: Arc::new(Registry::default()),
12311            snapshot,
12312            terminal_ring: Arc::clone(&runtime.terminal_ring),
12313            spawn_events: SpawnEventFeed::default(),
12314        }
12315    }
12316
12317    impl Fixture {
12318        /// Drain through the supervisor's own teardown path.
12319        async fn drain(&mut self) {
12320            let child = self
12321                .child
12322                .take()
12323                .expect("the fixture child is still present");
12324            drain_child_to_state(
12325                &self.module_id,
12326                ModuleProtocol::Subc,
12327                // No forwarding table in this fixture, so nothing reaches the
12328                // child over a connection.
12329                StopNotice::NotSent,
12330                &self.registry,
12331                None,
12332                &self.snapshot,
12333                &self.terminal_ring,
12334                &self.spawn_events,
12335                child,
12336                Duration::from_millis(500),
12337                ModuleState::Stopped,
12338                Some(false),
12339            )
12340            .await
12341            .expect("drain the supervised fixture");
12342        }
12343    }
12344
12345    /// Teardown reaps the grandchild, not merely the direct child.
12346    ///
12347    /// This is the assertion the change exists for. Before containment the
12348    /// grandchild survived: it is a separate process, and `start_kill` is
12349    /// `TerminateProcess` scoped to one pid.
12350    #[tokio::test]
12351    async fn teardown_reaps_the_grandchild() {
12352        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
12353        let grandchild = fixture.grandchild;
12354
12355        assert!(
12356            subc_jobobject::process_exists(grandchild),
12357            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
12358        );
12359
12360        fixture.drain().await;
12361
12362        assert!(
12363            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12364            "grandchild {grandchild} outlived module teardown: the tree was not contained"
12365        );
12366    }
12367
12368    /// The mutation control: with containment withheld, the grandchild survives
12369    /// the same kill.
12370    ///
12371    /// This is the defect reproduction from #109 — a direct-child kill reaches
12372    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
12373    /// supervisor because `spawn_and_mark_running` now always contains on
12374    /// Windows, which is the point: there is no longer a path that spawns
12375    /// uncontained, so the control has to construct one.
12376    ///
12377    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
12378    /// grandchild ever dies here, that test is passing for a reason unrelated to
12379    /// the job object and the containment claim is unproven.
12380    #[test]
12381    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12382        let dir = TestTempDir::new("teardown-uncontained");
12383        let pid_file = dir.join("grandchild.pid");
12384        let mut child = std::process::Command::new(stub_path())
12385            .env("FAKE_AFT_NEVER_CONNECT", "1")
12386            .env(
12387                "FAKE_AFT_GRANDCHILD_PID_FILE",
12388                pid_file.display().to_string(),
12389            )
12390            .stdin(std::process::Stdio::null())
12391            .stdout(std::process::Stdio::null())
12392            .stderr(std::process::Stdio::null())
12393            .spawn()
12394            .expect("spawn the uncontained fixture");
12395        let grandchild = read_grandchild_pid(&pid_file);
12396
12397        // Exactly what the pre-fix teardown did: kill the direct child.
12398        child.kill().expect("kill the direct child");
12399        let _ = child.wait();
12400
12401        assert!(
12402            subc_jobobject::process_exists(grandchild),
12403            "grandchild {grandchild} died with the direct child, so this control no longer \
12404             distinguishes contained from uncontained teardown and the regression test is \
12405             passing vacuously"
12406        );
12407
12408        // The orphan this control demonstrates is the leak the fix prevents, so
12409        // the control must not leave one behind.
12410        kill_tree(grandchild);
12411    }
12412
12413    /// Crash durability: closing the containment handle reaps the tree with no
12414    /// teardown code running at all.
12415    ///
12416    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
12417    /// call anything — and it is why containment is a kernel property of the
12418    /// handle rather than a step in the drain. Discovered by getting the
12419    /// mutation control wrong: clearing `job` to "disable" containment instead
12420    /// killed the tree, which is the guarantee, not a mistake.
12421    #[tokio::test]
12422    async fn dropping_containment_reaps_the_grandchild() {
12423        let mut fixture = fixture("drop-containment", "tree-drop");
12424        let grandchild = fixture.grandchild;
12425
12426        assert!(subc_jobobject::process_exists(grandchild));
12427
12428        // No `drain` call, no kill: dropping the handle is the entire mechanism.
12429        fixture.child.as_mut().expect("child present").job = None;
12430
12431        assert!(
12432            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12433            "grandchild {grandchild} survived the containment handle closing, so a daemon \
12434             crash would leave the tree behind"
12435        );
12436    }
12437
12438    /// Kill a pid and its tree, then confirm it is gone.
12439    fn kill_tree(pid: u32) {
12440        let _ = std::process::Command::new("taskkill.exe")
12441            .args(["/PID", &pid.to_string(), "/T", "/F"])
12442            .stdin(std::process::Stdio::null())
12443            .stdout(std::process::Stdio::null())
12444            .stderr(std::process::Stdio::null())
12445            .status();
12446        assert!(
12447            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12448            "could not clean up grandchild {pid}"
12449        );
12450    }
12451}
12452
12453#[cfg(test)]
12454mod privacy_trampoline_configuration_tests {
12455    #[tokio::test]
12456    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12457    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
12458        #[cfg(target_os = "macos")]
12459        {
12460            let supervisor = super::Supervisor::new(
12461                std::sync::Arc::new(crate::Registry::default()),
12462                super::RestartPolicy::default(),
12463            );
12464            let error = supervisor.spawn(spec()).unwrap_err();
12465            assert!(
12466                error
12467                    .to_string()
12468                    .contains("no privacy trampoline configured"),
12469                "{error}"
12470            );
12471        }
12472    }
12473
12474    #[tokio::test]
12475    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12476    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
12477        #[cfg(target_os = "macos")]
12478        {
12479            let supervisor = super::Supervisor::new(
12480                std::sync::Arc::new(crate::Registry::default()),
12481                super::RestartPolicy::default(),
12482            )
12483            .with_privacy_trampoline(std::env::current_exe().unwrap());
12484            let error = supervisor.spawn(spec()).unwrap_err();
12485            assert!(
12486                error
12487                    .to_string()
12488                    .contains("binary does not implement the privacy trampoline protocol"),
12489                "{error}"
12490            );
12491        }
12492    }
12493
12494    #[cfg(target_os = "macos")]
12495    fn spec() -> super::ModuleSpec {
12496        super::ModuleSpec {
12497            module_id: "privacy-configuration".into(),
12498            program: "/bin/sleep".into(),
12499            args: vec!["30".into()],
12500            env: vec![],
12501            reserved: false,
12502            reserved_prefixes: vec![],
12503            protocol: subc_control::ModuleProtocol::None,
12504            overlap: super::ModuleOverlap::Exclusive,
12505        }
12506    }
12507}
12508
12509/// The daemon's real spawn path hands a subc-wire child its launch nonce on
12510/// descriptor 3, without an environment copy. The shell records the nonce
12511/// and its environment after exec so these tests observe the real handover.
12512#[cfg(all(test, unix))]
12513mod launch_nonce_descriptor_tests {
12514    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
12515    use crate::stderr_tail::{StderrRing, StderrTailConfig};
12516    use std::{
12517        path::PathBuf,
12518        sync::{Arc, Mutex},
12519        time::{Duration, Instant},
12520    };
12521    use subc_test_support::TestTempDir;
12522
12523    async fn probe(role: super::SpawnRole) {
12524        let scratch = TestTempDir::new("launch-nonce-descriptor");
12525        let fd_copy = scratch.join("from-descriptor");
12526        let env_copy = scratch.join("environment");
12527        let script = format!(
12528            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
12529            fd = fd_copy.display(), env = env_copy.display(),
12530        );
12531        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
12532        let spec = ModuleSpec {
12533            module_id: "nonce-descriptor-probe".to_string(),
12534            program: PathBuf::from("/bin/sh"),
12535            args: vec!["-c".to_string(), script],
12536            env: vec![
12537                xdg("XDG_DATA_HOME"),
12538                xdg("XDG_RUNTIME_DIR"),
12539                xdg("XDG_CONFIG_HOME"),
12540                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
12541            ],
12542            reserved: true,
12543            reserved_prefixes: Vec::new(),
12544            protocol: ModuleProtocol::Subc,
12545            overlap: Default::default(),
12546        };
12547        let handle = SupervisorHandle::new();
12548        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12549        let roster = ChildRoster::default();
12550        #[cfg(target_os = "macos")]
12551        {
12552            let path = super::test_privacy_trampoline();
12553            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
12554        }
12555        let child = super::spawn_child_in_slot(
12556            &spec,
12557            None,
12558            Some(&handle),
12559            &ring,
12560            None,
12561            &roster,
12562            #[cfg(target_os = "linux")]
12563            None,
12564            role,
12565            matches!(role, super::SpawnRole::SwapCandidate),
12566        )
12567        .expect("spawn probe");
12568        let deadline = Instant::now() + Duration::from_secs(10);
12569        while !(fd_copy.exists() && env_copy.exists()) {
12570            assert!(Instant::now() < deadline, "probe never wrote its copies");
12571            tokio::time::sleep(Duration::from_millis(20)).await;
12572        }
12573        let nonce = std::fs::read_to_string(fd_copy).unwrap();
12574        assert!(!nonce.is_empty());
12575        let environment = std::fs::read_to_string(env_copy).unwrap();
12576        assert!(environment
12577            .lines()
12578            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
12579        let copy = environment
12580            .lines()
12581            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
12582        assert_eq!(
12583            copy, None,
12584            "Unix children must never receive the environment nonce"
12585        );
12586        if matches!(role, super::SpawnRole::Plain) {
12587            assert_eq!(
12588                handle.spawn_nonce(&spec.module_id).as_deref(),
12589                Some(nonce.as_str())
12590            );
12591        }
12592        drop(child);
12593    }
12594
12595    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12596    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
12597        probe(super::SpawnRole::Plain).await;
12598    }
12599
12600    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12601    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
12602        probe(super::SpawnRole::SwapCandidate).await;
12603    }
12604}
12605
12606#[cfg(all(test, target_os = "linux"))]
12607mod cgroup_containment_tests {
12608    use super::*;
12609    use subc_test_support::TestTempDir;
12610
12611    fn running(pid: u32) -> bool {
12612        // An orphan can remain a zombie until the container init reaps it.
12613        std::fs::read_to_string(format!("/proc/{pid}/stat"))
12614            .ok()
12615            .and_then(|stat| {
12616                stat.rsplit_once(") ")
12617                    .map(|(_, rest)| rest.starts_with('Z'))
12618            })
12619            .is_some_and(|zombie| !zombie)
12620    }
12621
12622    #[tokio::test]
12623    async fn linux_teardown_reaps_the_grandchild() {
12624        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
12625    }
12626
12627    #[tokio::test]
12628    async fn linux_shutdown_straggler_reaps_the_grandchild() {
12629        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
12630    }
12631
12632    async fn teardown_tree(test_name: &str, shutdown: bool) {
12633        let dir = TestTempDir::new(test_name);
12634        let root = PathBuf::from(format!(
12635            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
12636            std::process::id(),
12637            unix_ms_now()
12638        ));
12639        if let Err(error) = std::fs::create_dir(&root) {
12640            assert!(
12641                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12642                "required cgroup test cannot execute: {error}"
12643            );
12644            eprintln!(
12645                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
12646                root.display()
12647            );
12648            return;
12649        }
12650        let placement = subc_cgroup::prepare_at(&root)
12651            .expect("prepare isolated kernel cgroup")
12652            .expect("isolated cgroup is delegated");
12653        let module_id = "tree-teardown";
12654        let module = placement
12655            .module_path(module_id)
12656            .expect("create isolated module cgroup");
12657        if !module.join("cgroup.kill").exists() {
12658            std::fs::remove_dir(&module).unwrap();
12659            std::fs::remove_dir(root.join("subc-modules")).unwrap();
12660            std::fs::remove_dir(&root).unwrap();
12661            assert!(
12662                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12663                "required cgroup.kill interface unavailable"
12664            );
12665            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
12666            return;
12667        }
12668        let supervisor = Supervisor::new_for_test(
12669            Arc::new(Registry::default()),
12670            RestartPolicy::new(3, Duration::ZERO),
12671        )
12672        .with_cgroup_placement(Some(placement));
12673        let mut runtime = supervisor.runtime_config();
12674        runtime.child_roster = runtime
12675            .child_roster
12676            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
12677        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12678        let pid_file = dir.join("grandchild.pid");
12679        let spec = ModuleSpec {
12680            module_id: module_id.to_string(),
12681            program: PathBuf::from("/bin/sh"),
12682            args: vec![
12683                "-c".into(),
12684                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
12685                "fixture".into(),
12686                pid_file.display().to_string(),
12687            ],
12688            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
12689                .into_iter()
12690                .map(|key| (key.to_string(), dir.display().to_string()))
12691                .collect(),
12692            reserved: false,
12693            reserved_prefixes: Vec::new(),
12694            protocol: ModuleProtocol::None,
12695            overlap: Default::default(),
12696        };
12697        let child =
12698            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
12699        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
12700        let grandchild: u32 = loop {
12701            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
12702                if let Ok(pid) = contents.trim().parse() {
12703                    break pid;
12704                }
12705            }
12706            assert!(
12707                tokio::time::Instant::now() < deadline,
12708                "grandchild pid was not recorded"
12709            );
12710            tokio::time::sleep(Duration::from_millis(10)).await;
12711        };
12712        assert!(
12713            running(grandchild),
12714            "grandchild must be alive before teardown"
12715        );
12716        if shutdown {
12717            let mut child = child;
12718            crate::child_roster::end_children_for_daemon_shutdown(
12719                &runtime.child_roster,
12720                false,
12721                std::future::pending(),
12722            )
12723            .await;
12724            child.wait().await.expect("reap shutdown straggler");
12725        } else {
12726            drain_child_to_state(
12727                module_id,
12728                ModuleProtocol::None,
12729                StopNotice::NotSent,
12730                &Registry::default(),
12731                None,
12732                &snapshot,
12733                &runtime.terminal_ring,
12734                &SpawnEventFeed::default(),
12735                child,
12736                Duration::from_millis(100),
12737                ModuleState::Stopped,
12738                Some(false),
12739            )
12740            .await
12741            .expect("real supervisor teardown");
12742        }
12743        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
12744        while running(grandchild) && tokio::time::Instant::now() < deadline {
12745            tokio::time::sleep(Duration::from_millis(10)).await;
12746        }
12747        let survived = running(grandchild);
12748        // Kill a surviving grandchild so a failed test does not leave it behind.
12749        if survived {
12750            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
12751            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
12752            tokio::time::sleep(Duration::from_millis(100)).await;
12753        }
12754        if module.exists() {
12755            std::fs::remove_dir(&module).expect("remove empty module cgroup");
12756        }
12757        std::fs::remove_dir(root.join("subc-modules")).unwrap();
12758        std::fs::remove_dir(&root).unwrap();
12759        assert!(
12760            !survived,
12761            "grandchild {grandchild} outlived module teardown"
12762        );
12763        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
12764    }
12765}