Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113struct SupervisedChild {
114    child: Child,
115    /// The name of this process's cgroup: the module id, or for a swap
116    /// candidate the alternate name (see `swap::cgroup_name`).
117    #[cfg(target_os = "linux")]
118    module_id: String,
119    #[cfg(target_os = "linux")]
120    cgroup_placement: Option<subc_cgroup::Placement>,
121    /// The job that contains this child and every process it spawns (issue #109).
122    ///
123    /// Dropping this handle is what reaps a surviving tree when no supervisor
124    /// code runs — a daemon crash — because the job carries
125    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
126    ///
127    /// That limit is not crash-only, and the difference is worth knowing: a
128    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
129    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
130    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
131    /// module at once. Before this change they survived that, saw EOF on the
132    /// control socket, and ran their own teardown; Unix keeps that path
133    /// deliberately, so a module can seal a WAL or close a capture rather than
134    /// be killed mid-write. So this trades graceful teardown on every Windows
135    /// daemon stop for containment on a crash, which is the right way round
136    /// today: orphaned GPU workers are a reported, recurring problem, and the
137    /// modules that write most heavily do not run on Windows.
138    ///
139    /// The fix is a real Windows stop path — the daemon draining before it
140    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
141    /// reaches only what the drain left behind, which is what it should reach.
142    #[cfg(windows)]
143    job: Option<subc_jobobject::JobObject>,
144    stdout_pump: Option<JoinHandle<()>>,
145    stderr_pump: Option<StderrPump>,
146    stderr_ring: Arc<Mutex<StderrRing>>,
147    spawned_at_ms: u64,
148    spawned_from: PathBuf,
149    spawned_file_identity: Option<SpawnedFileIdentity>,
150    process_start_time: Option<u64>,
151    process_identity: Option<ProcessIdentity>,
152    pid: u32,
153    /// This process's entry in the daemon's child roster, released when the
154    /// process is reaped or this handle is dropped.
155    roster_guard: Option<crate::child_roster::RosterGuard>,
156}
157
158impl SupervisedChild {
159    fn id(&self) -> Option<u32> {
160        Some(self.pid)
161    }
162
163    fn process_identity(&self) -> Option<ProcessIdentity> {
164        self.process_identity
165    }
166
167    async fn wait(&mut self) -> io::Result<ExitStatus> {
168        // The roster entry is NOT released here. A daemon shutdown waits for the
169        // roster to empty and then exits the process, so releasing at the reap
170        // let it exit before the exit handler wrote this child's terminal record
171        // (the stderr drain and snapshot update sit in between), and the
172        // shutdown's own `daemon_shutdown` record was intermittently lost. The
173        // caller releases it after recording the exit (`release_roster`), and
174        // dropping the handle releases it too.
175        let result = self.child.wait().await;
176        #[cfg(target_os = "linux")]
177        if result.is_ok() {
178            if let Some(placement) = self.cgroup_placement.take() {
179                remove_module_cgroup(&placement, &self.module_id);
180            }
181        }
182        result
183    }
184
185    /// Releases this child's daemon-shutdown roster entry once its exit has
186    /// been recorded. The pid is already reaped and free for reuse, so the
187    /// entry must not outlive the record any longer than that.
188    fn release_roster(&mut self) {
189        self.roster_guard = None;
190    }
191
192    /// Kill the child and, where containment is available, its process tree.
193    ///
194    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
195    /// helper process leaked the helper — the Synapse embedding module's CUDA
196    /// worker holds the GPU allocation, so the leak cost VRAM until the next
197    /// restart of something else. Terminating the job reaches grandchildren that
198    /// a tree walk cannot, including one whose parent has already exited and
199    /// been reparented away.
200    ///
201    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
202    /// direct-child kill still decides the outcome, so containment can never
203    /// change whether a module is reported as stopped.
204    fn start_kill(&mut self) -> io::Result<()> {
205        #[cfg(windows)]
206        if let Some(job) = &self.job {
207            if let Err(error) = job.terminate() {
208                debug!(
209                    error = %error,
210                    "job termination failed; the direct-child kill still owns the outcome"
211                );
212            }
213        }
214        #[cfg(target_os = "linux")]
215        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
216        self.child.start_kill()
217    }
218
219    async fn drain_stderr(&mut self, module_id: &str) {
220        if let Some(mut pump) = self.stdout_pump.take() {
221            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
222                Ok(Ok(())) => {}
223                Ok(Err(error)) => {
224                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
225                }
226                Err(_) => {
227                    pump.abort();
228                    warn!(
229                        module_id,
230                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
231                        "stdout pump did not drain before restart; stopped it before the next process"
232                    );
233                }
234            }
235        }
236
237        let Some(pump) = self.stderr_pump.take() else {
238            return;
239        };
240        settle_stderr_pump(
241            module_id,
242            &self.stderr_ring,
243            pump,
244            STDERR_PUMP_DRAIN_TIMEOUT,
245        )
246        .await;
247    }
248}
249
250/// The reader task for one process's stderr, with the ring generation its
251/// lines are attributed to.
252struct StderrPump {
253    task: JoinHandle<()>,
254    generation: u64,
255}
256
257/// Retire an exited process's stderr reader and wait up to `bound` for it to
258/// reach EOF. A reader still running at the bound is detached, not stopped: it
259/// keeps filling the exited process's section of the ring until its pipe
260/// closes, and the tail reads `Incomplete` until then. See
261/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
262async fn settle_stderr_pump(
263    module_id: &str,
264    ring: &Arc<Mutex<StderrRing>>,
265    pump: StderrPump,
266    bound: Duration,
267) {
268    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
269    let StderrPump {
270        mut task,
271        generation,
272    } = pump;
273    lock().retire_pump(generation);
274    match timeout(bound, &mut task).await {
275        Ok(Ok(())) => {}
276        Ok(Err(err)) => {
277            let mut ring = lock();
278            ring.mark_incomplete(format!("stderr pump ended unexpectedly: {err}"));
279            ring.finish_pump(generation);
280            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
281        }
282        Err(_) => {
283            // Dropping the handle detaches the task; it ends at EOF on its pipe.
284            drop(task);
285            lock().mark_pump_late(
286                generation,
287                format!(
288                    "stderr of the exited process had not reached EOF {bound:?} after it was \
289                     retired (a descendant may still hold the pipe open); lines it still \
290                     writes are kept in that process's section"
291                ),
292            );
293            warn!(
294                module_id,
295                waited = ?bound,
296                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
297            );
298        }
299    }
300}
301
302fn registration_release_events() -> &'static watch::Sender<u64> {
303    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
304    EVENTS.get_or_init(|| {
305        let (sender, _receiver) = watch::channel(0);
306        sender
307    })
308}
309
310pub(crate) fn notify_registration_release() {
311    let events = registration_release_events();
312    let next_generation = (*events.borrow()).wrapping_add(1);
313    events.send_replace(next_generation);
314}
315
316/// How to launch one singleton module process.
317#[derive(Debug, Clone, PartialEq, Eq)]
318pub struct ModuleSpec {
319    pub module_id: String,
320    pub program: PathBuf,
321    pub args: Vec<String>,
322    pub env: Vec<(String, String)>,
323    /// When true this is a reserved module: each spawn gets a fresh one-time launch
324    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
325    /// process can register this module_id (a security-boundary module like the
326    /// credential vault must not be impersonable while it is down/restarting).
327    pub reserved: bool,
328    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
329    /// Prefixes come from daemon config and must end in `:` before they reach the
330    /// supervisor; the owner module's current spawn nonce authorizes claims under
331    /// each prefix.
332    pub reserved_prefixes: Vec<String>,
333    /// The wire protocol this module speaks, as DECLARED in daemon config.
334    ///
335    /// [`ModuleProtocol::None`] changes five things and nothing else: health
336    /// probing is suppressed, teardown sends SIGTERM before waiting,
337    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
338    /// and NO launch nonce, and a clean exit the daemon did not request is
339    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
340    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
341    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
342    /// because a process ignores an environment variable it does not read.
343    ///
344    /// The argument is the part that cannot be "harmless to a process that
345    /// ignores it": a stock binary exits on an unknown flag before it listens
346    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
347    /// first conformance run against this mode found it. The nonce is withheld
348    /// because a process that will never present it gains nothing from holding
349    /// it, and a secret in the environment of a process that does not need it is
350    /// a leak surface for no benefit.
351    pub protocol: ModuleProtocol,
352    /// Whether two processes of this module may run at once, which is what a
353    /// blue/green swap does for the length of its overlap. Declared in daemon
354    /// config because the daemon must be able to answer it while the module is
355    /// down, and so a module cannot talk itself into it after registering.
356    pub overlap: ModuleOverlap,
357}
358
359/// Whether a module tolerates a second process of itself running alongside.
360///
361/// Most modules are single-writer on their store (a WAL, a capture log, a
362/// resident index behind a writer barrier), and two processes on one store
363/// corrupt it. So a swap, which overlaps the old and new process by design,
364/// is refused unless the module's config opts in with `overlap: "safe"`.
365#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
366pub enum ModuleOverlap {
367    /// Never run two processes of this module at once. The default.
368    #[default]
369    Exclusive,
370    /// The module has said a second process of itself is harmless for the
371    /// length of a swap.
372    ///
373    /// Declare it only if a second instance can run for a few seconds without
374    /// touching ANY single-writer store: every database, WAL, index, projector
375    /// and scheduled job the module owns. A lease on part of that state is not
376    /// enough. broca's session lease guards WAL appends while its run index, its
377    /// store projector and its archive fold timer (which unlinks live WAL files)
378    /// stay single-writer, so broca is exclusive despite holding a lease. The
379    /// refusal only fires after this has been decided, so the decision is the
380    /// check.
381    Safe,
382}
383
384impl ModuleOverlap {
385    pub fn as_str(self) -> &'static str {
386        match self {
387            Self::Exclusive => "exclusive",
388            Self::Safe => "safe",
389        }
390    }
391}
392
393/// Environment variable telling a spawned module which case it was started
394/// for, before it sends HELLO. Only a swap candidate carries it, as
395/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
396///
397/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
398/// longer because nobody waits on it, while a plain restart must flip ready
399/// quickly because callers see `module_warming` until it does. Absence means
400/// plain restart, the safe reading. The daemon trusts nothing about it; the
401/// candidate is proven by its launch nonce at HELLO.
402pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
403/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
404pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
405/// How long a swap waits for its candidate to register and declare itself
406/// ready when the operator does not say. A module warming as a swap candidate
407/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
408/// daemon allows that plus time to start the process and send HELLO.
409pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
410
411/// Bounded restart policy for crash exits.
412///
413/// `max_restarts` is the number of replacement processes allowed after the
414/// initial spawn WITHIN `window`. After that many crash restarts inside one
415/// window the module enters [`ModuleState::Failed`] and the supervisor stops
416/// the crash loop.
417///
418/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
419/// and that only survived because crashes were rare: a module that crashed
420/// three times across a week was disabled forever by crashes that had nothing
421/// to do with each other. That stopped being survivable once modules began
422/// exiting non-zero whenever the daemon's connection to them drops, because
423/// then every daemon-side connection drop spends a unit of the same budget and
424/// one flappy hour permanently stops a healthy module. Restarts older than
425/// `window` release their slot, so a module that crashed twice yesterday has a
426/// full budget today, while a genuine crash loop -- which is fast by
427/// definition -- still reaches the cap and stops.
428#[derive(Debug, Clone, Copy, PartialEq, Eq)]
429pub struct RestartPolicy {
430    pub max_restarts: u32,
431    /// Base delay before a crash replacement. The actual delay escalates with
432    /// the number of recent crash replacements and is capped by `max_backoff`.
433    pub backoff: Duration,
434    /// Maximum delay before a crash replacement.
435    pub max_backoff: Duration,
436    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
437    /// budget effectively infinite (nothing is ever in-window), which is why
438    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
439    pub window: Duration,
440}
441
442impl RestartPolicy {
443    /// A policy with the default crash window. Callers that care about the
444    /// window say so with [`Self::with_window`]; the ones that do not are
445    /// asking for the standard rate limit, not for no limit.
446    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
447        Self {
448            max_restarts,
449            backoff,
450            max_backoff: DEFAULT_MAX_BACKOFF,
451            window: DEFAULT_RESTART_WINDOW,
452        }
453    }
454
455    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
456        self.max_backoff = max_backoff;
457        self
458    }
459
460    pub fn with_window(mut self, window: Duration) -> Self {
461        self.window = window;
462        self
463    }
464
465    /// Calculate the capped exponential delay for the next crash replacement.
466    /// `restart_in_window` is zero for the first replacement after an operator
467    /// action (restart, reload, re-enable) cleared the crash ring, or after all
468    /// older crash replacements have aged out of the window.
469    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
470        if self.backoff.is_zero() || self.max_backoff.is_zero() {
471            return Duration::ZERO;
472        }
473
474        let mut delay = self.backoff;
475        for _ in 0..restart_in_window {
476            if delay >= self.max_backoff {
477                return self.max_backoff;
478            }
479            delay = delay
480                .checked_mul(10)
481                .unwrap_or(self.max_backoff)
482                .min(self.max_backoff);
483        }
484        delay.min(self.max_backoff)
485    }
486
487    /// The one sentence that explains a budget-exhausted stop, used for both the
488    /// log line and the terminal record so the two cannot drift. It names the
489    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
490    /// exactly what this budget is not.
491    fn budget_exhausted_detail(&self) -> String {
492        format!(
493            "crash budget exhausted: max_restarts={} within window_secs={}",
494            self.max_restarts,
495            self.window.as_secs()
496        )
497    }
498}
499
500impl Default for RestartPolicy {
501    fn default() -> Self {
502        Self {
503            max_restarts: DEFAULT_MAX_RESTARTS,
504            backoff: DEFAULT_BACKOFF,
505            max_backoff: DEFAULT_MAX_BACKOFF,
506            window: DEFAULT_RESTART_WINDOW,
507        }
508    }
509}
510
511#[derive(Debug, Clone, Copy, PartialEq, Eq)]
512struct CrashRestartSchedule {
513    restart_in_window: u32,
514    delay: Duration,
515}
516
517/// Whether the daemon itself will bring this module back after the exit being
518/// handled: it is enabled AND its in-window crash restarts are below the cap.
519///
520/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
521/// the window are dropped here rather than by a timer, so the count is right
522/// the moment somebody asks and no bookkeeping runs for idle modules.
523fn daemon_will_restart(
524    state: &mut SupervisorSnapshot,
525    policy: &RestartPolicy,
526    now: Instant,
527) -> bool {
528    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
529}
530
531const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
532const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
533const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
534const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
535
536#[derive(Debug, Clone, Copy, PartialEq, Eq)]
537pub enum HealthAction {
538    Report,
539    Restart,
540    Alert,
541}
542
543impl fmt::Display for HealthAction {
544    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
545        f.write_str(match self {
546            Self::Report => "report",
547            Self::Restart => "restart",
548            Self::Alert => "alert",
549        })
550    }
551}
552
553#[derive(Debug, Clone, Copy, PartialEq, Eq)]
554pub struct HealthConfig {
555    pub cadence: Duration,
556    pub deadline: Duration,
557    pub failure_threshold: u32,
558    pub on_degraded: HealthAction,
559    pub on_failing: HealthAction,
560    pub critical: bool,
561}
562
563impl Default for HealthConfig {
564    fn default() -> Self {
565        Self {
566            cadence: DEFAULT_HEALTH_CADENCE,
567            deadline: DEFAULT_HEALTH_DEADLINE,
568            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
569            on_degraded: HealthAction::Report,
570            on_failing: HealthAction::Report,
571            critical: false,
572        }
573    }
574}
575
576/// The supervisor's view of one module's health, relayed to clients over
577/// channel-0 and rendered by `ck health`.
578///
579/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
580/// stated here rather than only at the wire type a consumer reads. A reader can
581/// look up what `None` means; only a writer can silently change it, and the
582/// writer has no reason to go looking at a downstream contract before editing.
583///
584/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
585/// back to `None` on re-registration precisely so a respawned module does not
586/// carry its predecessor's timestamp — so an old value and an absent one call for
587/// opposite readings, and anything that defaulted this to a number would make a
588/// never-probed module indistinguishable from one probed at the epoch.
589///
590/// `detail` and `metrics` are `None` when the module published none on this
591/// probe, which does not mean it reported nothing wrong — it is also the shape
592/// when the probe never reached it. `last_probe_ms` is what separates those.
593#[derive(Debug, Clone, PartialEq)]
594pub struct ModuleHealthStatus {
595    pub status: SupervisorHealthStatus,
596    pub last_probe_ms: Option<u64>,
597    pub detail: Option<String>,
598    pub metrics: Option<Value>,
599    pub consecutive_failures: u32,
600    /// Number of replies received after a recurring health probe's deadline.
601    /// Unlike a timeout, every increment proves the module was alive.
602    pub late_answer_count: u64,
603    /// End-to-end latency of the newest late reply, measured from probe start.
604    pub last_late_answer_latency_ms: Option<u64>,
605    pub last_action: Option<String>,
606    /// Set together with `last_action`; the pair moves as one, and both being
607    /// absent means no escalation has ever been taken rather than that the last
608    /// one succeeded.
609    pub last_action_ms: Option<u64>,
610}
611
612impl Default for ModuleHealthStatus {
613    fn default() -> Self {
614        Self {
615            status: SupervisorHealthStatus::Unknown,
616            last_probe_ms: None,
617            detail: None,
618            metrics: None,
619            consecutive_failures: 0,
620            late_answer_count: 0,
621            last_late_answer_latency_ms: None,
622            last_action: None,
623            last_action_ms: None,
624        }
625    }
626}
627
628/// Typed lifecycle state for a supervised module.
629#[derive(Debug, Clone, Copy, PartialEq, Eq)]
630pub enum ModuleState {
631    Starting,
632    Running,
633    Unresponsive,
634    Restarting,
635    Draining,
636    Stopped,
637    Failed,
638    Disabled,
639}
640
641impl fmt::Display for ModuleState {
642    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
643        f.write_str(match self {
644            Self::Starting => "starting",
645            Self::Running => "running",
646            Self::Unresponsive => "unresponsive",
647            Self::Restarting => "restarting",
648            Self::Draining => "draining",
649            Self::Stopped => "stopped",
650            Self::Failed => "failed",
651            Self::Disabled => "disabled",
652        })
653    }
654}
655
656/// Supervisor classification of a child-process exit.
657#[derive(Debug, Clone, Copy, PartialEq, Eq)]
658pub enum ExitKind {
659    Clean,
660    Crash,
661    DeliberateSeverance,
662}
663
664impl From<ExitKind> for TerminalExitKind {
665    fn from(kind: ExitKind) -> Self {
666        match kind {
667            ExitKind::Clean => Self::Clean,
668            ExitKind::Crash => Self::Crash,
669            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
670        }
671    }
672}
673
674/// Exact process identity retained when a supervised module registers its
675/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
676#[derive(Debug, Clone, Copy, PartialEq, Eq)]
677pub(crate) struct ProcessIdentity {
678    pub(crate) pid: u32,
679    pub(crate) start_time: u64,
680}
681
682/// Last observed child exit, if any.
683#[derive(Debug, Clone, PartialEq, Eq)]
684pub struct ExitReport {
685    pub kind: ExitKind,
686    pub code: Option<i32>,
687    pub signal: Option<i32>,
688    pub at_ms: u64,
689}
690
691/// Point-in-time module status answerable by subc without forwarding to the
692/// module process.
693#[derive(Debug, Clone, PartialEq)]
694pub struct ModuleStatus {
695    pub module_id: String,
696    pub state: ModuleState,
697    pub enabled: bool,
698    pub process_alive: bool,
699    pub registration_active: bool,
700    /// The module's declared wire protocol, carried beside `live` because it is
701    /// what makes `live` readable: the two fields answer one question together.
702    pub protocol: ModuleProtocol,
703    /// Whether the module is serving, under the strongest definition the daemon
704    /// can assert for its protocol.
705    ///
706    /// A subc module must also be REGISTERED: its process being alive says
707    /// nothing about whether it can take a request. A `protocol: "none"` module
708    /// never registers, so that term is dropped and this falls back to "enabled,
709    /// running, and the process the daemon launched is alive" -- which is all
710    /// the daemon observes about a process that speaks no subc wire. It stays a
711    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
712    /// rather than printing it bare.
713    pub live: bool,
714    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
715    /// restarts have already released their slot, so this count can go down
716    /// without anybody touching the module.
717    pub restart_count: u32,
718    /// Replacement processes spawned over this module's entire supervisor lifetime;
719    /// unlike `restart_count`, this value is never reset by an operator action
720    /// and never falls out of a window.
721    pub lifetime_restarts: u32,
722    pub spawn_generation: u64,
723    /// The budget `restart_count` is spent against. Carried alongside the count
724    /// because the count alone does not say how close the module is to being
725    /// disabled, and reporting one without the other is what makes an
726    /// about-to-be-retired module look ordinary.
727    pub max_restarts: u32,
728    /// The span `restart_count` is counted over. Carried with the pair above for
729    /// the same reason they are carried together: "2 of 3" means one thing for a
730    /// ten-minute window and something else entirely for a lifetime.
731    pub restart_window: Duration,
732    /// Effective drain and restart timing policy used by this running module.
733    /// These values are carried together with the restart budget so status
734    /// readers can compare configured intent with what the supervisor applied.
735    pub drain_timeout: Duration,
736    pub restart_backoff: Duration,
737    pub restart_max_backoff: Duration,
738    pub pid: Option<u32>,
739    pub spawned_at_ms: Option<u64>,
740    pub spawned_from: Option<PathBuf>,
741    pub process_start_time: Option<u64>,
742    pub last_exit: Option<ExitReport>,
743    pub health: ModuleHealthStatus,
744}
745
746#[derive(Debug, Clone, PartialEq)]
747struct SupervisorSnapshot {
748    state: ModuleState,
749    enabled: bool,
750    process_alive: bool,
751    /// When each crash restart was spent, oldest first. This IS the crash
752    /// budget: its in-window length is the count an operator sees and the count
753    /// the restart decision is made against, so there is no second counter that
754    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
755    /// operator actions that used to zero the old lifetime counter.
756    crash_restarts: VecDeque<Instant>,
757    lifetime_restarts: u32,
758    /// Successful child spawns in this daemon incarnation.
759    ///
760    /// `lifetime_restarts` was considered and rejected: it starts at zero
761    /// (line 640), successful initial/operator spawns in `set_running` do not
762    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
763    /// increments before a successful replacement exists (lines 604, 3846,
764    /// and 3921), so a failed spawn can consume it. This counter moves only
765    /// when a live PID is accepted below.
766    spawn_generation: u64,
767    pid: Option<u32>,
768    /// Last reaped child, retained after current process facts are cleared.
769    reaped_pid: Option<u32>,
770    /// Whether the command-serving supervision loop has a scheduled respawn.
771    respawn_pending: bool,
772    /// A second restart is waiting for the replacement already scheduled.
773    coalesced_restart_pending: bool,
774    spawned_at_ms: Option<u64>,
775    spawned_from: Option<PathBuf>,
776    spawned_file_identity: Option<SpawnedFileIdentity>,
777    process_start_time: Option<u64>,
778    deliberate_severance: Option<ProcessIdentity>,
779    last_exit: Option<ExitReport>,
780    /// Diagnostic attached to the next drain's terminal record, if any.
781    drain_disposition_detail: Option<String>,
782    health: ModuleHealthStatus,
783    /// Whether the current process was started as a swap candidate and so
784    /// lives in the module's alternate cgroup. The next swap's candidate takes
785    /// the other one, so the two processes of a swap never share a cgroup. A
786    /// plain spawn always uses the primary cgroup.
787    in_alternate_slot: bool,
788    /// Whether the current `Draining` state ends in a replacement process
789    /// (restart, reload, health restart) rather than a stop. Only meaningful
790    /// while `state` is `Draining`; every entry into that state rewrites it.
791    /// It is what lets route.open answer the retryable `module_reloading` to a
792    /// consumer that reaches a still-registered process mid-restart, instead of
793    /// the `supervisor_not_live` a stop or disable deserves.
794    draining_to_replace: bool,
795    /// Whether a configuration update has been applied since the current
796    /// process was spawned, so that process runs an older spec than the one
797    /// the supervisor now holds. A queued restart is only coalesced into a
798    /// fresher process when this is false: a restart requested to pick up a
799    /// new configuration must not be satisfied by a process that predates it.
800    configuration_updated_since_spawn: bool,
801}
802
803impl SupervisorSnapshot {
804    fn starting() -> Self {
805        Self::new(ModuleState::Starting, true)
806    }
807
808    fn disabled() -> Self {
809        Self::new(ModuleState::Disabled, false)
810    }
811
812    fn failed() -> Self {
813        Self::new(ModuleState::Failed, true)
814    }
815
816    /// Crash restarts still inside `window`, having dropped the ones that are
817    /// not. Pruning on read is what makes the budget a rate: an instant older
818    /// than the window stops holding a slot the moment anybody counts.
819    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
820        while let Some(oldest) = self.crash_restarts.front() {
821            if now.duration_since(*oldest) > window {
822                self.crash_restarts.pop_front();
823            } else {
824                break;
825            }
826        }
827        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
828    }
829
830    /// Spend one unit of the crash budget and record the restart in the ledger.
831    ///
832    /// The ring is bounded by the cap because more than `max_restarts` in-window
833    /// instants can never be reached (the caller refuses the restart first), so
834    /// anything beyond that is an unbounded queue waiting to happen.
835    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
836        self.crash_restarts.push_back(now);
837        while self.crash_restarts.len() > policy.max_restarts as usize {
838            self.crash_restarts.pop_front();
839        }
840        self.lifetime_restarts += 1;
841    }
842
843    /// Reserve one crash-restart slot and calculate the delay before respawning.
844    /// The count is captured before recording this restart, so the first retry
845    /// uses the base delay and each later in-window retry escalates once.
846    fn next_crash_restart(
847        &mut self,
848        policy: &RestartPolicy,
849        now: Instant,
850    ) -> Option<CrashRestartSchedule> {
851        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
852        if restart_in_window >= policy.max_restarts {
853            return None;
854        }
855        self.record_crash_restart(policy, now);
856        Some(CrashRestartSchedule {
857            restart_in_window,
858            delay: policy.delay_for_restart(restart_in_window),
859        })
860    }
861
862    /// Give the module its full budget back, as an operator restart, reload, or
863    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
864    /// ledger of what actually happened, and an operator action does not unmake
865    /// the crashes.
866    fn clear_crash_restarts(&mut self) {
867        self.crash_restarts.clear();
868    }
869
870    fn new(state: ModuleState, enabled: bool) -> Self {
871        Self {
872            state,
873            enabled,
874            process_alive: false,
875            crash_restarts: VecDeque::new(),
876            lifetime_restarts: 0,
877            spawn_generation: 0,
878            pid: None,
879            reaped_pid: None,
880            respawn_pending: false,
881            coalesced_restart_pending: false,
882            spawned_at_ms: None,
883            spawned_from: None,
884            spawned_file_identity: None,
885            process_start_time: None,
886            deliberate_severance: None,
887            last_exit: None,
888            drain_disposition_detail: None,
889            health: ModuleHealthStatus::default(),
890            in_alternate_slot: false,
891            draining_to_replace: false,
892            configuration_updated_since_spawn: false,
893        }
894    }
895}
896
897type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
898
899type SpawnSubscriberKey = (ConnectionId, u64);
900
901#[derive(Debug)]
902struct SpawnSubscriber {
903    version: u8,
904    frames: mpsc::Sender<Frame>,
905    /// Tells this subscriber's forwarder that it was dropped for lagging, and
906    /// from which event. The full frame channel cannot carry that news, so it
907    /// travels beside it; see `SpawnEventFeed::subscribe`.
908    lagged: Option<oneshot::Sender<SpawnCursor>>,
909}
910
911#[derive(Debug)]
912struct SpawnEventState {
913    daemon_incarnation: String,
914    seq: u64,
915    capacity: usize,
916    live: HashMap<String, LiveSpawn>,
917    generations: HashMap<String, u64>,
918    events: VecDeque<SpawnEvent>,
919    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
920}
921
922impl Default for SpawnEventState {
923    fn default() -> Self {
924        Self {
925            daemon_incarnation: "unconfigured".to_string(),
926            seq: 0,
927            capacity: SPAWN_EVENT_RING_CAPACITY,
928            live: HashMap::new(),
929            generations: HashMap::new(),
930            events: VecDeque::new(),
931            subscribers: HashMap::new(),
932        }
933    }
934}
935
936#[derive(Debug, Clone, Default)]
937struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
938
939#[derive(Debug, Clone, PartialEq, Eq)]
940pub(crate) enum SpawnSubscribeRefusal {
941    ForeignIncarnation { current: String },
942    TooOld { oldest: SpawnCursor },
943    Frame(String),
944}
945
946impl SpawnEventFeed {
947    fn configure_incarnation(&self, daemon_incarnation: String) {
948        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
949        state.daemon_incarnation = daemon_incarnation;
950        state.seq = 0;
951        state.live.clear();
952        state.generations.clear();
953        state.events.clear();
954        state.subscribers.clear();
955    }
956
957    fn cursor(state: &SpawnEventState) -> SpawnCursor {
958        SpawnCursor {
959            daemon_incarnation: state.daemon_incarnation.clone(),
960            seq: state.seq,
961        }
962    }
963
964    fn snapshot(&self) -> SpawnSnapshot {
965        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
966        let mut live = state.live.values().cloned().collect::<Vec<_>>();
967        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
968        SpawnSnapshot {
969            cursor: Self::cursor(&state),
970            ring_bound: state.capacity as u64,
971            live,
972        }
973    }
974
975    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
976        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
977        let generation = state
978            .generations
979            .get(module_id)
980            .copied()
981            .unwrap_or(0)
982            .checked_add(1)
983            .expect("spawn generation exhausted");
984        state.generations.insert(module_id.to_string(), generation);
985        let live = LiveSpawn {
986            module_id: module_id.to_string(),
987            spawn_generation: generation,
988            pid,
989            spawned_at_ms,
990        };
991        state.live.insert(module_id.to_string(), live);
992        Self::emit_locked(
993            &mut state,
994            SpawnEventKind::Spawned,
995            module_id.to_string(),
996            generation,
997            pid,
998            None,
999            None,
1000        );
1001        generation
1002    }
1003
1004    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1005        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1006        let Some(live) = state.live.remove(module_id) else {
1007            warn!(
1008                module_id,
1009                "terminal record had no live spawn event identity"
1010            );
1011            return;
1012        };
1013        Self::emit_locked(
1014            &mut state,
1015            SpawnEventKind::Exited,
1016            module_id.to_string(),
1017            live.spawn_generation,
1018            live.pid,
1019            exit_code,
1020            exit_signal,
1021        );
1022    }
1023
1024    /// Report the exit of a process that a swap has already replaced.
1025    ///
1026    /// `emit_exited` removes the module's live entry, which after a swap's
1027    /// cutover describes the promoted candidate, not the old process now
1028    /// exiting. This emits the old generation's exit and leaves the live entry
1029    /// alone unless it still names that generation.
1030    fn emit_superseded_exited(
1031        &self,
1032        module_id: &str,
1033        spawn_generation: u64,
1034        pid: u32,
1035        exit_code: Option<i32>,
1036        exit_signal: Option<i32>,
1037    ) {
1038        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1039        if state
1040            .live
1041            .get(module_id)
1042            .is_some_and(|live| live.spawn_generation == spawn_generation)
1043        {
1044            state.live.remove(module_id);
1045        }
1046        Self::emit_locked(
1047            &mut state,
1048            SpawnEventKind::Exited,
1049            module_id.to_string(),
1050            spawn_generation,
1051            pid,
1052            exit_code,
1053            exit_signal,
1054        );
1055    }
1056
1057    #[allow(clippy::too_many_arguments)]
1058    fn emit_locked(
1059        state: &mut SpawnEventState,
1060        kind: SpawnEventKind,
1061        module_id: String,
1062        spawn_generation: u64,
1063        pid: u32,
1064        exit_code: Option<i32>,
1065        exit_signal: Option<i32>,
1066    ) {
1067        state.seq = state
1068            .seq
1069            .checked_add(1)
1070            .expect("spawn event sequence exhausted");
1071        let event = SpawnEvent {
1072            cursor: Self::cursor(state),
1073            kind,
1074            module_id,
1075            spawn_generation,
1076            pid,
1077            exit_code,
1078            exit_signal,
1079        };
1080        state.events.push_back(event.clone());
1081        while state.events.len() > state.capacity {
1082            state.events.pop_front();
1083        }
1084        let body = match serde_json::to_vec(&event) {
1085            Ok(body) => body,
1086            Err(error) => {
1087                error!(%error, "failed to serialize supervisor spawn event");
1088                return;
1089            }
1090        };
1091        state.subscribers.retain(|(connection_id, corr), subscriber| {
1092            let frame = Frame::build_with_version(
1093                subscriber.version,
1094                FrameType::StreamData,
1095                control_flags(),
1096                0,
1097                0,
1098                *corr,
1099                body.clone(),
1100            );
1101            match frame {
1102                Ok(frame) => {
1103                    if subscriber.frames.try_send(frame).is_ok() {
1104                        true
1105                    } else {
1106                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1107                        if let Some(lagged) = subscriber.lagged.take() {
1108                            let _ = lagged.send(event.cursor.clone());
1109                        }
1110                        false
1111                    }
1112                }
1113                Err(error) => {
1114                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1115                    false
1116                }
1117            }
1118        });
1119    }
1120
1121    fn subscribe(
1122        &self,
1123        connection_id: ConnectionId,
1124        corr: u64,
1125        version: u8,
1126        since: Option<SpawnCursor>,
1127        sink: FrameSink,
1128    ) -> Result<(), SpawnSubscribeRefusal> {
1129        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1130        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1131        {
1132            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1133            let replay = if let Some(since) = since {
1134                if since.daemon_incarnation != state.daemon_incarnation {
1135                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1136                        current: state.daemon_incarnation.clone(),
1137                    });
1138                }
1139                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1140                    if since.seq < oldest.seq.saturating_sub(1) {
1141                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1142                    }
1143                }
1144                state
1145                    .events
1146                    .iter()
1147                    .filter(|event| event.cursor.seq > since.seq)
1148                    .cloned()
1149                    .collect::<Vec<_>>()
1150            } else {
1151                Vec::new()
1152            };
1153            for event in replay {
1154                let body = serde_json::to_vec(&event)
1155                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1156                let frame = Frame::build_with_version(
1157                    version,
1158                    FrameType::StreamData,
1159                    control_flags(),
1160                    0,
1161                    0,
1162                    corr,
1163                    body,
1164                )
1165                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1166                frames
1167                    .try_send(frame)
1168                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1169            }
1170            state.subscribers.insert(
1171                (connection_id, corr),
1172                SpawnSubscriber {
1173                    version,
1174                    frames,
1175                    lagged: Some(lagged),
1176                },
1177            );
1178        }
1179        // The lagged terminal is sent here, by the forwarder, rather than by
1180        // the emitter: at the moment of the drop the subscriber's own channel
1181        // is full, and writing to the connection sink directly from the emitter
1182        // would put the Error AHEAD of the events still queued in that channel
1183        // (and the emitter holds the feed lock, so it cannot await the sink).
1184        // Dropping the subscriber drops the only sender, so `recv` drains every
1185        // queued event and then returns `None`; only then is the Error sent, so
1186        // the client sees each event it can keep, then the reason it was cut.
1187        // Cancel and connection removal drop the oneshot unsent, so they end
1188        // the stream with no Error.
1189        tokio::spawn(async move {
1190            while let Some(frame) = receiver.recv().await {
1191                if sink.send(frame).await.is_err() {
1192                    return;
1193                }
1194            }
1195            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1196                return;
1197            };
1198            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1199                Ok(frame) => {
1200                    let _ = sink.send(frame).await;
1201                }
1202                Err(error) => {
1203                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1204                }
1205            }
1206        });
1207        Ok(())
1208    }
1209
1210    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1211        let Some(subscriber) = self
1212            .0
1213            .lock()
1214            .unwrap_or_else(|p| p.into_inner())
1215            .subscribers
1216            .remove(&(connection_id, corr))
1217        else {
1218            return false;
1219        };
1220        if let Ok(frame) = Frame::build_with_version(
1221            subscriber.version,
1222            FrameType::StreamEnd,
1223            control_flags(),
1224            0,
1225            0,
1226            corr,
1227            Vec::new(),
1228        ) {
1229            tokio::spawn(async move {
1230                let _ = subscriber.frames.send(frame).await;
1231            });
1232        }
1233        true
1234    }
1235
1236    fn remove_connection(&self, connection_id: ConnectionId) {
1237        self.0
1238            .lock()
1239            .unwrap_or_else(|p| p.into_inner())
1240            .subscribers
1241            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1242    }
1243
1244    #[cfg(any(test, feature = "test-support"))]
1245    fn set_capacity(&self, capacity: usize) {
1246        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1247    }
1248
1249    #[cfg(any(test, feature = "test-support"))]
1250    fn subscriber_count(&self) -> usize {
1251        self.0
1252            .lock()
1253            .unwrap_or_else(|p| p.into_inner())
1254            .subscribers
1255            .len()
1256    }
1257}
1258
1259/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1260/// The terminal Error a lagged spawn subscriber receives after its queued events.
1261fn spawn_subscriber_lagged_frame(
1262    version: u8,
1263    corr: u64,
1264    first_undelivered: SpawnCursor,
1265) -> Result<Frame, String> {
1266    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1267        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1268        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1269            .to_string(),
1270        detail: Some(serde_json::json!({
1271            "first_undelivered_cursor": first_undelivered
1272        })),
1273    })
1274    .map_err(|error| error.to_string())?;
1275    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1276        .map_err(|error| error.to_string())
1277}
1278
1279pub trait ModuleProcessLiveness: Send + Sync {
1280    fn process_live(&self, module_id: &str) -> Option<bool>;
1281
1282    /// Whether the supervisor is replacing this module's process right now: an
1283    /// operator restart or reload, a health restart, or a crash respawn whose
1284    /// backoff is running. A module in that state is not live, but a consumer
1285    /// refused now should retry shortly rather than treat the target as gone.
1286    /// Stopped, failed, and disabled modules are not replacing.
1287    fn process_replacing(&self, _module_id: &str) -> bool {
1288        false
1289    }
1290}
1291
1292/// Shared process-liveness registry keyed by supervised `module_id`.
1293#[derive(Debug, Clone, Default)]
1294pub struct SupervisorProcessLiveness {
1295    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1296}
1297
1298impl SupervisorProcessLiveness {
1299    pub fn new() -> Self {
1300        Self::default()
1301    }
1302
1303    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1304        let mut snapshots = self
1305            .snapshots
1306            .lock()
1307            .unwrap_or_else(|poisoned| poisoned.into_inner());
1308        snapshots.insert(module_id, snapshot);
1309    }
1310
1311    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1312        let mut snapshots = self
1313            .snapshots
1314            .lock()
1315            .unwrap_or_else(|poisoned| poisoned.into_inner());
1316        let is_current = snapshots
1317            .get(module_id)
1318            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1319            .unwrap_or(false);
1320        if is_current {
1321            snapshots.remove(module_id);
1322        }
1323    }
1324}
1325
1326impl ModuleProcessLiveness for SupervisorProcessLiveness {
1327    fn process_live(&self, module_id: &str) -> Option<bool> {
1328        let snapshot = {
1329            let snapshots = self
1330                .snapshots
1331                .lock()
1332                .unwrap_or_else(|poisoned| poisoned.into_inner());
1333            snapshots.get(module_id).cloned()
1334        }?;
1335        let snapshot = snapshot
1336            .lock()
1337            .unwrap_or_else(|poisoned| poisoned.into_inner());
1338        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1339    }
1340
1341    fn process_replacing(&self, module_id: &str) -> bool {
1342        let Some(snapshot) = self
1343            .snapshots
1344            .lock()
1345            .unwrap_or_else(|poisoned| poisoned.into_inner())
1346            .get(module_id)
1347            .cloned()
1348        else {
1349            return false;
1350        };
1351        let snapshot = snapshot
1352            .lock()
1353            .unwrap_or_else(|poisoned| poisoned.into_inner());
1354        snapshot.enabled
1355            && match snapshot.state {
1356                ModuleState::Restarting => true,
1357                ModuleState::Draining => snapshot.draining_to_replace,
1358                ModuleState::Starting
1359                | ModuleState::Running
1360                | ModuleState::Unresponsive
1361                | ModuleState::Stopped
1362                | ModuleState::Failed
1363                | ModuleState::Disabled => false,
1364            }
1365    }
1366}
1367
1368#[cfg(test)]
1369#[derive(Debug, Default)]
1370struct ReloadExitRecordGate {
1371    reached: tokio::sync::Notify,
1372    resume: tokio::sync::Notify,
1373}
1374
1375#[derive(Debug, Clone, Copy)]
1376enum RespawnKind {
1377    Spawn,
1378    Reload,
1379}
1380
1381#[derive(Debug, Clone, Copy)]
1382struct PendingRespawn {
1383    deadline: Instant,
1384    kind: RespawnKind,
1385}
1386
1387type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1388
1389#[derive(Debug, Clone)]
1390struct SupervisorRuntimeConfig {
1391    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1392    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1393    /// A reload acknowledges completion only after its replacement registers.
1394    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1395    restart_policy: RestartPolicy,
1396    /// This module's RESOLVED drain budget: per-module config when present,
1397    /// else `default_drain_timeout`.
1398    drain_timeout: Duration,
1399    /// Shared with the status handle so the attested value changes atomically
1400    /// when a rescan updates the running drain policy.
1401    effective_drain_timeout: Arc<Mutex<Duration>>,
1402    /// The supervisor-wide fallback, kept so a configuration update that
1403    /// REMOVES the per-module override can re-resolve to it.
1404    default_drain_timeout: Duration,
1405    health: HealthConfig,
1406    connection_file_path: Option<PathBuf>,
1407    capture_logs_dir: Option<PathBuf>,
1408    forwarding: Option<Arc<ForwardingTable>>,
1409    /// The shared handle, so every spawn path (initial, restart, reload) records the
1410    /// reserved-module launch nonce the HELLO verifier checks against.
1411    supervisor_handle: Option<SupervisorHandle>,
1412    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1413    /// status queries.
1414    ///
1415    /// One ring per module, held across every respawn. The lines explaining an exit
1416    /// are written BEFORE that exit, so a ring recreated per process would be empty
1417    /// exactly when it is asked for.
1418    stderr_ring: Arc<Mutex<StderrRing>>,
1419    terminal_ring: Arc<Mutex<TerminalRing>>,
1420    spawn_events: SpawnEventFeed,
1421    child_roster: ChildRoster,
1422    #[cfg(target_os = "linux")]
1423    cgroup_placement: Option<subc_cgroup::Placement>,
1424    #[cfg(test)]
1425    test_seed_stale_facts_before_enable_spawn: bool,
1426    #[cfg(test)]
1427    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1428}
1429
1430#[derive(Debug, Clone, PartialEq, Eq)]
1431struct SupervisedConfiguration {
1432    spec: ModuleSpec,
1433    health: HealthConfig,
1434}
1435
1436/// Shared daemon lookup table for supervised module handles.
1437///
1438/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1439/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1440/// launch nonces recorded at spawn are checked by the same daemon instance.
1441#[derive(Debug, Clone, Default)]
1442pub struct SupervisorHandle {
1443    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1444    spawn_events: SpawnEventFeed,
1445    /// The current expected launch nonce for each reserved module_id. Set when the
1446    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1447    /// non-reserved module never has an entry here and is never nonce-checked.
1448    /// Reserved module ids and the nonce that authorizes their next HELLO.
1449    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1450    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1451    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1452    /// had NO entry and admitted anyone: the reservation protected the nonce
1453    /// holder, not the NAME (found live by CKCRED's canary probe registering
1454    /// against a reserved scratch id).
1455    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1456    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1457    ///
1458    /// This is deliberately in-memory only: subc is state-free across daemon
1459    /// restarts, and the tombstone only explains the hours-after-removal window
1460    /// while this executing daemon is still alive. Do not persist it in a store.
1461    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1462    /// The current launch nonce for every supervised spawn. This is separate from
1463    /// reserved_nonces because consumer route.open attestation applies to all spawned
1464    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1465    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1466    /// Reserved namespace prefixes mapped to the supervised owner module whose
1467    /// current spawn nonce authorizes HELLO claims below the prefix.
1468    ///
1469    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1470    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1471    /// accidental collisions and lower-trust processes from squatting protected
1472    /// namespaces.
1473    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1474    /// Blue/green swaps in progress, by module id. An entry exists from just
1475    /// before the candidate process is spawned until the swap has failed, or
1476    /// has cut over and the old process is gone. While it exists, HELLO for the
1477    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1478    /// consumer attestation accepts both processes' nonces.
1479    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1480    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1481    promotion_observer: PromotionObserverSlot,
1482    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1483    /// this daemon-wide ordering, a rescan could retire or update a module while a
1484    /// concurrent reload still held its old handle and launch specification.
1485    operation_lock: Arc<AsyncMutex<()>>,
1486}
1487
1488/// Told when a swap has promoted its candidate to be the module's active
1489/// registration.
1490///
1491/// An ordinary HELLO runs the control plane's registration side effects (the
1492/// capability cache, the deny census, the requirement recompute) as it
1493/// registers. A swap candidate's HELLO does not, because it is not routable;
1494/// promotion is when those must run instead, and promotion happens in the
1495/// supervisor, which has no other way into the control handler.
1496pub(crate) trait SwapPromotionObserver: Send + Sync {
1497    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1498}
1499
1500/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1501/// control handler) owns this handle, so a strong reference back would be a
1502/// cycle that keeps both alive.
1503#[derive(Clone, Default)]
1504struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1505
1506impl fmt::Debug for PromotionObserverSlot {
1507    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1508        f.write_str("PromotionObserverSlot")
1509    }
1510}
1511
1512/// The nonces of one open swap.
1513#[derive(Debug, Clone)]
1514struct OpenSwap {
1515    /// The launch nonce minted for the candidate process. It is the swap
1516    /// token: the only thing that admits a HELLO into the candidate slot.
1517    candidate_nonce: String,
1518    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1519    /// here because cutover moves the module's recorded spawn nonce to the
1520    /// candidate while the incumbent is still draining and its consumers are
1521    /// still attesting with this one.
1522    incumbent_nonce: Option<String>,
1523    /// Set once a HELLO has been admitted with the swap token, so the token
1524    /// admits one registration and cannot be replayed after cutover empties
1525    /// the candidate slot.
1526    candidate_admitted: bool,
1527}
1528
1529/// What the swap gate says about a HELLO. See
1530/// [`SupervisorHandle::swap_hello_admission`].
1531#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1532pub(crate) enum SwapHelloAdmission {
1533    /// No swap is open for the id (or the HELLO carries the incumbent's own
1534    /// nonce); the ordinary gates decide.
1535    NotSwapping,
1536    /// The HELLO carries the swap token: register it into the candidate slot.
1537    Candidate,
1538    /// A swap is open and the HELLO carries a nonce the supervisor did not
1539    /// mint for this id, no nonce, or a token already used.
1540    Refused,
1541}
1542
1543#[derive(Debug, Clone, PartialEq, Eq)]
1544pub(crate) enum ReservedHelloRejection {
1545    Exact {
1546        module_id: String,
1547    },
1548    Prefix {
1549        prefix: String,
1550        owner_module_id: String,
1551    },
1552}
1553
1554impl SupervisorHandle {
1555    pub fn new() -> Self {
1556        Self::default()
1557    }
1558
1559    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1560        self.spawn_events.snapshot()
1561    }
1562
1563    pub(crate) fn subscribe_spawns(
1564        &self,
1565        connection_id: ConnectionId,
1566        corr: u64,
1567        version: u8,
1568        since: Option<SpawnCursor>,
1569        sink: FrameSink,
1570    ) -> Result<(), SpawnSubscribeRefusal> {
1571        self.spawn_events
1572            .subscribe(connection_id, corr, version, since, sink)
1573    }
1574
1575    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1576        self.spawn_events.cancel(connection_id, corr)
1577    }
1578
1579    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1580        self.spawn_events.remove_connection(connection_id);
1581    }
1582
1583    #[cfg(any(test, feature = "test-support"))]
1584    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1585        assert!(capacity > 0, "spawn event capacity must be non-zero");
1586        self.spawn_events.set_capacity(capacity);
1587    }
1588
1589    #[cfg(any(test, feature = "test-support"))]
1590    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1591        self.spawn_events.subscriber_count()
1592    }
1593
1594    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1595    /// a respawn invalidates stale consumer identities.
1596    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1597        self.spawn_nonces
1598            .lock()
1599            .unwrap_or_else(|poisoned| poisoned.into_inner())
1600            .insert(module_id.to_string(), nonce);
1601    }
1602
1603    /// Record the launch nonce expected from the next HELLO for a reserved module,
1604    /// replacing any prior nonce (a respawn invalidates the previous one).
1605    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1606        self.reserved_nonces
1607            .lock()
1608            .unwrap_or_else(|poisoned| poisoned.into_inner())
1609            .insert(module_id.to_string(), Some(nonce));
1610    }
1611
1612    /// Record namespace prefixes owned by a supervised module.
1613    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1614        let mut owners = self
1615            .reserved_prefix_owners
1616            .lock()
1617            .unwrap_or_else(|poisoned| poisoned.into_inner());
1618        owners.retain(|_, owner| owner != owner_module_id);
1619        for prefix in prefixes {
1620            owners.insert(prefix.clone(), owner_module_id.to_string());
1621        }
1622    }
1623
1624    /// The launch nonce most recently minted for a module's spawn, if any.
1625    #[cfg(test)]
1626    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1627        self.spawn_nonces
1628            .lock()
1629            .unwrap_or_else(|poisoned| poisoned.into_inner())
1630            .get(module_id)
1631            .cloned()
1632    }
1633
1634    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1635        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1636        let spawn_nonce = self
1637            .spawn_nonces
1638            .lock()
1639            .unwrap_or_else(|poisoned| poisoned.into_inner())
1640            .get(&spec.module_id)
1641            .cloned();
1642        let mut reserved_nonces = self
1643            .reserved_nonces
1644            .lock()
1645            .unwrap_or_else(|poisoned| poisoned.into_inner());
1646        if spec.reserved {
1647            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1648            // reserved name whose module has never spawned has no legitimate
1649            // holder, and the entry's absence is what used to leave the name
1650            // open to the first claimant.
1651            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1652        }
1653        drop(reserved_nonces);
1654        // A later unreserved declaration must not silently unreserve an id that
1655        // was retained after its reserved configuration was removed. The explicit
1656        // release ceremony is the only operation that retires that gate.
1657        self.removal_tombstones
1658            .lock()
1659            .unwrap_or_else(|poisoned| poisoned.into_inner())
1660            .remove(&spec.module_id);
1661    }
1662
1663    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
1664    /// authorized only by its expected nonce; otherwise a matching reserved prefix
1665    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
1666    /// with no matching prefix are always authorized.
1667    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1668        self.reserved_hello_rejection(module_id, presented)
1669            .is_none()
1670    }
1671
1672    pub(crate) fn reserved_hello_rejection(
1673        &self,
1674        module_id: &str,
1675        presented: Option<&str>,
1676    ) -> Option<ReservedHelloRejection> {
1677        let nonces = self
1678            .reserved_nonces
1679            .lock()
1680            .unwrap_or_else(|poisoned| poisoned.into_inner());
1681        if let Some(expected) = nonces.get(module_id) {
1682            // `None` = reserved with no legitimate holder: refuse every
1683            // presentation, because no process can hold a nonce that was never
1684            // minted. Only a real minted nonce admits, in constant time.
1685            let authorized = match expected {
1686                Some(expected) => {
1687                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1688                }
1689                None => false,
1690            };
1691            if authorized {
1692                return None;
1693            }
1694            return Some(ReservedHelloRejection::Exact {
1695                module_id: module_id.to_string(),
1696            });
1697        }
1698        drop(nonces);
1699
1700        let matched_prefix = self
1701            .reserved_prefix_owners
1702            .lock()
1703            .unwrap_or_else(|poisoned| poisoned.into_inner())
1704            .iter()
1705            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
1706            .max_by_key(|(prefix, _)| prefix.len())
1707            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
1708        let (prefix, owner_module_id) = matched_prefix?;
1709
1710        let authorized = presented.is_some_and(|presented| {
1711            self.spawn_nonces
1712                .lock()
1713                .unwrap_or_else(|poisoned| poisoned.into_inner())
1714                .get(&owner_module_id)
1715                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
1716                // While the owner is being swapped, children started by
1717                // either of its two processes hold that process's nonce.
1718                || self.swap_nonce_matches(&owner_module_id, presented)
1719        });
1720        if authorized {
1721            None
1722        } else {
1723            Some(ReservedHelloRejection::Prefix {
1724                prefix,
1725                owner_module_id,
1726            })
1727        }
1728    }
1729
1730    /// Whether a consumer connection proved it came from a daemon-spawned module.
1731    ///
1732    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
1733    /// accepted only for module ids the supervisor has spawned.
1734    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
1735        if presented.is_empty() {
1736            return false;
1737        }
1738        let nonces = self
1739            .spawn_nonces
1740            .lock()
1741            .unwrap_or_else(|poisoned| poisoned.into_inner());
1742        let current = nonces
1743            .get(module_id)
1744            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
1745        drop(nonces);
1746        // During a swap two processes of the module are alive, and a consumer
1747        // started by either one presents that process's nonce. Accepting only
1748        // the recorded one would fail the incumbent's consumers for the whole
1749        // overlap once cutover moves the record to the candidate.
1750        current || self.swap_nonce_matches(module_id, presented)
1751    }
1752
1753    /// Whether `presented` is either nonce of an open swap for `module_id`.
1754    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
1755        let swaps = self
1756            .swaps
1757            .lock()
1758            .unwrap_or_else(|poisoned| poisoned.into_inner());
1759        swaps.get(module_id).is_some_and(|swap| {
1760            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
1761                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
1762                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
1763                })
1764        })
1765    }
1766
1767    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
1768    /// Called before the candidate process exists.
1769    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
1770        let incumbent_nonce = self
1771            .spawn_nonces
1772            .lock()
1773            .unwrap_or_else(|poisoned| poisoned.into_inner())
1774            .get(module_id)
1775            .cloned();
1776        self.swaps
1777            .lock()
1778            .unwrap_or_else(|poisoned| poisoned.into_inner())
1779            .insert(
1780                module_id.to_string(),
1781                OpenSwap {
1782                    candidate_nonce,
1783                    incumbent_nonce,
1784                    candidate_admitted: false,
1785                },
1786            );
1787    }
1788
1789    /// Close the swap for `module_id`, releasing whichever nonce is no longer
1790    /// the module's recorded one.
1791    pub(crate) fn close_swap(&self, module_id: &str) {
1792        self.swaps
1793            .lock()
1794            .unwrap_or_else(|poisoned| poisoned.into_inner())
1795            .remove(module_id);
1796    }
1797
1798    /// Install the observer told about swap promotions, replacing any earlier
1799    /// one.
1800    pub(crate) fn set_swap_promotion_observer(
1801        &self,
1802        observer: std::sync::Weak<dyn SwapPromotionObserver>,
1803    ) {
1804        *self
1805            .promotion_observer
1806            .0
1807            .lock()
1808            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
1809    }
1810
1811    /// Tell the installed observer, if it is still alive, that a swap promoted
1812    /// `registration`.
1813    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
1814        let observer = self
1815            .promotion_observer
1816            .0
1817            .lock()
1818            .unwrap_or_else(|poisoned| poisoned.into_inner())
1819            .as_ref()
1820            .and_then(std::sync::Weak::upgrade);
1821        if let Some(observer) = observer {
1822            observer.swap_promoted(registration);
1823        }
1824    }
1825
1826    /// Whether a swap is open for `module_id`.
1827    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
1828        self.swaps
1829            .lock()
1830            .unwrap_or_else(|poisoned| poisoned.into_inner())
1831            .contains_key(module_id)
1832    }
1833
1834    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
1835    /// respawn would, once cutover has made the candidate the module's process.
1836    /// The swap stays open so the incumbent's nonce keeps attesting until the
1837    /// incumbent has drained and exited.
1838    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
1839        let candidate_nonce = self
1840            .swaps
1841            .lock()
1842            .unwrap_or_else(|poisoned| poisoned.into_inner())
1843            .get(module_id)
1844            .map(|swap| swap.candidate_nonce.clone());
1845        let Some(nonce) = candidate_nonce else {
1846            return;
1847        };
1848        self.set_spawn_nonce(module_id, nonce.clone());
1849        if reserved {
1850            self.set_reserved_nonce(module_id, nonce);
1851        }
1852    }
1853
1854    /// The swap gate for a HELLO claiming `module_id`.
1855    ///
1856    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
1857    /// presents the candidate nonce, which the reserved gate (holding the
1858    /// incumbent's nonce) would refuse as `reserved_module` before swap
1859    /// admission was ever reached. And it applies to unreserved ids too: for an
1860    /// unreserved id the only thing that ever stopped a second process claiming
1861    /// a live id was the `duplicate_module_id` refusal, which is exactly the
1862    /// refusal a swap lifts for its candidate.
1863    ///
1864    /// The incumbent's own nonce falls through to the ordinary gates, which
1865    /// treat it as they always have (a live incumbent is refused as a
1866    /// duplicate). Anything else while a swap is open is refused, including an
1867    /// absent nonce.
1868    pub(crate) fn swap_hello_admission(
1869        &self,
1870        module_id: &str,
1871        presented: Option<&str>,
1872    ) -> SwapHelloAdmission {
1873        let swaps = self
1874            .swaps
1875            .lock()
1876            .unwrap_or_else(|poisoned| poisoned.into_inner());
1877        let Some(swap) = swaps.get(module_id) else {
1878            return SwapHelloAdmission::NotSwapping;
1879        };
1880        let Some(presented) = presented else {
1881            return SwapHelloAdmission::Refused;
1882        };
1883        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
1884            return if swap.candidate_admitted {
1885                SwapHelloAdmission::Refused
1886            } else {
1887                SwapHelloAdmission::Candidate
1888            };
1889        }
1890        if swap
1891            .incumbent_nonce
1892            .as_deref()
1893            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
1894        {
1895            return SwapHelloAdmission::NotSwapping;
1896        }
1897        SwapHelloAdmission::Refused
1898    }
1899
1900    /// Record that the swap token has registered a candidate, so it admits no
1901    /// second HELLO.
1902    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
1903        if let Some(swap) = self
1904            .swaps
1905            .lock()
1906            .unwrap_or_else(|poisoned| poisoned.into_inner())
1907            .get_mut(module_id)
1908        {
1909            swap.candidate_admitted = true;
1910        }
1911    }
1912
1913    /// Test/support lookup for the current launch nonce of a supervised spawn.
1914    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1915        self.spawn_nonces
1916            .lock()
1917            .unwrap_or_else(|poisoned| poisoned.into_inner())
1918            .get(module_id)
1919            .cloned()
1920    }
1921
1922    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
1923    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1924        self.reserved_nonces
1925            .lock()
1926            .unwrap_or_else(|poisoned| poisoned.into_inner())
1927            .get(module_id)
1928            .cloned()
1929            .flatten()
1930    }
1931
1932    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
1933        let mut modules = self
1934            .modules
1935            .lock()
1936            .unwrap_or_else(|poisoned| poisoned.into_inner());
1937        modules.insert(module.module_id().to_string(), module)
1938    }
1939
1940    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
1941        let modules = self
1942            .modules
1943            .lock()
1944            .unwrap_or_else(|poisoned| poisoned.into_inner());
1945        modules.get(module_id).cloned()
1946    }
1947
1948    pub(crate) fn record_late_health_answer(
1949        &self,
1950        module_id: &str,
1951        latency_ms: u64,
1952    ) -> Result<bool, SuperviseError> {
1953        let Some(module) = self.get(module_id) else {
1954            return Ok(false);
1955        };
1956        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
1957            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
1958            state.health.last_late_answer_latency_ms = Some(latency_ms);
1959            // A late answer is an answer: the module served the probe, just past
1960            // the deadline. Leaving the miss streak in place while logging
1961            // "proves the module is alive" is how a CPU-starved module that
1962            // answers every probe a few seconds late still marches to the
1963            // threshold and gets killed — the exact kill class `NoAnswer` is
1964            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
1965            // is degradation, and degradation reports; it does not restart.
1966            state.health.consecutive_failures = 0;
1967        })?;
1968        Ok(true)
1969    }
1970
1971    /// Arm the one-shot marker for the module process that this caller
1972    /// deliberately initiated severance against. Generic connection teardown
1973    /// must not call this:
1974    /// a surviving process would otherwise retain an exemption for a later
1975    /// genuine crash.
1976    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
1977        let Some(module) = self.get(module_id) else {
1978            return Ok(false);
1979        };
1980        let status = module.status()?;
1981        let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
1982            return Ok(false);
1983        };
1984        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
1985    }
1986
1987    pub fn list(&self) -> Vec<SupervisedModule> {
1988        let modules = self
1989            .modules
1990            .lock()
1991            .unwrap_or_else(|poisoned| poisoned.into_inner());
1992        let mut modules = modules.values().cloned().collect::<Vec<_>>();
1993        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
1994        modules
1995    }
1996
1997    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
1998        self.spawn_nonces
1999            .lock()
2000            .unwrap_or_else(|poisoned| poisoned.into_inner())
2001            .remove(module_id);
2002        self.close_swap(module_id);
2003        let mut reserved_nonces = self
2004            .reserved_nonces
2005            .lock()
2006            .unwrap_or_else(|poisoned| poisoned.into_inner());
2007        if reserved_nonces.contains_key(module_id) {
2008            // The old nonce must die with the removed process, but the exact-id
2009            // gate remains until an operator explicitly releases it.
2010            reserved_nonces.insert(module_id.to_string(), None);
2011        }
2012        drop(reserved_nonces);
2013        self.reserved_prefix_owners
2014            .lock()
2015            .unwrap_or_else(|poisoned| poisoned.into_inner())
2016            .retain(|_, owner| owner != module_id);
2017        self.modules
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner())
2020            .remove(module_id)
2021    }
2022
2023    /// Remember a module removed by a non-preview rescan so route.open can
2024    /// distinguish that intentional removal from an unknown id.
2025    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2026        self.removal_tombstones
2027            .lock()
2028            .unwrap_or_else(|poisoned| poisoned.into_inner())
2029            .insert(module_id.to_string(), unix_ms_now());
2030    }
2031
2032    /// Return how long ago a rescan removed this module in milliseconds.
2033    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2034        self.removal_tombstones
2035            .lock()
2036            .unwrap_or_else(|poisoned| poisoned.into_inner())
2037            .get(module_id)
2038            .copied()
2039            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2040    }
2041
2042    /// Retire a reserved-id gate only after its module has left supervision.
2043    ///
2044    /// A retained gate has no live nonce (`None`), so releasing any other entry
2045    /// would weaken a currently configured or otherwise active reservation.
2046    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2047        if self.get(module_id).is_some() {
2048            return false;
2049        }
2050        let mut reserved_nonces = self
2051            .reserved_nonces
2052            .lock()
2053            .unwrap_or_else(|poisoned| poisoned.into_inner());
2054        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2055            return false;
2056        }
2057        reserved_nonces.remove(module_id);
2058        true
2059    }
2060
2061    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2062        Arc::clone(&self.operation_lock)
2063    }
2064}
2065
2066/// Process supervisor for subc-owned singleton modules.
2067#[derive(Debug, Clone)]
2068pub struct Supervisor {
2069    registry: Arc<Registry>,
2070    restart_policy: RestartPolicy,
2071    drain_timeout: Duration,
2072    connection_file_path: Option<PathBuf>,
2073    capture_logs_dir: Option<PathBuf>,
2074    forwarding: Option<Arc<ForwardingTable>>,
2075    process_liveness: Arc<SupervisorProcessLiveness>,
2076    supervisor_handle: Option<SupervisorHandle>,
2077    health: HealthConfig,
2078    daemon_start_clock: crate::clock::StartClock,
2079    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2080    spawn_events: SpawnEventFeed,
2081    provenance_probe: ExecutableIdentityProbe,
2082    /// Every process spawned through this supervisor (and its clones) and not
2083    /// yet reaped, so daemon shutdown can end them.
2084    child_roster: ChildRoster,
2085    #[cfg(target_os = "linux")]
2086    cgroup_placement: Option<subc_cgroup::Placement>,
2087}
2088
2089impl Supervisor {
2090    /// The first step of an announced daemon shutdown, before the notice and
2091    /// before any connection is closed.
2092    ///
2093    /// Sets the daemon-shutdown flag first: from here on no module is
2094    /// respawned (crash restart, operator restart, or swap), and every child
2095    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2096    /// the module exits on the EOF this shutdown gives it or is signalled by a
2097    /// service manager that kills the whole cgroup. Then writes the journal's
2098    /// shutdown marker, which records the instant and closes this daemon
2099    /// incarnation's stretch of the journal.
2100    #[cfg(unix)]
2101    pub(crate) fn begin_daemon_shutdown(&self) {
2102        self.child_roster.close();
2103        if let Some(journal) = &self.terminal_journal {
2104            journal.stamp_shutdown();
2105        }
2106    }
2107
2108    /// Announce a cut while established connections can still carry replies.
2109    /// These budgets promise notice and a bounded wait, not child completion;
2110    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2111    #[cfg(unix)]
2112    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2113        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2114        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2115        let Some(forwarding) = &self.forwarding else {
2116            return Ok(());
2117        };
2118        let module_ids = forwarding
2119            .begin_daemon_drain()
2120            .map_err(SuperviseError::Forwarding)?;
2121        let deadline_ms =
2122            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2123        let mut notices = tokio::task::JoinSet::new();
2124        let mut drains = Vec::new();
2125        for module_id in module_ids {
2126            let Some(target) = forwarding
2127                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2128                .map_err(SuperviseError::Forwarding)?
2129            else {
2130                continue;
2131            };
2132            let routes = forwarding
2133                .endpoint_routes(target.endpoint)
2134                .map_err(SuperviseError::Forwarding)?;
2135            // Restart allows deployed consumers to reopen after the new daemon
2136            // appears. The wire reason stays `restart`; what tells a daemon cut
2137            // apart from a module restart afterwards is the terminal record
2138            // itself, whose disposition is `daemon_shutdown` for every exit
2139            // observed once `begin_daemon_shutdown` has run.
2140            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2141                reason: RouteCloseReason::Restart,
2142                deadline_ms,
2143            })
2144            .expect("module draining serializes");
2145            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2146            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2147            for route in routes {
2148                let client = route.goodbye_target;
2149                if let Some((_, channels)) = clients
2150                    .iter_mut()
2151                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2152                {
2153                    channels.push(client.channel);
2154                } else {
2155                    let channel = client.channel;
2156                    clients.push((client, vec![channel]));
2157                }
2158            }
2159            for (client, mut channels) in clients {
2160                channels.sort_unstable();
2161                channels.dedup();
2162                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2163                    module_id: module_id.clone(),
2164                    channels,
2165                    reason: RouteCloseReason::Restart,
2166                })
2167                .expect("route closing serializes");
2168                recipients.push((client.sink, client.negotiated_ver, closing));
2169            }
2170            for (sink, version, body) in recipients {
2171                notices.spawn(async move {
2172                    let frame = Frame::build_with_version(
2173                        version,
2174                        FrameType::Push,
2175                        control_flags(),
2176                        0,
2177                        0,
2178                        0,
2179                        body,
2180                    )
2181                    .expect("bounded lifecycle notice frame builds");
2182                    sink.send_flushed(frame).await
2183                });
2184            }
2185            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2186            drains.push((module_id, target.endpoint, gauges));
2187        }
2188        // A quiet forwarding table is not proof that queued notices reached the
2189        // socket. Wait for writer flush acknowledgements before testing quiescence.
2190        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2191        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2192            if !matches!(result, Ok(Ok(()))) {
2193                warn!(?result, "daemon shutdown notice delivery failed");
2194            }
2195        }
2196        notices.abort_all();
2197        let deadline = Instant::now() + DRAIN_BUDGET;
2198        let mut waits = tokio::task::JoinSet::new();
2199        for (module_id, endpoint, gauges) in drains {
2200            let forwarding = Arc::clone(forwarding);
2201            let mut runtime = self.runtime_config();
2202            runtime.health.cadence = Duration::from_millis(100);
2203            waits.spawn(async move {
2204                wait_for_forwarding_quiescence(
2205                    &forwarding,
2206                    &module_id,
2207                    &runtime,
2208                    endpoint,
2209                    deadline,
2210                    &gauges,
2211                    DrainScope::Active,
2212                )
2213                .await
2214            });
2215        }
2216        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2217            if !matches!(result, Ok(Ok(true))) {
2218                warn!(?result, "daemon shutdown drain did not reach quiescence");
2219            }
2220        }
2221        Ok(())
2222    }
2223
2224    /// The last step of an announced daemon shutdown, after the notice and the
2225    /// drain: send every registered module a module GOODBYE, the same planned
2226    /// stop signal `ck module stop` gives, then close every connection so each
2227    /// subc module sees EOF and starts its own teardown, then end every
2228    /// supervised child that has not exited
2229    /// by its own deadline (its drain budget, capped). Modules lead their own
2230    /// process groups, so a
2231    /// service manager's group kill no longer reaches them; without this a
2232    /// child that does not stop on EOF (every `protocol: "none"` child, which
2233    /// has no connection) would outlive the daemon. Every wait is bounded (see
2234    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2235    #[cfg(unix)]
2236    pub(crate) async fn end_children_for_daemon_shutdown(
2237        &self,
2238        already_escalated: bool,
2239        escalate: impl std::future::Future<Output = ()>,
2240    ) {
2241        tokio::pin!(escalate);
2242        let mut escalated = already_escalated;
2243        if let Some(forwarding) = &self.forwarding {
2244            let reason = CloseReason::new(
2245                "daemon_shutdown",
2246                "the daemon is exiting after its shutdown notice and drain",
2247            );
2248            if escalated {
2249                // The operator asked to stop waiting: queue the GOODBYEs but
2250                // do not wait for them to be written.
2251                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2252            } else {
2253                tokio::select! {
2254                    biased;
2255                    _ = escalate.as_mut() => {
2256                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2257                        escalated = true;
2258                    }
2259                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2260                }
2261            }
2262            let closed = forwarding.close_all_connections(&reason);
2263            debug!(closed, "closed established connections for daemon shutdown");
2264        }
2265        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2266        // already completed and must not be polled again; the child shutdown
2267        // wait is told it is escalated and gets a future that never fires.
2268        let escalated_here = escalated && !already_escalated;
2269        let remaining_escalate = async move {
2270            if escalated_here {
2271                std::future::pending::<()>().await;
2272            } else {
2273                escalate.await;
2274            }
2275        };
2276        crate::child_roster::end_children_for_daemon_shutdown(
2277            &self.child_roster,
2278            escalated,
2279            remaining_escalate,
2280        )
2281        .await;
2282    }
2283
2284    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2285        Self {
2286            registry,
2287            restart_policy,
2288            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2289            connection_file_path: None,
2290            capture_logs_dir: None,
2291            forwarding: None,
2292            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2293            supervisor_handle: None,
2294            health: HealthConfig::default(),
2295            daemon_start_clock: crate::clock::StartClock::capture(),
2296            terminal_journal: None,
2297            spawn_events: SpawnEventFeed::default(),
2298            provenance_probe: ExecutableIdentityProbe::default(),
2299            child_roster: ChildRoster::default(),
2300            #[cfg(target_os = "linux")]
2301            cgroup_placement: None,
2302        }
2303    }
2304
2305    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2306        self.drain_timeout = drain_timeout;
2307        self
2308    }
2309
2310    pub fn with_process_liveness(
2311        mut self,
2312        process_liveness: Arc<SupervisorProcessLiveness>,
2313    ) -> Self {
2314        self.process_liveness = process_liveness;
2315        self
2316    }
2317
2318    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2319        self.connection_file_path = Some(connection_file_path.into());
2320        self
2321    }
2322
2323    /// Enables daemon-owned capture files for supervised stdout and stderr.
2324    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2325        self.capture_logs_dir = Some(logs_dir.into());
2326        self
2327    }
2328
2329    /// Names this daemon lifetime in spawn events, independently of whether a
2330    /// terminal journal is configured.
2331    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2332        // A millisecond start stamp can repeat after clock rollback or a rapid
2333        // restart. Use the connection file's random daemon_id instead: it already
2334        // identifies this daemon lifetime independently of the wall clock.
2335        self.spawn_events.configure_incarnation(daemon_incarnation);
2336        self
2337    }
2338
2339    /// Enables best-effort history shared by every supervised module. Without
2340    /// it, terminal history is kept only in each module's in-memory ring.
2341    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2342        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2343        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2344            path,
2345            daemon_incarnation,
2346        )));
2347        this
2348    }
2349
2350    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2351        self.forwarding = Some(forwarding);
2352        self
2353    }
2354
2355    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2356        self.spawn_events = supervisor_handle.spawn_events.clone();
2357        self.supervisor_handle = Some(supervisor_handle);
2358        self
2359    }
2360
2361    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2362        self.health = health;
2363        self
2364    }
2365
2366    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2367    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2368    /// record is kept.
2369    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2370        self.child_roster.record_to(path.into());
2371        self
2372    }
2373
2374    #[cfg(target_os = "linux")]
2375    pub fn with_cgroup_placement(
2376        mut self,
2377        cgroup_placement: Option<subc_cgroup::Placement>,
2378    ) -> Self {
2379        self.cgroup_placement = cgroup_placement;
2380        self
2381    }
2382
2383    /// Spawn `spec.program` and start monitoring it.
2384    ///
2385    /// The child is expected to parse `--subc <connection-file-path>`, read the
2386    /// TCP+key connection file, authenticate to the already-running listener, and
2387    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2388    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2389        validate_spec(&spec)?;
2390
2391        let runtime = self.runtime_config();
2392        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2393        let child = spawn_child(
2394            &spec,
2395            runtime.connection_file_path.as_deref(),
2396            self.supervisor_handle.as_ref(),
2397            &runtime.stderr_ring,
2398            runtime.capture_logs_dir.as_deref(),
2399            &runtime.child_roster,
2400            #[cfg(target_os = "linux")]
2401            runtime.cgroup_placement.as_ref(),
2402        )?;
2403        set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2404        self.process_liveness
2405            .track(spec.module_id.clone(), Arc::clone(&snapshot));
2406
2407        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2408    }
2409
2410    /// Start supervising a module declared in daemon configuration.
2411    ///
2412    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2413    /// failures in the supervisor handle so operator-facing `supervisor.list`
2414    /// reflects every configured module while daemon startup continues.
2415    pub fn supervise_configured(
2416        &self,
2417        spec: ModuleSpec,
2418        enabled: bool,
2419    ) -> Result<SupervisedModule, SuperviseError> {
2420        validate_spec(&spec)?;
2421
2422        let runtime = self.runtime_config();
2423        if !enabled {
2424            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2425            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2426        }
2427
2428        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2429        match spawn_child(
2430            &spec,
2431            runtime.connection_file_path.as_deref(),
2432            self.supervisor_handle.as_ref(),
2433            &runtime.stderr_ring,
2434            runtime.capture_logs_dir.as_deref(),
2435            &runtime.child_roster,
2436            #[cfg(target_os = "linux")]
2437            runtime.cgroup_placement.as_ref(),
2438        ) {
2439            Ok(child) => {
2440                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2441                self.process_liveness
2442                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2443                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2444            }
2445            Err(err) => {
2446                error!(
2447                    module_id = %spec.module_id,
2448                    program = %spec.program.display(),
2449                    error = %err,
2450                    "configured module failed to spawn; marking failed and continuing"
2451                );
2452                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2453                Ok(self.supervised_module(spec, runtime, snapshot, None))
2454            }
2455        }
2456    }
2457
2458    /// Supervise a configured module with its own health, drain, and crash
2459    /// budget. The restart policy is per-module because the config file is:
2460    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2461    /// module that is expensive to restart should not be forced onto the same
2462    /// budget as one that is cheap.
2463    pub fn supervise_configured_with_health(
2464        &self,
2465        spec: ModuleSpec,
2466        enabled: bool,
2467        health: HealthConfig,
2468        drain_timeout_ms: Option<u64>,
2469        restart_policy: RestartPolicy,
2470    ) -> Result<SupervisedModule, SuperviseError> {
2471        validate_spec(&spec)?;
2472
2473        let mut runtime = self.runtime_config();
2474        runtime.health = health;
2475        runtime.restart_policy = restart_policy;
2476        if let Some(ms) = drain_timeout_ms {
2477            runtime.drain_timeout = Duration::from_millis(ms);
2478            *runtime
2479                .effective_drain_timeout
2480                .lock()
2481                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2482        }
2483        if !enabled {
2484            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2485            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2486        }
2487
2488        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2489        match spawn_child(
2490            &spec,
2491            runtime.connection_file_path.as_deref(),
2492            self.supervisor_handle.as_ref(),
2493            &runtime.stderr_ring,
2494            runtime.capture_logs_dir.as_deref(),
2495            &runtime.child_roster,
2496            #[cfg(target_os = "linux")]
2497            runtime.cgroup_placement.as_ref(),
2498        ) {
2499            Ok(child) => {
2500                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2501                self.process_liveness
2502                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2503                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2504            }
2505            Err(err) => {
2506                if health.critical {
2507                    error!(
2508                        module_id = %spec.module_id,
2509                        program = %spec.program.display(),
2510                        error = %err,
2511                        "critical configured module failed to spawn; marking failed and alerting"
2512                    );
2513                } else {
2514                    error!(
2515                        module_id = %spec.module_id,
2516                        program = %spec.program.display(),
2517                        error = %err,
2518                        "configured module failed to spawn; marking failed and continuing"
2519                    );
2520                }
2521                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2522                Ok(self.supervised_module(spec, runtime, snapshot, None))
2523            }
2524        }
2525    }
2526
2527    fn runtime_config(&self) -> SupervisorRuntimeConfig {
2528        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2529        SupervisorRuntimeConfig {
2530            scheduled_respawn: Arc::default(),
2531            deferred_reload_reply: Arc::default(),
2532            restart_policy: self.restart_policy,
2533            drain_timeout: self.drain_timeout,
2534            // Shared with this module's roster copy: daemon shutdown waits on
2535            // each child for the module's own drain budget, as resolved now.
2536            child_roster: self
2537                .child_roster
2538                .for_module(Arc::clone(&effective_drain_timeout)),
2539            effective_drain_timeout,
2540            default_drain_timeout: self.drain_timeout,
2541            health: self.health,
2542            connection_file_path: self.connection_file_path.clone(),
2543            capture_logs_dir: self.capture_logs_dir.clone(),
2544            forwarding: self.forwarding.clone(),
2545            supervisor_handle: self.supervisor_handle.clone(),
2546            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
2547            terminal_ring: Arc::new(Mutex::new(
2548                TerminalRing::new(
2549                    TerminalRingConfig::default(),
2550                    self.daemon_start_clock.started_at_ms(),
2551                )
2552                .with_start_clock(self.daemon_start_clock)
2553                .with_journal(self.terminal_journal.clone())
2554                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
2555            )),
2556            spawn_events: self.spawn_events.clone(),
2557            #[cfg(target_os = "linux")]
2558            cgroup_placement: self.cgroup_placement.clone(),
2559            #[cfg(test)]
2560            test_seed_stale_facts_before_enable_spawn: false,
2561            #[cfg(test)]
2562            test_reload_exit_record_gate: None,
2563        }
2564    }
2565
2566    fn supervised_module(
2567        &self,
2568        spec: ModuleSpec,
2569        runtime: SupervisorRuntimeConfig,
2570        snapshot: SharedSnapshot,
2571        child: Option<SupervisedChild>,
2572    ) -> SupervisedModule {
2573        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
2574            spec: spec.clone(),
2575            health: runtime.health,
2576        }));
2577        let stderr_ring = Arc::clone(&runtime.stderr_ring);
2578        let terminal_ring = Arc::clone(&runtime.terminal_ring);
2579        // The module's OWN policy, which may be its per-module config rather than
2580        // the supervisor-wide one; status must report the budget the supervise
2581        // loop actually enforces.
2582        let restart_policy = runtime.restart_policy;
2583        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
2584        let (tx, rx) = mpsc::channel(4);
2585        let monitor = tokio::spawn(supervise_loop(
2586            spec.clone(),
2587            runtime,
2588            Arc::clone(&self.registry),
2589            Arc::clone(&self.process_liveness),
2590            Arc::clone(&snapshot),
2591            child,
2592            rx,
2593        ));
2594
2595        let module_id = spec.module_id.clone();
2596        let module = SupervisedModule {
2597            inner: Arc::new(SupervisedModuleInner {
2598                module_id: module_id.clone(),
2599                registry: Arc::clone(&self.registry),
2600                snapshot,
2601                configuration,
2602                stderr_ring,
2603                terminal_ring,
2604                commands: tx,
2605                monitor: Mutex::new(Some(monitor)),
2606                restart_policy,
2607                effective_drain_timeout,
2608                provenance_probe: self.provenance_probe.clone(),
2609            }),
2610        };
2611        if let Some(supervisor_handle) = &self.supervisor_handle {
2612            supervisor_handle.apply_identity_configuration(&spec);
2613            supervisor_handle.insert(module.clone());
2614        }
2615        module
2616    }
2617}
2618
2619impl Default for Supervisor {
2620    fn default() -> Self {
2621        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
2622    }
2623}
2624
2625/// Handle to one supervised child process.
2626#[derive(Clone)]
2627pub struct SupervisedModule {
2628    inner: Arc<SupervisedModuleInner>,
2629}
2630
2631struct SupervisedModuleInner {
2632    module_id: String,
2633    registry: Arc<Registry>,
2634    snapshot: SharedSnapshot,
2635    configuration: Arc<Mutex<SupervisedConfiguration>>,
2636    stderr_ring: Arc<Mutex<StderrRing>>,
2637    terminal_ring: Arc<Mutex<TerminalRing>>,
2638    commands: mpsc::Sender<SupervisorCommand>,
2639    monitor: Mutex<Option<JoinHandle<()>>>,
2640    /// Copied from the supervisor's runtime config at spawn so `status()` can
2641    /// report the restart budget without reaching back into the supervisor. The
2642    /// policy is fixed for the process's lifetime, so a copy cannot drift.
2643    restart_policy: RestartPolicy,
2644    effective_drain_timeout: Arc<Mutex<Duration>>,
2645    provenance_probe: ExecutableIdentityProbe,
2646}
2647
2648impl fmt::Debug for SupervisedModule {
2649    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2650        f.debug_struct("SupervisedModule")
2651            .field("module_id", &self.inner.module_id)
2652            .field("status", &self.status())
2653            .finish_non_exhaustive()
2654    }
2655}
2656
2657impl SupervisedModule {
2658    pub fn module_id(&self) -> &str {
2659        &self.inner.module_id
2660    }
2661
2662    /// Test-only: put one probe miss on the streak, the way
2663    /// `handle_health_probe_failure` does, so tests can assert what a later
2664    /// event does to the streak without driving the whole probe loop.
2665    #[cfg(test)]
2666    pub(crate) fn record_health_probe_failure_for_test(
2667        &self,
2668        detail: &str,
2669    ) -> Result<(), SuperviseError> {
2670        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
2671            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
2672            state.health.detail = Some(detail.to_string());
2673        })
2674    }
2675
2676    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
2677        Ok(lock_snapshot(&self.inner.snapshot)?.state)
2678    }
2679
2680    /// The module's retained stderr, newest lines last.
2681    ///
2682    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
2683    /// module, `supervisor.list` renders every module, and putting it in the
2684    /// shared snapshot would make each status read carry a payload almost nobody
2685    /// asked for. Callers that want the text ask for it.
2686    pub fn stderr_tail(
2687        &self,
2688        max_lines: Option<usize>,
2689        max_bytes: Option<usize>,
2690    ) -> StderrTailSnapshot {
2691        self.inner
2692            .stderr_ring
2693            .lock()
2694            .unwrap_or_else(|poisoned| poisoned.into_inner())
2695            .snapshot(max_lines, max_bytes)
2696    }
2697
2698    /// The module's bounded terminal history, oldest retained exit first.
2699    ///
2700    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
2701    /// daemon whose in-memory history was necessarily reset.
2702    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
2703        self.inner
2704            .terminal_ring
2705            .lock()
2706            .unwrap_or_else(|poisoned| poisoned.into_inner())
2707            .snapshot()
2708    }
2709
2710    /// Retained observations from the current ring and all journal generations.
2711    ///
2712    /// Blocking: this reads the journal files. Async callers use
2713    /// [`Self::read_durable_terminal_history`].
2714    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
2715        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
2716    }
2717
2718    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
2719    /// read (up to every retained generation) never occupies a runtime worker.
2720    /// Fails only if the blocking task could not finish (runtime shutdown or a
2721    /// panic in the read).
2722    pub(crate) async fn read_durable_terminal_history(
2723        &self,
2724    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
2725        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
2726        let module_id = self.inner.module_id.clone();
2727        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
2728            .await
2729    }
2730
2731    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
2732        self.status_with_snapshot_lock(&self.inner.snapshot, None)
2733    }
2734
2735    pub(crate) fn record_deliberate_severance(
2736        &self,
2737        identity: ProcessIdentity,
2738    ) -> Result<bool, SuperviseError> {
2739        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2740        if snapshot.pid != Some(identity.pid)
2741            || snapshot.process_start_time != Some(identity.start_time)
2742        {
2743            return Ok(false);
2744        }
2745        snapshot.deliberate_severance = Some(identity);
2746        Ok(true)
2747    }
2748
2749    /// Read status for a channel-0 renderer and report a contended snapshot lock.
2750    ///
2751    /// Internal supervision callers use [`Self::status`] so writer-side machinery
2752    /// does not produce reader-observability logs.
2753    pub(crate) fn status_for_control(
2754        &self,
2755        caller: &'static str,
2756    ) -> Result<ModuleStatus, SuperviseError> {
2757        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
2758    }
2759
2760    fn status_with_snapshot_lock(
2761        &self,
2762        snapshot: &SharedSnapshot,
2763        caller: Option<&'static str>,
2764    ) -> Result<ModuleStatus, SuperviseError> {
2765        let mut guard = match caller {
2766            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
2767            None => lock_snapshot(snapshot)?,
2768        };
2769        // Read the budget through the pruning path so a reader sees the same
2770        // in-window count the restart decision would use, not a stale total.
2771        let restart_count =
2772            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
2773        let snapshot = guard.clone();
2774        drop(guard);
2775        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
2776            SuperviseError::StatePoisoned {
2777                module_id: Some(self.inner.module_id.clone()),
2778            }
2779        })?;
2780        let registration_active = self
2781            .inner
2782            .registry
2783            .get_module(&self.inner.module_id)
2784            .map_err(SuperviseError::Registry)?
2785            .is_some();
2786        let protocol = self.declared_protocol()?;
2787        let running_process =
2788            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
2789        // Registration is the difference between the two protocols and the only
2790        // one: a subc module that has not registered cannot serve a request even
2791        // though its process is up, and a `none` module never registers at all,
2792        // so requiring it there would pin `live` to false for the whole life of
2793        // a perfectly healthy process.
2794        let live = match protocol {
2795            ModuleProtocol::Subc => running_process && registration_active,
2796            ModuleProtocol::None => running_process,
2797        };
2798
2799        Ok(ModuleStatus {
2800            module_id: self.inner.module_id.clone(),
2801            state: snapshot.state,
2802            enabled: snapshot.enabled,
2803            process_alive: snapshot.process_alive,
2804            registration_active,
2805            protocol,
2806            live,
2807            restart_count,
2808            lifetime_restarts: snapshot.lifetime_restarts,
2809            spawn_generation: snapshot.spawn_generation,
2810            max_restarts: self.inner.restart_policy.max_restarts,
2811            restart_window: self.inner.restart_policy.window,
2812            drain_timeout,
2813            restart_backoff: self.inner.restart_policy.backoff,
2814            restart_max_backoff: self.inner.restart_policy.max_backoff,
2815            pid: snapshot.pid,
2816            spawned_at_ms: snapshot.spawned_at_ms,
2817            spawned_from: snapshot.spawned_from,
2818            process_start_time: snapshot.process_start_time,
2819            last_exit: snapshot.last_exit,
2820            health: snapshot.health,
2821        })
2822    }
2823
2824    #[cfg(test)]
2825    pub(crate) fn hold_snapshot_for_test(
2826        &self,
2827        acquired: std::sync::mpsc::Sender<()>,
2828        hold: Duration,
2829    ) -> std::thread::JoinHandle<()> {
2830        let snapshot = Arc::clone(&self.inner.snapshot);
2831        std::thread::spawn(move || {
2832            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
2833            acquired
2834                .send(())
2835                .expect("test receiver waits for snapshot lock");
2836            std::thread::sleep(hold);
2837        })
2838    }
2839
2840    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
2841        let snapshot = match lock_snapshot(&self.inner.snapshot) {
2842            Ok(snapshot) => snapshot.clone(),
2843            Err(_) => {
2844                return subc_control::RunningImageAgreement::Unavailable {
2845                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
2846                };
2847            }
2848        };
2849        self.inner
2850            .provenance_probe
2851            .observe(
2852                snapshot.pid,
2853                snapshot.spawned_from.as_deref(),
2854                snapshot.spawned_file_identity,
2855                snapshot.process_start_time,
2856            )
2857            .await
2858    }
2859
2860    /// Memory and CPU time of the module's current process, read now. Only the
2861    /// process the supervisor spawned is read, not processes it has started.
2862    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
2863        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
2864            Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
2865            Err(_) => {
2866                return subc_control::ChildResourceUsage::Unavailable {
2867                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
2868                }
2869            }
2870        };
2871        crate::child_resources::read(pid, start_time)
2872    }
2873
2874    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
2875        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2876        Ok(match snapshot.state {
2877            ModuleState::Restarting => true,
2878            ModuleState::Failed | ModuleState::Disabled => false,
2879            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
2880        })
2881    }
2882
2883    #[cfg(test)]
2884    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
2885        self.is_warming_with_snapshot_lock(None)
2886    }
2887
2888    pub(crate) fn is_warming_for_control(
2889        &self,
2890        caller: &'static str,
2891    ) -> Result<bool, SuperviseError> {
2892        self.is_warming_with_snapshot_lock(Some(caller))
2893    }
2894
2895    fn is_warming_with_snapshot_lock(
2896        &self,
2897        caller: Option<&'static str>,
2898    ) -> Result<bool, SuperviseError> {
2899        let snapshot = match caller {
2900            Some(caller) => {
2901                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
2902            }
2903            None => lock_snapshot(&self.inner.snapshot)?,
2904        }
2905        .clone();
2906        Ok(matches!(
2907            snapshot.state,
2908            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
2909        ))
2910    }
2911
2912    /// Drain the module and stop monitoring it.
2913    pub async fn drain(&self) -> Result<(), SuperviseError> {
2914        self.stop().await
2915    }
2916
2917    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
2918        match self.state()? {
2919            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2920            ModuleState::Starting
2921            | ModuleState::Running
2922            | ModuleState::Unresponsive
2923            | ModuleState::Restarting
2924            | ModuleState::Draining
2925            | ModuleState::Disabled => {}
2926        }
2927
2928        let (reply_tx, reply_rx) = oneshot::channel();
2929        self.inner
2930            .commands
2931            .send(SupervisorCommand::Retire { reply: reply_tx })
2932            .await
2933            .map_err(|_| SuperviseError::CommandClosed {
2934                module_id: self.inner.module_id.clone(),
2935            })?;
2936        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2937            module_id: self.inner.module_id.clone(),
2938        })?
2939    }
2940
2941    pub async fn stop(&self) -> Result<(), SuperviseError> {
2942        match self.state()? {
2943            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2944            ModuleState::Starting
2945            | ModuleState::Running
2946            | ModuleState::Unresponsive
2947            | ModuleState::Restarting
2948            | ModuleState::Draining
2949            | ModuleState::Disabled => {}
2950        }
2951
2952        let (reply_tx, reply_rx) = oneshot::channel();
2953        self.inner
2954            .commands
2955            .send(SupervisorCommand::Drain { reply: reply_tx })
2956            .await
2957            .map_err(|_| SuperviseError::CommandClosed {
2958                module_id: self.inner.module_id.clone(),
2959            })?;
2960        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2961            module_id: self.inner.module_id.clone(),
2962        })?
2963    }
2964
2965    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
2966        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
2967        let (reply_tx, reply_rx) = oneshot::channel();
2968        self.inner
2969            .commands
2970            .send(SupervisorCommand::Restart {
2971                drain_timeout_ms,
2972                received_at_generation,
2973                queued_at: Instant::now(),
2974                reply: reply_tx,
2975            })
2976            .await
2977            .map_err(|_| SuperviseError::CommandClosed {
2978                module_id: self.inner.module_id.clone(),
2979            })?;
2980        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2981            module_id: self.inner.module_id.clone(),
2982        })?
2983    }
2984
2985    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
2986    /// `supervisor_swap` module. Returns once the swap has cut over (the old
2987    /// process then drains in the background of the supervise loop) or has
2988    /// failed, leaving the old process serving.
2989    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
2990        let (reply_tx, reply_rx) = oneshot::channel();
2991        self.inner
2992            .commands
2993            .send(SupervisorCommand::Swap {
2994                ready_timeout,
2995                reply: reply_tx,
2996            })
2997            .await
2998            .map_err(|_| SuperviseError::CommandClosed {
2999                module_id: self.inner.module_id.clone(),
3000            })?;
3001        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3002            module_id: self.inner.module_id.clone(),
3003        })?
3004    }
3005
3006    pub async fn reload(&self) -> Result<(), SuperviseError> {
3007        let (reply_tx, reply_rx) = oneshot::channel();
3008        self.inner
3009            .commands
3010            .send(SupervisorCommand::Reload { reply: reply_tx })
3011            .await
3012            .map_err(|_| SuperviseError::CommandClosed {
3013                module_id: self.inner.module_id.clone(),
3014            })?;
3015        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3016            module_id: self.inner.module_id.clone(),
3017        })?
3018    }
3019
3020    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3021        let (reply_tx, reply_rx) = oneshot::channel();
3022        self.inner
3023            .commands
3024            .send(SupervisorCommand::SetEnabled {
3025                enabled,
3026                reply: reply_tx,
3027            })
3028            .await
3029            .map_err(|_| SuperviseError::CommandClosed {
3030                module_id: self.inner.module_id.clone(),
3031            })?;
3032        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3033            module_id: self.inner.module_id.clone(),
3034        })?
3035    }
3036
3037    /// This module's declared protocol, read from the same stored configuration
3038    /// the rescan diff compares and `update_configuration` rewrites, so a status
3039    /// read and the supervise loop can never disagree about which protocol is in
3040    /// force.
3041    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3042        Ok(self
3043            .inner
3044            .configuration
3045            .lock()
3046            .map_err(|_| SuperviseError::StatePoisoned {
3047                module_id: Some(self.inner.module_id.clone()),
3048            })?
3049            .spec
3050            .protocol)
3051    }
3052
3053    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3054        let configuration =
3055            self.inner
3056                .configuration
3057                .lock()
3058                .map_err(|_| SuperviseError::StatePoisoned {
3059                    module_id: Some(self.inner.module_id.clone()),
3060                })?;
3061        Ok((configuration.spec.clone(), configuration.health))
3062    }
3063
3064    /// Replace this module's launch spec, keeping its health and drain policy,
3065    /// the way a rescan does for a changed config entry. The running process is
3066    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3067    #[cfg(any(test, feature = "test-support"))]
3068    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3069        let (_, health) = self.configuration()?;
3070        let drain_timeout_ms = u64::try_from(
3071            self.inner
3072                .effective_drain_timeout
3073                .lock()
3074                .unwrap_or_else(|poisoned| poisoned.into_inner())
3075                .as_millis(),
3076        )
3077        .ok();
3078        self.update_configuration(spec, health, drain_timeout_ms)
3079            .await
3080    }
3081
3082    pub(crate) async fn update_configuration(
3083        &self,
3084        spec: ModuleSpec,
3085        health: HealthConfig,
3086        drain_timeout_ms: Option<u64>,
3087    ) -> Result<(), SuperviseError> {
3088        if spec.module_id != self.inner.module_id {
3089            return Err(SuperviseError::InvalidSpec {
3090                reason: "a supervised module's module_id cannot be changed".to_string(),
3091            });
3092        }
3093        validate_spec(&spec)?;
3094        let (reply_tx, reply_rx) = oneshot::channel();
3095        self.inner
3096            .commands
3097            .send(SupervisorCommand::UpdateConfiguration {
3098                spec: spec.clone(),
3099                health,
3100                drain_timeout_ms,
3101                reply: reply_tx,
3102            })
3103            .await
3104            .map_err(|_| SuperviseError::CommandClosed {
3105                module_id: self.inner.module_id.clone(),
3106            })?;
3107        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3108            module_id: self.inner.module_id.clone(),
3109        })?;
3110        let mut configuration =
3111            self.inner
3112                .configuration
3113                .lock()
3114                .map_err(|_| SuperviseError::StatePoisoned {
3115                    module_id: Some(self.inner.module_id.clone()),
3116                })?;
3117        configuration.spec = spec;
3118        configuration.health = health;
3119        Ok(())
3120    }
3121}
3122
3123impl Drop for SupervisedModuleInner {
3124    fn drop(&mut self) {
3125        let Ok(mut monitor) = self.monitor.lock() else {
3126            return;
3127        };
3128        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3129            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3130                state.state = ModuleState::Stopped;
3131                clear_current_process_facts(state);
3132            });
3133            monitor.abort();
3134        }
3135        let _ = monitor.take();
3136    }
3137}
3138
3139#[derive(Debug)]
3140enum SupervisorCommand {
3141    Drain {
3142        reply: oneshot::Sender<Result<(), SuperviseError>>,
3143    },
3144    Retire {
3145        reply: oneshot::Sender<Result<(), SuperviseError>>,
3146    },
3147    Restart {
3148        /// Operator override for this one restart's drain budget, in ms. `None`
3149        /// uses the module's configured/default budget; `Some(0)` cuts
3150        /// immediately (wedge bounce: a stuck request never settles, so
3151        /// waiting only delays recovery).
3152        drain_timeout_ms: Option<u64>,
3153        /// The module's `spawn_generation` when the request was received, before
3154        /// it waited in the command queue. A queued restart whose module has
3155        /// since spawned a newer process is already satisfied (see the handler).
3156        received_at_generation: u64,
3157        /// When the request entered the command queue, so the handler can log
3158        /// how long it waited behind the loop's other work.
3159        queued_at: Instant,
3160        reply: oneshot::Sender<Result<(), SuperviseError>>,
3161    },
3162    Reload {
3163        reply: oneshot::Sender<Result<(), SuperviseError>>,
3164    },
3165    SetEnabled {
3166        enabled: bool,
3167        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3168    },
3169    UpdateConfiguration {
3170        spec: ModuleSpec,
3171        health: HealthConfig,
3172        /// Per-module drain override from the new config; `None` re-resolves to
3173        /// the supervisor-wide default.
3174        drain_timeout_ms: Option<u64>,
3175        reply: oneshot::Sender<()>,
3176    },
3177    Swap {
3178        /// How long the candidate may take to register and declare itself
3179        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3180        ready_timeout: Option<Duration>,
3181        /// Answered at cutover or failure; the incumbent's drain follows.
3182        reply: oneshot::Sender<Result<(), SuperviseError>>,
3183    },
3184}
3185
3186#[derive(Debug)]
3187pub enum SuperviseError {
3188    InvalidSpec {
3189        reason: String,
3190    },
3191    Spawn {
3192        program: PathBuf,
3193        source: io::Error,
3194        cgroup_path: Option<PathBuf>,
3195    },
3196    Cgroup {
3197        module_id: String,
3198        source: io::Error,
3199    },
3200    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3201    /// than spawn a reserved module without its identity binding.
3202    LaunchNonce {
3203        reason: String,
3204    },
3205    Wait {
3206        module_id: String,
3207        source: io::Error,
3208    },
3209    Kill {
3210        module_id: String,
3211        source: io::Error,
3212    },
3213    Forwarding(ForwardingError),
3214    Registry(RegistryError),
3215    ReloadUnavailable {
3216        module_id: String,
3217        reason: String,
3218    },
3219    /// An operator restart/reload was requested for a module that is currently
3220    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3221    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3222    /// by a restart, so these commands are rejected instead of re-enabling it.
3223    Disabled {
3224        module_id: String,
3225    },
3226    ReloadFailed {
3227        module_id: String,
3228        reason: String,
3229    },
3230    RegistrationStillActive {
3231        module_id: String,
3232        waited: Duration,
3233    },
3234    StatePoisoned {
3235        module_id: Option<String>,
3236    },
3237    CommandClosed {
3238        module_id: String,
3239    },
3240    /// A restart or reload arrived while a swap's candidate was warming. The
3241    /// swap owns the module until it cuts over or fails; a stop or disable
3242    /// would have aborted it instead.
3243    SwapInProgress {
3244        module_id: String,
3245    },
3246    /// A swap was refused before anything was spawned.
3247    SwapRefused {
3248        module_id: String,
3249        reason: SwapRefusal,
3250    },
3251    /// A swap spawned a candidate and gave up on it. The candidate has been
3252    /// killed and its slot freed; the incumbent was left serving and was never
3253    /// drained, except in the one `CutoverLost` case described on that arm.
3254    SwapFailed {
3255        module_id: String,
3256        arm: SwapFailureArm,
3257        detail: String,
3258        /// How the candidate exited, when it exited on its own before the
3259        /// supervisor gave up on it.
3260        candidate_exit: Option<ExitReport>,
3261    },
3262}
3263
3264/// Why a swap was refused before a candidate was spawned.
3265#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3266pub enum SwapRefusal {
3267    /// The module's config does not declare `overlap: "safe"`.
3268    OverlapExclusive,
3269    /// The module is not registered, so there is no incumbent to keep serving
3270    /// and nothing a swap would improve on; a plain restart is the tool.
3271    NotRegistered,
3272    /// The module does not speak the subc wire, so a candidate could never
3273    /// register or declare itself ready.
3274    ProtocolNone,
3275    /// The supervisor lacks the forwarding table (to cut routes over) or the
3276    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3277    NotConfigured,
3278    /// A swap is already open for this module.
3279    AlreadySwapping,
3280}
3281
3282impl SwapRefusal {
3283    pub fn as_str(self) -> &'static str {
3284        match self {
3285            Self::OverlapExclusive => "overlap_exclusive",
3286            Self::NotRegistered => "not_registered",
3287            Self::ProtocolNone => "protocol_none",
3288            Self::NotConfigured => "not_configured",
3289            Self::AlreadySwapping => "already_swapping",
3290        }
3291    }
3292}
3293
3294/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3295/// serving and undrained; see `CutoverLost`.
3296#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3297pub enum SwapFailureArm {
3298    /// The candidate process could not be started.
3299    SpawnFailed,
3300    /// The candidate did not register within the readiness budget.
3301    NeverRegistered,
3302    /// The candidate registered but did not declare itself ready in time.
3303    NeverReady,
3304    /// The candidate exited before cutover.
3305    CandidateExited,
3306    /// The candidate declared itself ready but failed its health probe.
3307    CandidateUnhealthy,
3308    /// An operator stop, disable or retire arrived while the candidate warmed.
3309    /// The candidate was killed and the operator's command then carried out on
3310    /// the incumbent.
3311    Interrupted,
3312    /// The candidate's connection closed at the moment of cutover. If it
3313    /// closed before forwarding moved, the incumbent is untouched. If it closed
3314    /// between the forwarding and registry halves of cutover, forwarding can no
3315    /// longer route to the incumbent, so the module is restarted plainly.
3316    CutoverLost,
3317}
3318
3319impl SwapFailureArm {
3320    pub fn as_str(self) -> &'static str {
3321        match self {
3322            Self::SpawnFailed => "spawn_failed",
3323            Self::NeverRegistered => "never_registered",
3324            Self::NeverReady => "never_ready",
3325            Self::CandidateExited => "candidate_exited",
3326            Self::CandidateUnhealthy => "candidate_unhealthy",
3327            Self::Interrupted => "interrupted",
3328            Self::CutoverLost => "cutover_lost",
3329        }
3330    }
3331}
3332
3333impl fmt::Display for SuperviseError {
3334    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3335        match self {
3336            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3337            Self::Spawn {
3338                program,
3339                source,
3340                cgroup_path: Some(cgroup_path),
3341            } => write!(
3342                f,
3343                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3344                cgroup_path.display(),
3345                program.display()
3346            ),
3347            Self::Spawn {
3348                program,
3349                source,
3350                cgroup_path: None,
3351            } => write!(
3352                f,
3353                "failed to spawn module '{}': {source}",
3354                program.display()
3355            ),
3356            Self::Cgroup { module_id, source } => {
3357                write!(
3358                    f,
3359                    "failed to prepare cgroup for module '{module_id}': {source}"
3360                )
3361            }
3362            Self::LaunchNonce { reason } => {
3363                write!(
3364                    f,
3365                    "failed to generate reserved-module launch nonce: {reason}"
3366                )
3367            }
3368            Self::Wait { module_id, source } => {
3369                write!(f, "failed to wait for module '{module_id}': {source}")
3370            }
3371            Self::Kill { module_id, source } => {
3372                write!(f, "failed to kill module '{module_id}': {source}")
3373            }
3374            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3375            Self::Registry(err) => write!(f, "registry error: {err}"),
3376            Self::ReloadUnavailable { module_id, reason } => {
3377                write!(f, "reload unavailable for module '{module_id}': {reason}")
3378            }
3379            Self::Disabled { module_id } => {
3380                write!(
3381                    f,
3382                    "module '{module_id}' is disabled; enable it before restart or reload"
3383                )
3384            }
3385            Self::ReloadFailed { module_id, reason } => {
3386                write!(f, "reload failed for module '{module_id}': {reason}")
3387            }
3388            Self::RegistrationStillActive { module_id, waited } => write!(
3389                f,
3390                "module '{module_id}' registration remained active after waiting {waited:?}"
3391            ),
3392            Self::StatePoisoned { module_id } => match module_id {
3393                Some(module_id) => {
3394                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3395                }
3396                None => write!(f, "supervisor state was poisoned"),
3397            },
3398            Self::CommandClosed { module_id } => {
3399                write!(
3400                    f,
3401                    "supervisor command channel for module '{module_id}' is closed"
3402                )
3403            }
3404            Self::SwapInProgress { module_id } => write!(
3405                f,
3406                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3407            ),
3408            Self::SwapRefused { module_id, reason } => match reason {
3409                SwapRefusal::OverlapExclusive => write!(
3410                    f,
3411                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3412                ),
3413                SwapRefusal::NotRegistered => write!(
3414                    f,
3415                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3416                ),
3417                SwapRefusal::ProtocolNone => write!(
3418                    f,
3419                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3420                ),
3421                SwapRefusal::NotConfigured => write!(
3422                    f,
3423                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3424                ),
3425                SwapRefusal::AlreadySwapping => {
3426                    write!(f, "module '{module_id}' is already being swapped")
3427                }
3428            },
3429            Self::SwapFailed {
3430                module_id,
3431                arm,
3432                detail,
3433                ..
3434            } => write!(
3435                f,
3436                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3437                arm.as_str()
3438            ),
3439        }
3440    }
3441}
3442
3443impl Error for SuperviseError {
3444    fn source(&self) -> Option<&(dyn Error + 'static)> {
3445        match self {
3446            Self::Spawn { source, .. }
3447            | Self::Cgroup { source, .. }
3448            | Self::Wait { source, .. }
3449            | Self::Kill { source, .. } => Some(source),
3450            Self::Forwarding(err) => Some(err),
3451            Self::Registry(err) => Some(err),
3452            Self::LaunchNonce { .. }
3453            | Self::InvalidSpec { .. }
3454            | Self::ReloadUnavailable { .. }
3455            | Self::Disabled { .. }
3456            | Self::ReloadFailed { .. }
3457            | Self::RegistrationStillActive { .. }
3458            | Self::StatePoisoned { .. }
3459            | Self::CommandClosed { .. }
3460            | Self::SwapInProgress { .. }
3461            | Self::SwapRefused { .. }
3462            | Self::SwapFailed { .. } => None,
3463        }
3464    }
3465}
3466
3467pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3468    if spec.module_id.trim().is_empty() {
3469        return Err(SuperviseError::InvalidSpec {
3470            reason: "module_id must not be empty".to_string(),
3471        });
3472    }
3473
3474    Ok(())
3475}
3476
3477#[derive(Debug, Default)]
3478struct HealthProbeRuntime {
3479    registered_connection: Option<crate::ConnectionId>,
3480    advertised: bool,
3481    next_probe_at: Option<Instant>,
3482    probe_index: u64,
3483}
3484
3485impl HealthProbeRuntime {
3486    fn refresh_registration(
3487        &mut self,
3488        spec: &ModuleSpec,
3489        runtime: &SupervisorRuntimeConfig,
3490        registry: &Registry,
3491        snapshot: &SharedSnapshot,
3492    ) {
3493        // THE PROBE GATE FOR A MODULE THAT SPEAKS NO SUBC WIRE, placed here
3494        // because this is the only place that ever arms a probe: leaving
3495        // `advertised` false and `next_probe_at` empty makes `due()` false
3496        // forever, so `run_health_probe_cycle` -- and with it every arm of
3497        // `probe_module_health`, including the one that reads an absent
3498        // registration as proof the module is gone and escalates to a restart --
3499        // is unreachable for this module.
3500        //
3501        // That arm is right for a subc module and is exactly wrong here: a
3502        // `protocol: "none"` module never registers by declaration, so the
3503        // absence it would classify is the module working as configured.
3504        if spec.protocol == ModuleProtocol::None {
3505            self.registered_connection = None;
3506            self.advertised = false;
3507            self.next_probe_at = None;
3508            return;
3509        }
3510
3511        let registration = match registry.get_module(&spec.module_id) {
3512            Ok(registration) => registration,
3513            Err(err) => {
3514                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3515                self.advertised = false;
3516                self.next_probe_at = None;
3517                return;
3518            }
3519        };
3520
3521        let Some(registration) = registration else {
3522            self.registered_connection = None;
3523            self.advertised = false;
3524            self.next_probe_at = None;
3525            return;
3526        };
3527
3528        let advertised = registration
3529            .control_ops
3530            .iter()
3531            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
3532        if !advertised {
3533            self.registered_connection = Some(registration.connection_id);
3534            self.advertised = false;
3535            self.next_probe_at = None;
3536            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3537                state.health.status = SupervisorHealthStatus::Unknown;
3538                state.health.consecutive_failures = 0;
3539                state.health.last_probe_ms = None;
3540                state.health.detail = None;
3541                state.health.metrics = None;
3542            });
3543            return;
3544        }
3545
3546        let reregistered = self.registered_connection != Some(registration.connection_id);
3547        self.registered_connection = Some(registration.connection_id);
3548        self.advertised = true;
3549        if reregistered || self.next_probe_at.is_none() {
3550            self.probe_index = 0;
3551            self.next_probe_at = Some(
3552                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3553            );
3554            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3555                state.health.status = SupervisorHealthStatus::Unknown;
3556                state.health.consecutive_failures = 0;
3557                state.health.detail = None;
3558                state.health.metrics = None;
3559            });
3560        }
3561    }
3562
3563    fn wake_after(&self) -> Duration {
3564        if !self.advertised {
3565            return REGISTRY_RELEASE_POLL;
3566        }
3567        self.next_probe_at
3568            .map(|next| next.saturating_duration_since(Instant::now()))
3569            .unwrap_or(REGISTRY_RELEASE_POLL)
3570    }
3571
3572    fn due(&self) -> bool {
3573        self.advertised
3574            && self
3575                .next_probe_at
3576                .is_some_and(|next| Instant::now() >= next)
3577    }
3578
3579    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
3580        self.probe_index = self.probe_index.wrapping_add(1);
3581        self.next_probe_at = Some(
3582            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
3583        );
3584    }
3585}
3586
3587/// What a failed health probe actually OBSERVED, kept apart from how it reads.
3588///
3589/// This was a struct with a single `message: String`, and every one of the
3590/// fifteen construction sites collapsed into it. Each site knows exactly what it
3591/// saw -- the lane is gone, the module did not answer in time, the module
3592/// answered with the wrong thing -- and `handle_health_probe_failure` then
3593/// treated all of them identically: increment a counter, compare to a threshold,
3594/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
3595/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
3596///
3597/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
3598///
3599/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
3600///   answer on it again.
3601/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
3602///   AND with a perfectly healthy one that lost a CPU race -- which is what
3603///   happens under machine load, and is how this supervisor killed a healthy
3604///   module three times in one day.
3605/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
3606///   Restarting on it is defensible, but it is not the silence case and should
3607///   never be counted as one.
3608/// * `Misconfigured` is a daemon-side fault. The module has not been asked
3609///   anything, so it cannot be evidence about the module at all.
3610///
3611/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
3612/// one that fires most often, and while every variant collapsed into one string
3613/// it carried the same weight as the strongest.
3614///
3615/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
3616/// DESIGN and a reader stopping at it gets the build backwards: the restart
3617/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
3618/// probes still increment the failure streak and drive escalation at the
3619/// threshold (see `is_proof_of_death` below for why that is deliberate and
3620/// what gates the change). Absence of evidence restarts modules today.
3621#[derive(Debug)]
3622enum HealthProbeEvidence {
3623    /// The module's control lane is gone. Proof of death.
3624    LaneDead,
3625    /// No reply within the deadline. Proves nothing about the module's state.
3626    NoAnswer,
3627    /// The module replied, but not with a usable health report. Proves it is alive.
3628    BadAnswer,
3629    /// The daemon could not ask. Says nothing about the module.
3630    Misconfigured,
3631}
3632
3633#[derive(Debug)]
3634struct HealthProbeError {
3635    evidence: HealthProbeEvidence,
3636    message: String,
3637}
3638
3639impl HealthProbeError {
3640    fn lane_dead(message: impl Into<String>) -> Self {
3641        Self::with(HealthProbeEvidence::LaneDead, message)
3642    }
3643
3644    fn no_answer(message: impl Into<String>) -> Self {
3645        Self::with(HealthProbeEvidence::NoAnswer, message)
3646    }
3647
3648    fn bad_answer(message: impl Into<String>) -> Self {
3649        Self::with(HealthProbeEvidence::BadAnswer, message)
3650    }
3651
3652    fn misconfigured(message: impl Into<String>) -> Self {
3653        Self::with(HealthProbeEvidence::Misconfigured, message)
3654    }
3655
3656    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
3657        Self {
3658            evidence,
3659            message: message.into(),
3660        }
3661    }
3662
3663    /// Whether this observation is proof the module cannot serve.
3664    ///
3665    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
3666    /// variant that fires under CPU starvation, and treating it as proof is the
3667    /// defect this enum exists to make impossible to reintroduce silently.
3668    ///
3669    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
3670    /// to restart also needs a bound for the case it excludes -- a genuinely
3671    /// wedged module, alive but never answering -- and that bound must come from
3672    /// the distribution of real late-answer latencies, which nothing measures
3673    /// yet. Landing the classification first makes the later change a one-line
3674    /// decision against evidence that already exists, rather than two unproven
3675    /// changes at once.
3676    #[allow(dead_code)]
3677    fn is_proof_of_death(&self) -> bool {
3678        matches!(self.evidence, HealthProbeEvidence::LaneDead)
3679    }
3680
3681    /// Short stable label for logs and the health snapshot.
3682    ///
3683    /// An operator reading `ck health` currently cannot tell "the module is gone"
3684    /// from "the module did not answer in five seconds", because both render as
3685    /// prose in the same field. These labels are what make the two
3686    /// distinguishable at a glance, and they are what a later restart-policy
3687    /// change will be argued from.
3688    fn label(&self) -> &'static str {
3689        match self.evidence {
3690            HealthProbeEvidence::LaneDead => "lane-dead",
3691            HealthProbeEvidence::NoAnswer => "no-answer",
3692            HealthProbeEvidence::BadAnswer => "bad-answer",
3693            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
3694        }
3695    }
3696}
3697
3698impl fmt::Display for HealthProbeError {
3699    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3700        f.write_str(&self.message)
3701    }
3702}
3703
3704async fn run_health_probe_cycle(
3705    spec: &ModuleSpec,
3706    runtime: &SupervisorRuntimeConfig,
3707    registry: &Registry,
3708    process_liveness: &SupervisorProcessLiveness,
3709    snapshot: &SharedSnapshot,
3710    child: &mut Option<SupervisedChild>,
3711) {
3712    let now_ms = unix_ms_now();
3713    match probe_module_health(&spec.module_id, runtime, None).await {
3714        Ok(report) => {
3715            handle_health_report(
3716                spec,
3717                runtime,
3718                registry,
3719                process_liveness,
3720                snapshot,
3721                child,
3722                report,
3723                now_ms,
3724            )
3725            .await;
3726        }
3727        Err(err) => {
3728            handle_health_probe_failure(
3729                spec,
3730                runtime,
3731                registry,
3732                process_liveness,
3733                snapshot,
3734                child,
3735                err,
3736                now_ms,
3737            )
3738            .await;
3739        }
3740    }
3741}
3742
3743async fn probe_module_health(
3744    module_id: &str,
3745    runtime: &SupervisorRuntimeConfig,
3746    drain_deadline: Option<Instant>,
3747) -> Result<HealthReport, HealthProbeError> {
3748    let Some(forwarding) = runtime.forwarding.as_ref() else {
3749        return Err(HealthProbeError::misconfigured(
3750            "supervisor was not configured with a forwarding table",
3751        ));
3752    };
3753    let probe_started_at = Instant::now();
3754    let mut deadline = probe_started_at + runtime.health.deadline;
3755    if let Some(drain_deadline) = drain_deadline {
3756        deadline = deadline.min(drain_deadline);
3757    }
3758    let pending = if drain_deadline.is_some() {
3759        forwarding.begin_drain_health_probe_rpc_for(
3760            module_id,
3761            MODULE_CONTROL_OP_HEALTH_CHECK,
3762            probe_started_at,
3763            deadline,
3764        )
3765    } else {
3766        forwarding.begin_health_probe_rpc_for(
3767            module_id,
3768            MODULE_CONTROL_OP_HEALTH_CHECK,
3769            probe_started_at,
3770            deadline,
3771        )
3772    }
3773    .map_err(|err| {
3774        // The endpoint is not registered, so there is no live control lane to
3775        // ask. That is the module being absent, not slow.
3776        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3777    })?;
3778    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3779}
3780
3781/// [`probe_module_health`] for one endpoint rather than the id's active one.
3782///
3783/// A swap probes two processes that no by-id lookup reaches: its candidate
3784/// before cutover, and its superseded incumbent (for busy gauges) while the
3785/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
3786/// bounds the by-id drain probe.
3787async fn probe_endpoint_health(
3788    endpoint: crate::ModuleEndpointId,
3789    runtime: &SupervisorRuntimeConfig,
3790    deadline_cap: Option<Instant>,
3791) -> Result<HealthReport, HealthProbeError> {
3792    let Some(forwarding) = runtime.forwarding.as_ref() else {
3793        return Err(HealthProbeError::misconfigured(
3794            "supervisor was not configured with a forwarding table",
3795        ));
3796    };
3797    let probe_started_at = Instant::now();
3798    let mut deadline = probe_started_at + runtime.health.deadline;
3799    if let Some(cap) = deadline_cap {
3800        deadline = deadline.min(cap);
3801    }
3802    let pending = forwarding
3803        .begin_endpoint_health_probe_rpc_for(
3804            endpoint,
3805            MODULE_CONTROL_OP_HEALTH_CHECK,
3806            probe_started_at,
3807            deadline,
3808        )
3809        .map_err(|err| {
3810            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3811        })?;
3812    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3813}
3814
3815/// Send a begun health probe and classify its answer.
3816async fn await_health_probe(
3817    forwarding: &ForwardingTable,
3818    pending: PendingModuleControlRpc,
3819    deadline: Instant,
3820    probe_budget: Duration,
3821) -> Result<HealthReport, HealthProbeError> {
3822    let PendingModuleControlRpc {
3823        endpoint,
3824        module_sink,
3825        negotiated_ver,
3826        corr,
3827        receiver,
3828    } = pending;
3829    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
3830        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
3831    })?;
3832    let frame = Frame::build_with_version(
3833        negotiated_ver,
3834        FrameType::Request,
3835        control_flags(),
3836        0,
3837        0,
3838        corr,
3839        body,
3840    )
3841    .map_err(|err| {
3842        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
3843    })?;
3844
3845    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
3846    // blocks waiting for capacity when the module's egress queue is full, and an
3847    // unbounded await here freezes the whole supervision actor (it stops polling
3848    // Child::wait and supervisor commands), making the module unrecoverable
3849    // in-band. On timeout the probe fails like any transport failure.
3850    match timeout_at(deadline, module_sink.send(frame)).await {
3851        Ok(Ok(())) => {}
3852        Ok(Err(err)) => {
3853            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3854            // A closed sink means the module's egress channel is gone -- the
3855            // receiving half is dropped when its connection tears down. Proof.
3856            return Err(HealthProbeError::lane_dead(format!(
3857                "failed to send health.check: {err}"
3858            )));
3859        }
3860        Err(_elapsed) => {
3861            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3862            // A full egress queue means the module is not draining its socket, which
3863            // is consistent with a wedged module AND with one whose reader is merely
3864            // starved. Silence, not proof.
3865            return Err(HealthProbeError::no_answer(
3866                "health.check send timed out before enqueue (module egress full)",
3867            ));
3868        }
3869    }
3870
3871    match timeout_at(deadline, receiver).await {
3872        // Each arm records WHAT WAS OBSERVED. Four of them are the module
3873        // demonstrably answering -- rejected, non-health, malformed, wrong op --
3874        // and those prove it is alive even though the probe failed.
3875        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
3876            response.health_report().ok_or_else(|| {
3877                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
3878            })
3879        }
3880        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
3881            format!("health.check rejected: {}", body.message),
3882        )),
3883        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
3884            Err(HealthProbeError::lane_dead(message))
3885        }
3886        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
3887            Err(HealthProbeError::bad_answer(message))
3888        }
3889        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
3890            Err(HealthProbeError::bad_answer(format!(
3891                "expected module-control op '{expected}', got '{actual}'"
3892            )))
3893        }
3894        // A reply that crosses the deadline before this waiter observes it is
3895        // still proof of life. The forwarding path records its end-to-end latency
3896        // before delivering this classification.
3897        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
3898            "module answered health.check after its daemon deadline",
3899        )),
3900        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
3901            "health.check waiter was canceled before the module responded",
3902        )),
3903        Err(_) => {
3904            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
3905            Err(HealthProbeError::no_answer(format!(
3906                "module did not answer health.check within {probe_budget:?}"
3907            )))
3908        }
3909    }
3910}
3911
3912#[allow(clippy::too_many_arguments)]
3913async fn handle_health_report(
3914    spec: &ModuleSpec,
3915    runtime: &SupervisorRuntimeConfig,
3916    registry: &Registry,
3917    process_liveness: &SupervisorProcessLiveness,
3918    snapshot: &SharedSnapshot,
3919    child: &mut Option<SupervisedChild>,
3920    report: HealthReport,
3921    now_ms: u64,
3922) {
3923    let status = supervisor_health_status(report.status);
3924    let detail = report.detail.clone();
3925    let metrics = truncate_health_metrics(report.metrics);
3926    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3927        state.health.status = status;
3928        state.health.last_probe_ms = Some(now_ms);
3929        state.health.detail = detail.clone();
3930        state.health.metrics = metrics.clone();
3931        state.health.consecutive_failures = 0;
3932    });
3933
3934    let action = match report.status {
3935        HealthStatus::Ok => return,
3936        HealthStatus::Degraded => runtime.health.on_degraded,
3937        HealthStatus::Failing => runtime.health.on_failing,
3938    };
3939    apply_l3_health_action(
3940        spec,
3941        runtime,
3942        registry,
3943        process_liveness,
3944        snapshot,
3945        child,
3946        status,
3947        detail.as_deref(),
3948        action,
3949        now_ms,
3950    )
3951    .await;
3952}
3953
3954#[allow(clippy::too_many_arguments)]
3955async fn handle_health_probe_failure(
3956    spec: &ModuleSpec,
3957    runtime: &SupervisorRuntimeConfig,
3958    registry: &Registry,
3959    process_liveness: &SupervisorProcessLiveness,
3960    snapshot: &SharedSnapshot,
3961    child: &mut Option<SupervisedChild>,
3962    err: HealthProbeError,
3963    now_ms: u64,
3964) {
3965    let threshold = runtime.health.failure_threshold.max(1);
3966    let mut failures = 0;
3967    // Carry the evidence class into the operator-visible detail. Without it,
3968    // "module did not answer within 5s" and "the control lane is gone" are two
3969    // prose strings in the same field, and the reader has to know the codebase to
3970    // tell which one is proof of anything.
3971    let detail = format!("[{}] {err}", err.label());
3972    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3973        state.health.last_probe_ms = Some(now_ms);
3974        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3975        state.health.detail = Some(detail.clone());
3976        state.health.metrics = None;
3977        failures = state.health.consecutive_failures;
3978    });
3979
3980    if failures < threshold {
3981        warn!(
3982            module_id = %spec.module_id,
3983            consecutive_failures = failures,
3984            threshold,
3985            evidence = err.label(),
3986            detail = %detail,
3987            "health.check probe failed"
3988        );
3989        return;
3990    }
3991
3992    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3993        state.state = ModuleState::Unresponsive;
3994        state.health.status = SupervisorHealthStatus::Unresponsive;
3995    });
3996    // The evidence class is logged at the kill site because this is the line an
3997    // operator reads after an unexplained restart. A streak of `no-answer` under
3998    // machine load is the known false-positive shape; a `lane-dead` is not.
3999    if runtime.health.critical {
4000        error!(
4001            module_id = %spec.module_id,
4002            status = "unresponsive",
4003            evidence = err.label(),
4004            detail = %detail,
4005            "critical module health alert"
4006        );
4007    } else {
4008        warn!(
4009            module_id = %spec.module_id,
4010            status = "unresponsive",
4011            evidence = err.label(),
4012            detail = %detail,
4013            "module health threshold breached"
4014        );
4015    }
4016    if let Err(err) = health_restart_child(
4017        spec,
4018        runtime,
4019        registry,
4020        process_liveness,
4021        snapshot,
4022        child,
4023        SupervisorHealthStatus::Unresponsive,
4024        Some(&detail),
4025        now_ms,
4026    )
4027    .await
4028    {
4029        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4030    }
4031}
4032
4033#[allow(clippy::too_many_arguments)]
4034async fn apply_l3_health_action(
4035    spec: &ModuleSpec,
4036    runtime: &SupervisorRuntimeConfig,
4037    registry: &Registry,
4038    process_liveness: &SupervisorProcessLiveness,
4039    snapshot: &SharedSnapshot,
4040    child: &mut Option<SupervisedChild>,
4041    status: SupervisorHealthStatus,
4042    detail: Option<&str>,
4043    action: HealthAction,
4044    now_ms: u64,
4045) {
4046    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4047    match action {
4048        HealthAction::Report => {
4049            info!(
4050                module_id = %spec.module_id,
4051                status = ?status,
4052                detail,
4053                "module reported non-ok health"
4054            );
4055        }
4056        HealthAction::Alert => {
4057            error!(
4058                module_id = %spec.module_id,
4059                status = ?status,
4060                detail,
4061                "module health alert"
4062            );
4063        }
4064        HealthAction::Restart => {
4065            if let Err(err) = health_restart_child(
4066                spec,
4067                runtime,
4068                registry,
4069                process_liveness,
4070                snapshot,
4071                child,
4072                status,
4073                detail,
4074                now_ms,
4075            )
4076            .await
4077            {
4078                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4079            }
4080        }
4081    }
4082}
4083
4084#[allow(clippy::too_many_arguments)]
4085async fn health_restart_child(
4086    spec: &ModuleSpec,
4087    runtime: &SupervisorRuntimeConfig,
4088    registry: &Registry,
4089    process_liveness: &SupervisorProcessLiveness,
4090    snapshot: &SharedSnapshot,
4091    child: &mut Option<SupervisedChild>,
4092    status: SupervisorHealthStatus,
4093    detail: Option<&str>,
4094    now_ms: u64,
4095) -> Result<(), SuperviseError> {
4096    let (enabled, schedule) = {
4097        let mut state = lock_snapshot(snapshot)?;
4098        let enabled = state.enabled;
4099        let schedule = if enabled {
4100            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4101        } else {
4102            None
4103        };
4104        (enabled, schedule)
4105    };
4106
4107    if !enabled {
4108        return Err(SuperviseError::Disabled {
4109            module_id: spec.module_id.clone(),
4110        });
4111    }
4112
4113    if schedule.is_none() {
4114        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4115        error!(
4116            module_id = %spec.module_id,
4117            status = ?status,
4118            detail,
4119            max_restarts = runtime.restart_policy.max_restarts,
4120            window_secs = runtime.restart_policy.window.as_secs(),
4121            reason = %runtime.restart_policy.budget_exhausted_detail(),
4122            "health restart budget exhausted; marking module failed"
4123        );
4124        let stop_notice = begin_forwarding_drain_if_configured(
4125            spec,
4126            runtime,
4127            registry,
4128            snapshot,
4129            Some(true),
4130            RouteCloseReason::Disable,
4131        )
4132        .await?;
4133        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4134            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4135        })?;
4136        drain_optional_child(
4137            &spec.module_id,
4138            spec.protocol,
4139            stop_notice,
4140            registry,
4141            runtime.forwarding.as_deref(),
4142            snapshot,
4143            &runtime.terminal_ring,
4144            &runtime.spawn_events,
4145            child,
4146            runtime.drain_timeout,
4147            ModuleState::Failed,
4148            Some(true),
4149        )
4150        .await?;
4151        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4152        return Ok(());
4153    }
4154
4155    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4156    let mut restart_count = 0;
4157    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4158        restart_count = state.crash_restarts.len();
4159        state.state = ModuleState::Unresponsive;
4160        state.health.status = status;
4161        state.health.last_action = Some(HealthAction::Restart.to_string());
4162        state.health.last_action_ms = Some(now_ms);
4163    })?;
4164    warn!(
4165        module_id = %spec.module_id,
4166        status = ?status,
4167        detail,
4168        restart_count,
4169        restart_in_window = schedule.restart_in_window,
4170        delay_ms = schedule.delay.as_millis() as u64,
4171        "health-triggered module restart"
4172    );
4173
4174    let stop_notice = begin_forwarding_drain_if_configured(
4175        spec,
4176        runtime,
4177        registry,
4178        snapshot,
4179        Some(true),
4180        RouteCloseReason::Restart,
4181    )
4182    .await?;
4183    drain_optional_child(
4184        &spec.module_id,
4185        spec.protocol,
4186        stop_notice,
4187        registry,
4188        runtime.forwarding.as_deref(),
4189        snapshot,
4190        &runtime.terminal_ring,
4191        &runtime.spawn_events,
4192        child,
4193        runtime.drain_timeout,
4194        ModuleState::Restarting,
4195        Some(true),
4196    )
4197    .await?;
4198    schedule_respawn(
4199        runtime,
4200        snapshot,
4201        &spec.module_id,
4202        schedule.delay,
4203        RespawnKind::Spawn,
4204    )
4205}
4206
4207fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4208    if let Some(reply) = runtime
4209        .deferred_reload_reply
4210        .lock()
4211        .unwrap_or_else(|p| p.into_inner())
4212        .take()
4213    {
4214        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4215            module_id: module_id.to_string(),
4216            reason: reason.to_string(),
4217        }));
4218    }
4219}
4220
4221fn schedule_respawn(
4222    runtime: &SupervisorRuntimeConfig,
4223    snapshot: &SharedSnapshot,
4224    module_id: &str,
4225    delay: Duration,
4226    kind: RespawnKind,
4227) -> Result<(), SuperviseError> {
4228    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4229    update_snapshot(snapshot, Some(module_id), |state| {
4230        state.respawn_pending = true
4231    })?;
4232    *runtime
4233        .scheduled_respawn
4234        .lock()
4235        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
4236        deadline: Instant::now() + delay,
4237        kind,
4238    });
4239    Ok(())
4240}
4241
4242fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4243    let _ = update_snapshot(snapshot, Some(module_id), |state| {
4244        state.health.last_action = Some(action);
4245        state.health.last_action_ms = Some(now_ms);
4246    });
4247}
4248
4249fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4250    match status {
4251        HealthStatus::Ok => SupervisorHealthStatus::Ok,
4252        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4253        HealthStatus::Failing => SupervisorHealthStatus::Failing,
4254    }
4255}
4256
4257/// Caps the metrics blob stored in the cached supervisor snapshot, which is
4258/// returned to every `supervisor.list` and `supervisor.health` caller.
4259///
4260/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
4261/// path: that request exists to return a module's complete metrics object, and
4262/// `ck health <module-id>` documents it as the way to see what the cached view
4263/// truncates. The asymmetry is the feature.
4264///
4265/// So a new caller must decide which side it is on rather than assume the cap is
4266/// universal. Reaching for it on a fresh-probe path would silently reintroduce
4267/// the truncation that path exists to avoid.
4268fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4269    let metrics = metrics?;
4270    match serde_json::to_vec(&metrics) {
4271        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4272            "truncated": true,
4273            "original_bytes": encoded.len(),
4274        })),
4275        Ok(_) | Err(_) => Some(metrics),
4276    }
4277}
4278
4279/// Spread health probes so a fleet-wide restart does not converge them.
4280///
4281/// The delay is derived from the module id and probe index rather than a random
4282/// source, so it is deterministic per module: a module keeps its own offset
4283/// across daemon restarts instead of re-rolling into a collision.
4284fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4285    if cadence.is_zero() {
4286        return Duration::ZERO;
4287    }
4288    let cadence_ms = cadence.as_millis() as u64;
4289    // This early return is REDUNDANT, deliberately, and a mutation run will show
4290    // it surviving removal. Recording why here so the next person to notice does
4291    // not have to re-derive it:
4292    //
4293    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
4294    //   a zero cadence and builds the Duration from whole milliseconds, so a
4295    //   sub-millisecond cadence cannot come from config.
4296    // - Even if reached it changes no answer. The `.max(1)` below makes the span
4297    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
4298    //   -- exactly what this returns.
4299    //
4300    // Kept as a guard against a future widening of the config parser (accepting
4301    // microseconds, say), which would make the sub-millisecond case reachable.
4302    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
4303    // divides by zero. Remove this and nothing changes.
4304    if cadence_ms == 0 {
4305        return cadence;
4306    }
4307    // Note that this never returns less than one cadence, including for the FIRST
4308    // probe. So a freshly registered module reports health `unknown` for a full
4309    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
4310    // ready to answer.
4311    //
4312    // That is a property of the supervisor's schedule, not of any module: an
4313    // operator watching a restart sees `unknown` and cannot tell it from a module
4314    // that is slow to warm. Measured on two unrelated modules, both flipping to
4315    // `ok` between 22s and 32s after restart.
4316    //
4317    // Left as-is because spreading the first probe is what keeps a fleet-wide
4318    // restart from firing fourteen simultaneous probes into a cold machine. The
4319    // alternative -- probe at t+0 and jitter only from the second onward -- trades
4320    // that thundering herd for a faster first reading.
4321    let jitter_span = (cadence_ms / 10).max(1);
4322    let hash = module_id.as_bytes().iter().fold(
4323        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
4324        |acc, byte| {
4325            acc.wrapping_mul(1099511628211)
4326                .wrapping_add(u64::from(*byte))
4327        },
4328    );
4329    cadence + Duration::from_millis(hash % jitter_span)
4330}
4331
4332#[cfg(test)]
4333mod tests {
4334    use super::*;
4335
4336    #[test]
4337    fn readding_a_module_clears_its_rescan_removal_tombstone() {
4338        let handle = SupervisorHandle::new();
4339        let module_id = "readded-tombstone";
4340        handle.record_rescan_removal(module_id);
4341        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
4342
4343        handle.apply_identity_configuration(&ModuleSpec {
4344            module_id: module_id.to_string(),
4345            program: PathBuf::from("/test/module"),
4346            args: Vec::new(),
4347            env: Vec::new(),
4348            reserved: false,
4349            reserved_prefixes: Vec::new(),
4350            protocol: ModuleProtocol::Subc,
4351            overlap: Default::default(),
4352        });
4353
4354        assert!(
4355            handle.removal_tombstone_age_ms(module_id).is_none(),
4356            "a re-added module must not retain a stale removal tombstone"
4357        );
4358    }
4359
4360    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
4361        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
4362        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
4363            snapshot.process_alive = true;
4364            snapshot.pid = Some(41);
4365            snapshot.spawned_at_ms = Some(42);
4366            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
4367            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
4368                device: 43,
4369                inode: 44,
4370            });
4371        })
4372        .unwrap();
4373        snapshot
4374    }
4375
4376    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
4377        let snapshot = lock_snapshot(snapshot).unwrap();
4378        assert!(!snapshot.process_alive);
4379        assert_eq!(snapshot.pid, None);
4380        assert_eq!(snapshot.spawned_at_ms, None);
4381        assert_eq!(snapshot.spawned_from, None);
4382        assert_eq!(snapshot.spawned_file_identity, None);
4383    }
4384
4385    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4386    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
4387        let supervisor = Supervisor::default();
4388        let mut runtime = supervisor.runtime_config();
4389        runtime.test_seed_stale_facts_before_enable_spawn = true;
4390        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
4391        let mut child = None;
4392        let spec = ModuleSpec {
4393            module_id: "failed-enable-clears-facts".to_string(),
4394            program: PathBuf::from("/definitely/missing/failed-enable-module"),
4395            args: Vec::new(),
4396            env: Vec::new(),
4397            reserved: false,
4398            reserved_prefixes: Vec::new(),
4399            protocol: ModuleProtocol::Subc,
4400            overlap: Default::default(),
4401        };
4402
4403        let result = set_child_enabled(
4404            &spec,
4405            &runtime,
4406            &supervisor.registry,
4407            &supervisor.process_liveness,
4408            &snapshot,
4409            &mut child,
4410            true,
4411        )
4412        .await;
4413
4414        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
4415        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4416        assert_snapshot_process_facts_cleared(&snapshot);
4417    }
4418
4419    #[tokio::test]
4420    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
4421        let supervisor = Supervisor::default();
4422        let runtime = supervisor.runtime_config();
4423        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
4424            ModuleState::Restarting,
4425            true,
4426        )));
4427        let spec = ModuleSpec {
4428            module_id: "start-stranded-restarting".to_string(),
4429            program: super::terminal_history_tests::fake_aft_stub_path(),
4430            args: Vec::new(),
4431            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
4432            reserved: false,
4433            reserved_prefixes: Vec::new(),
4434            protocol: ModuleProtocol::None,
4435            overlap: Default::default(),
4436        };
4437        let mut child = None;
4438        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
4439        assert!(!super::set_child_enabled(
4440            &spec,
4441            &runtime,
4442            &Registry::default(),
4443            &supervisor.process_liveness,
4444            &snapshot,
4445            &mut child,
4446            true
4447        )
4448        .await
4449        .unwrap());
4450        assert!(child.is_none());
4451        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
4452        assert!(super::set_child_enabled(
4453            &spec,
4454            &runtime,
4455            &Registry::default(),
4456            &supervisor.process_liveness,
4457            &snapshot,
4458            &mut child,
4459            true
4460        )
4461        .await
4462        .unwrap());
4463        assert_eq!(
4464            lock_snapshot(&snapshot).unwrap().state,
4465            ModuleState::Running
4466        );
4467        let mut child = child.unwrap();
4468        child.start_kill().unwrap();
4469        child.wait().await.unwrap();
4470    }
4471
4472    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4473    async fn failed_reload_spawn_clears_current_process_facts() {
4474        let supervisor = Supervisor::default();
4475        let mut runtime = supervisor.runtime_config();
4476        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
4477        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4478        let mut child = None;
4479        let spec = ModuleSpec {
4480            module_id: "failed-reload-clears-facts".to_string(),
4481            program: PathBuf::from("/unused/failed-reload-module"),
4482            args: Vec::new(),
4483            env: Vec::new(),
4484            reserved: false,
4485            reserved_prefixes: Vec::new(),
4486            protocol: ModuleProtocol::Subc,
4487            overlap: Default::default(),
4488        };
4489
4490        let result = handle_reload_spawn_failure(
4491            &spec,
4492            &runtime,
4493            &supervisor.process_liveness,
4494            &snapshot,
4495            &mut child,
4496            "forced reload spawn failure".to_string(),
4497        )
4498        .await;
4499
4500        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
4501        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4502        assert_snapshot_process_facts_cleared(&snapshot);
4503    }
4504
4505    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4506    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
4507        let supervisor = Supervisor::default();
4508        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4509        let module = supervisor.supervised_module(
4510            ModuleSpec {
4511                module_id: "drop-clears-facts".to_string(),
4512                program: PathBuf::from("/unused/drop-module"),
4513                args: Vec::new(),
4514                env: Vec::new(),
4515                reserved: false,
4516                reserved_prefixes: Vec::new(),
4517                protocol: ModuleProtocol::Subc,
4518                overlap: Default::default(),
4519            },
4520            supervisor.runtime_config(),
4521            Arc::clone(&snapshot),
4522            None,
4523        );
4524        assert!(!module
4525            .inner
4526            .monitor
4527            .lock()
4528            .unwrap()
4529            .as_ref()
4530            .unwrap()
4531            .is_finished());
4532
4533        drop(module);
4534
4535        assert_eq!(
4536            lock_snapshot(&snapshot).unwrap().state,
4537            ModuleState::Stopped
4538        );
4539        assert_snapshot_process_facts_cleared(&snapshot);
4540    }
4541
4542    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4543    async fn configuration_update_does_not_replace_captured_running_process_facts() {
4544        let supervisor = Supervisor::default();
4545        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4546        let initial = ModuleSpec {
4547            module_id: "rescan-preserves-spawn-facts".to_string(),
4548            program: PathBuf::from("/spawned/module"),
4549            args: Vec::new(),
4550            env: Vec::new(),
4551            reserved: false,
4552            reserved_prefixes: Vec::new(),
4553            protocol: ModuleProtocol::Subc,
4554            overlap: Default::default(),
4555        };
4556        let module = supervisor.supervised_module(
4557            initial.clone(),
4558            supervisor.runtime_config(),
4559            snapshot,
4560            None,
4561        );
4562        let before = module.status().unwrap();
4563        let mut replacement = initial;
4564        replacement.program = PathBuf::from("/rescanned/replacement-module");
4565
4566        module
4567            .update_configuration(replacement, HealthConfig::default(), None)
4568            .await
4569            .unwrap();
4570
4571        let after = module.status().unwrap();
4572        assert_eq!(after.pid, before.pid);
4573        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
4574        assert_eq!(after.spawned_from, before.spawned_from);
4575        drop(module);
4576    }
4577}
4578
4579fn unix_ms_now() -> u64 {
4580    SystemTime::now()
4581        .duration_since(UNIX_EPOCH)
4582        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
4583        .unwrap_or(0)
4584}
4585
4586async fn supervise_loop(
4587    mut spec: ModuleSpec,
4588    mut runtime: SupervisorRuntimeConfig,
4589    registry: Arc<Registry>,
4590    process_liveness: Arc<SupervisorProcessLiveness>,
4591    snapshot: SharedSnapshot,
4592    mut child: Option<SupervisedChild>,
4593    mut commands: mpsc::Receiver<SupervisorCommand>,
4594) {
4595    let mut health_probe = HealthProbeRuntime::default();
4596    // All restart backoffs run here, including health and operator requests.
4597    // While one is pending the loop serves commands, so disable or drain can
4598    // cancel the replacement without spawning a process just to stop it.
4599    let mut pending_respawn: Option<PendingRespawn> = None;
4600    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
4601    // before anything else so a stop that interrupted a swap runs at once.
4602    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
4603    loop {
4604        if let Some(scheduled) = runtime
4605            .scheduled_respawn
4606            .lock()
4607            .unwrap_or_else(|p| p.into_inner())
4608            .take()
4609        {
4610            pending_respawn = Some(scheduled);
4611        }
4612        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
4613            pending_respawn = None;
4614            cancel_deferred_reload(
4615                &runtime,
4616                &spec.module_id,
4617                "respawn cancelled by a supervisor command",
4618            );
4619        }
4620        if child.is_none() && pending_respawn.is_none() {
4621            cancel_deferred_reload(
4622                &runtime,
4623                &spec.module_id,
4624                "respawn cancelled before a replacement was spawned",
4625            );
4626            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4627                state.respawn_pending = false;
4628                state.coalesced_restart_pending = false;
4629                if matches!(
4630                    state.state,
4631                    ModuleState::Restarting
4632                        | ModuleState::Starting
4633                        | ModuleState::Draining
4634                        | ModuleState::Unresponsive
4635                ) {
4636                    error!(module_id = %spec.module_id, state = ?state.state,
4637                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
4638                    state.state = ModuleState::Failed;
4639                    clear_current_process_facts(state);
4640                }
4641            });
4642        }
4643        if let Some(command) = requeued.pop_front() {
4644            if !handle_supervisor_command(
4645                command,
4646                &mut spec,
4647                &mut runtime,
4648                &registry,
4649                &process_liveness,
4650                &snapshot,
4651                &mut child,
4652                &mut commands,
4653                &mut requeued,
4654            )
4655            .await
4656            {
4657                return;
4658            }
4659            if child.is_some() || !respawn_still_pending(&snapshot) {
4660                pending_respawn = None;
4661                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4662                    state.respawn_pending = false
4663                });
4664            }
4665            continue;
4666        }
4667        if child.is_some() {
4668            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
4669            let probe_sleep = sleep(health_probe.wake_after());
4670            tokio::pin!(probe_sleep);
4671            let active_child = child.as_mut().expect("child checked above");
4672            tokio::select! {
4673                wait_result = active_child.wait() => {
4674                    // Every arm below that gives up on the CHILD must keep the
4675                    // supervision task itself alive (child = None, loop
4676                    // continues into command-serving mode). Returning here
4677                    // closes the command channel, which makes the module
4678                    // permanently unrestartable in-band: a clean child exit
4679                    // of an enabled module once wedged the fleet this way
4680                    // ('supervisor command channel is closed') and required a
4681                    // full daemon restart to recover.
4682                    let exit_report = match wait_result {
4683                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
4684                        Err(err) => {
4685                            active_child.drain_stderr(&spec.module_id).await;
4686                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4687                            // Every other exit path (on_child_exit's Clean/Crash arms,
4688                            // the reload-registration-failure path) records a terminal
4689                            // before moving on. Without one here, a module whose wait()
4690                            // itself errored (e.g. already reaped) leaves no terminal
4691                            // record at all -- an empty ring reads as "nothing died".
4692                            record_wait_error_terminal(
4693                                &spec.module_id,
4694                                &runtime.terminal_ring,
4695                                &runtime.spawn_events,
4696                            );
4697                            untrack_if_registration_released(
4698                                &process_liveness,
4699                                &registry,
4700                                &spec.module_id,
4701                                &snapshot,
4702                            );
4703                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
4704                            child = None;
4705                            continue;
4706                        }
4707                    };
4708                    active_child.drain_stderr(&spec.module_id).await;
4709
4710                    let next = on_child_exit(
4711                        &spec,
4712                        runtime.restart_policy,
4713                        &registry,
4714                        &snapshot,
4715                        &runtime.terminal_ring,
4716                        &runtime.spawn_events,
4717                        &runtime.child_roster,
4718                        exit_report,
4719                    ).await;
4720                    // The exit is recorded, so a daemon shutdown may stop
4721                    // waiting for this child (see `SupervisedChild::wait`).
4722                    active_child.release_roster();
4723                    match next {
4724                        NextAction::Stop { registration_released } => {
4725                            if registration_released {
4726                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4727                            }
4728                            child = None;
4729                        }
4730                        NextAction::Restart { schedule } => {
4731                            let delay = schedule.map_or(
4732                                runtime.restart_policy.delay_for_restart(0),
4733                                |schedule| schedule.delay,
4734                            );
4735                            if let Some(schedule) = schedule {
4736                                log_crash_respawn(&spec.module_id, schedule);
4737                            }
4738                            // The exited child is fully recorded at this point,
4739                            // so release it and count the backoff down in the
4740                            // command-serving branch below rather than sleeping
4741                            // here: commands cannot be received from inside this
4742                            // select arm, and an operator disable or drain that
4743                            // arrives during the backoff must cancel the pending
4744                            // respawn instead of waiting for it to spawn first.
4745                            child = None;
4746                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
4747                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
4748                        }
4749                    }
4750                }
4751                command = commands.recv() => {
4752                    let Some(command) = command else {
4753                        return;
4754                    };
4755                    if !handle_supervisor_command(
4756                        command,
4757                        &mut spec,
4758                        &mut runtime,
4759                        &registry,
4760                        &process_liveness,
4761                        &snapshot,
4762                        &mut child,
4763                        &mut commands,
4764                        &mut requeued,
4765                    ).await {
4766                        return;
4767                    }
4768                }
4769                _ = &mut probe_sleep => {
4770                    if health_probe.due() {
4771                        run_health_probe_cycle(
4772                            &spec,
4773                            &runtime,
4774                            &registry,
4775                            &process_liveness,
4776                            &snapshot,
4777                            &mut child,
4778                        ).await;
4779                        if child.is_some() {
4780                            health_probe.schedule_next(&spec, runtime.health.cadence);
4781                        }
4782                    }
4783                }
4784            }
4785        } else if let Some(pending) = pending_respawn {
4786            tokio::select! {
4787                _ = sleep_until(pending.deadline) => {
4788                    pending_respawn = None;
4789                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
4790                    // A command handled below while the backoff elapsed may
4791                    // have stopped the module; never respawn past an operator's
4792                    // disable or drain.
4793                    if !respawn_still_pending(&snapshot) {
4794                        continue;
4795                    }
4796                    // The daemon began shutting down during the backoff: the
4797                    // spawn would be refused anyway, and refusing it here
4798                    // leaves the module stopped instead of reporting a
4799                    // failed restart.
4800                    if runtime.child_roster.is_closed() {
4801                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4802                            state.state = ModuleState::Stopped;
4803                        });
4804                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
4805                        continue;
4806                    }
4807                    if let Err(err) = release_dead_registration(
4808                        &registry,
4809                        runtime.forwarding.as_deref(),
4810                        &snapshot,
4811                        &spec.module_id,
4812                    ).await {
4813                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
4814                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
4815                        continue;
4816                    }
4817
4818                    if matches!(pending.kind, RespawnKind::Reload) {
4819                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
4820                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
4821                        if let Some(reply) = reply { let _ = reply.send(result); }
4822                        continue;
4823                    }
4824                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
4825                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
4826                        Ok(next_child) => {
4827                            child = Some(next_child);
4828                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
4829                        }
4830                        Err(err) => {
4831                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4832                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4833                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
4834                        }
4835                    }
4836                }
4837                command = commands.recv() => {
4838                    let Some(command) = command else {
4839                        return;
4840                    };
4841                    if !handle_supervisor_command(
4842                        command,
4843                        &mut spec,
4844                        &mut runtime,
4845                        &registry,
4846                        &process_liveness,
4847                        &snapshot,
4848                        &mut child,
4849                        &mut commands,
4850                        &mut requeued,
4851                    ).await {
4852                        return;
4853                    }
4854                    // Reconcile the pending respawn with what the command did:
4855                    // a start may already have spawned a fresh child,
4856                    // while a disable or drain moved the snapshot out of the
4857                    // state the respawn was counting down from.
4858                    if child.is_some() || !respawn_still_pending(&snapshot) {
4859                        pending_respawn = None;
4860                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
4861                    }
4862                }
4863            }
4864        } else {
4865            let Some(command) = commands.recv().await else {
4866                return;
4867            };
4868            if !handle_supervisor_command(
4869                command,
4870                &mut spec,
4871                &mut runtime,
4872                &registry,
4873                &process_liveness,
4874                &snapshot,
4875                &mut child,
4876                &mut commands,
4877                &mut requeued,
4878            )
4879            .await
4880            {
4881                return;
4882            }
4883        }
4884    }
4885}
4886
4887fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
4888    info!(
4889        module_id,
4890        restart_in_window = schedule.restart_in_window,
4891        delay_ms = schedule.delay.as_millis() as u64,
4892        "respawning after crash"
4893    );
4894}
4895
4896/// Whether the respawn a backoff was counting down to is still wanted. A
4897/// disable or drain handled while the backoff elapsed moves the snapshot out
4898/// of `Restarting`, and the operator's stop must win over the pending respawn,
4899/// so every sleep-then-spawn path re-validates against the live snapshot
4900/// instead of assuming the state it left behind still holds.
4901fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
4902    matches!(
4903        lock_snapshot(snapshot),
4904        Ok(state) if state.enabled && state.state == ModuleState::Restarting
4905    )
4906}
4907
4908enum NextAction {
4909    Stop {
4910        registration_released: bool,
4911    },
4912    Restart {
4913        schedule: Option<CrashRestartSchedule>,
4914    },
4915}
4916
4917#[allow(clippy::too_many_arguments)]
4918async fn handle_supervisor_command(
4919    command: SupervisorCommand,
4920    spec: &mut ModuleSpec,
4921    runtime: &mut SupervisorRuntimeConfig,
4922    registry: &Registry,
4923    process_liveness: &SupervisorProcessLiveness,
4924    snapshot: &SharedSnapshot,
4925    child: &mut Option<SupervisedChild>,
4926    commands: &mut mpsc::Receiver<SupervisorCommand>,
4927    requeued: &mut VecDeque<SupervisorCommand>,
4928) -> bool {
4929    match command {
4930        SupervisorCommand::Drain { reply } => {
4931            // A plain stop runs no forwarding drain, so nothing reaches the
4932            // module over its connection before the wait: ask by signal.
4933            let result = drain_optional_child(
4934                &spec.module_id,
4935                spec.protocol,
4936                StopNotice::NotSent,
4937                registry,
4938                runtime.forwarding.as_deref(),
4939                snapshot,
4940                &runtime.terminal_ring,
4941                &runtime.spawn_events,
4942                child,
4943                runtime.drain_timeout,
4944                ModuleState::Stopped,
4945                None,
4946            )
4947            .await;
4948            let registration_released = result.is_ok();
4949            let _ = reply.send(result);
4950            if registration_released {
4951                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4952            }
4953            false
4954        }
4955        SupervisorCommand::Retire { reply } => {
4956            let result = async {
4957                let stop_notice = begin_forwarding_drain_if_configured(
4958                    spec,
4959                    runtime,
4960                    registry,
4961                    snapshot,
4962                    None,
4963                    RouteCloseReason::Disable,
4964                )
4965                .await?;
4966                drain_optional_child(
4967                    &spec.module_id,
4968                    spec.protocol,
4969                    stop_notice,
4970                    registry,
4971                    runtime.forwarding.as_deref(),
4972                    snapshot,
4973                    &runtime.terminal_ring,
4974                    &runtime.spawn_events,
4975                    child,
4976                    runtime.drain_timeout,
4977                    ModuleState::Stopped,
4978                    None,
4979                )
4980                .await
4981            }
4982            .await;
4983            let registration_released = result.is_ok();
4984            let _ = reply.send(result);
4985            if registration_released {
4986                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4987            }
4988            false
4989        }
4990        SupervisorCommand::Restart {
4991            drain_timeout_ms,
4992            received_at_generation,
4993            queued_at,
4994            reply,
4995        } => {
4996            // Without this line a restart that waited in the queue (behind a
4997            // health probe cycle or another command) was invisible: the log
4998            // showed only the drain timing out, minutes after the operator's call.
4999            info!(
5000                module_id = %spec.module_id,
5001                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
5002                "restart command dequeued"
5003            );
5004            // ACK AT INITIATION, not completion. The blocking form deadlocked any
5005            // caller whose own request lane rides the module being restarted: the
5006            // caller's in-flight request keeps the drain from quiescing, the drain
5007            // keeps the restart from completing, and the completion keeps the reply
5008            // from releasing the caller — so the drain always timed out and cut the
5009            // initiator with a GOODBYE, even on a healthy module. Replying once the
5010            // restart is validated lets a self-lane caller settle, which is exactly
5011            // what makes the drain succeed. Completion is observable via
5012            // supervisor.list / module status; a post-ack failure lands the module
5013            // in a visible terminal state below rather than in a reply nobody can
5014            // receive.
5015            let validation = match lock_snapshot(snapshot) {
5016                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
5017                    module_id: spec.module_id.clone(),
5018                }),
5019                Ok(_) => Ok(()),
5020                Err(err) => Err(err),
5021            };
5022            let initiated = validation.is_ok();
5023            let _ = reply.send(validation);
5024            // A restart asks for a fresh process. Commands run one at a time,
5025            // so a restart queued behind another restart (two operator calls
5026            // in quick succession) is dequeued the moment the first one has
5027            // spawned its replacement -- before that process has sent HELLO.
5028            // Running it would drain and kill the process the first restart
5029            // just produced, which is the opposite of what both callers asked
5030            // for. If a process spawned after this request was received is
5031            // still supervised, the request is already satisfied. Not when the
5032            // configuration changed since that spawn: then the newer process
5033            // predates the spec this restart may exist to apply.
5034            let satisfied_by_generation = if initiated && child.is_some() {
5035                lock_snapshot(snapshot).ok().and_then(|state| {
5036                    (state.spawn_generation > received_at_generation
5037                        && !state.configuration_updated_since_spawn)
5038                        .then_some(state.spawn_generation)
5039                })
5040            } else {
5041                None
5042            };
5043            let satisfied_by_pending = initiated
5044                && child.is_none()
5045                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
5046                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
5047                    if pending {
5048                        state.coalesced_restart_pending = true;
5049                    }
5050                    pending
5051                });
5052            if satisfied_by_pending {
5053                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
5054            } else if let Some(generation) = satisfied_by_generation {
5055                info!(
5056                    module_id = %spec.module_id,
5057                    received_at_generation,
5058                    "restart already satisfied by generation {generation}; not restarting again"
5059                );
5060            } else if initiated {
5061                // Precedence: this restart's operator override, else the module's
5062                // configured budget (already resolved into the runtime).
5063                let drain_timeout = drain_timeout_ms
5064                    .map(Duration::from_millis)
5065                    .unwrap_or(runtime.drain_timeout);
5066                if let Err(err) = restart_child(
5067                    spec,
5068                    runtime,
5069                    registry,
5070                    process_liveness,
5071                    snapshot,
5072                    child,
5073                    drain_timeout,
5074                )
5075                .await
5076                {
5077                    warn!(
5078                        module_id = %spec.module_id,
5079                        error = %err,
5080                        "operator restart failed after initiation ack; module state carries the outcome"
5081                    );
5082                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5083                        state.state = ModuleState::Failed;
5084                        clear_current_process_facts(state);
5085                    });
5086                }
5087            }
5088            true
5089        }
5090        SupervisorCommand::Reload { reply } => {
5091            let result =
5092                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
5093            if result.is_ok()
5094                && runtime
5095                    .scheduled_respawn
5096                    .lock()
5097                    .unwrap_or_else(|p| p.into_inner())
5098                    .as_ref()
5099                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
5100            {
5101                *runtime
5102                    .deferred_reload_reply
5103                    .lock()
5104                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
5105            } else {
5106                let _ = reply.send(result);
5107            }
5108            true
5109        }
5110        SupervisorCommand::SetEnabled { enabled, reply } => {
5111            let result = set_child_enabled(
5112                spec,
5113                runtime,
5114                registry,
5115                process_liveness,
5116                snapshot,
5117                child,
5118                enabled,
5119            )
5120            .await;
5121            let _ = reply.send(result);
5122            true
5123        }
5124        SupervisorCommand::UpdateConfiguration {
5125            spec: next_spec,
5126            health,
5127            drain_timeout_ms,
5128            reply,
5129        } => {
5130            if let Some(handle) = &runtime.supervisor_handle {
5131                handle.apply_identity_configuration(&next_spec);
5132            }
5133            *spec = next_spec;
5134            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5135                state.configuration_updated_since_spawn = true;
5136            });
5137            runtime.health = health;
5138            runtime.drain_timeout = drain_timeout_ms
5139                .map(Duration::from_millis)
5140                .unwrap_or(runtime.default_drain_timeout);
5141            *runtime
5142                .effective_drain_timeout
5143                .lock()
5144                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
5145            let _ = reply.send(());
5146            true
5147        }
5148        SupervisorCommand::Swap {
5149            ready_timeout,
5150            reply,
5151        } => {
5152            let end = swap::run_swap(
5153                spec,
5154                runtime,
5155                registry,
5156                process_liveness,
5157                snapshot,
5158                child,
5159                commands,
5160                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
5161                reply,
5162            )
5163            .await;
5164            requeued.extend(end.requeue);
5165            true
5166        }
5167    }
5168}
5169
5170async fn restart_child(
5171    spec: &ModuleSpec,
5172    runtime: &SupervisorRuntimeConfig,
5173    registry: &Registry,
5174    process_liveness: &SupervisorProcessLiveness,
5175    snapshot: &SharedSnapshot,
5176    child: &mut Option<SupervisedChild>,
5177    drain_timeout: Duration,
5178) -> Result<(), SuperviseError> {
5179    // Restart cycles a running module; it must not silently start a disabled one.
5180    if !lock_snapshot(snapshot)?.enabled {
5181        return Err(SuperviseError::Disabled {
5182            module_id: spec.module_id.clone(),
5183        });
5184    }
5185    let stop_notice = begin_forwarding_drain_with_timeout(
5186        spec,
5187        runtime,
5188        registry,
5189        snapshot,
5190        None,
5191        RouteCloseReason::Restart,
5192        drain_timeout,
5193    )
5194    .await?;
5195
5196    if child.is_some() {
5197        drain_optional_child(
5198            &spec.module_id,
5199            spec.protocol,
5200            stop_notice,
5201            registry,
5202            runtime.forwarding.as_deref(),
5203            snapshot,
5204            &runtime.terminal_ring,
5205            &runtime.spawn_events,
5206            child,
5207            drain_timeout,
5208            ModuleState::Restarting,
5209            Some(true),
5210        )
5211        .await?;
5212    } else {
5213        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5214            state.enabled = true;
5215            state.state = ModuleState::Restarting;
5216            clear_current_process_facts(state);
5217        })?;
5218        release_dead_registration(
5219            registry,
5220            runtime.forwarding.as_deref(),
5221            snapshot,
5222            &spec.module_id,
5223        )
5224        .await?;
5225    }
5226
5227    reset_restart_count(snapshot, &spec.module_id)?;
5228    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5229    schedule_respawn(
5230        runtime,
5231        snapshot,
5232        &spec.module_id,
5233        runtime.restart_policy.backoff,
5234        RespawnKind::Spawn,
5235    )
5236}
5237
5238async fn reload_child(
5239    spec: &ModuleSpec,
5240    runtime: &SupervisorRuntimeConfig,
5241    registry: &Registry,
5242    process_liveness: &SupervisorProcessLiveness,
5243    snapshot: &SharedSnapshot,
5244    child: &mut Option<SupervisedChild>,
5245) -> Result<(), SuperviseError> {
5246    // Reload cycles a running module; it must not silently start a disabled one.
5247    if !lock_snapshot(snapshot)?.enabled {
5248        return Err(SuperviseError::Disabled {
5249            module_id: spec.module_id.clone(),
5250        });
5251    }
5252    let stop_notice = begin_forwarding_drain(
5253        spec,
5254        runtime,
5255        registry,
5256        snapshot,
5257        Some(true),
5258        RouteCloseReason::Reload,
5259    )
5260    .await?;
5261
5262    if child.is_some() {
5263        drain_optional_child(
5264            &spec.module_id,
5265            spec.protocol,
5266            stop_notice,
5267            registry,
5268            runtime.forwarding.as_deref(),
5269            snapshot,
5270            &runtime.terminal_ring,
5271            &runtime.spawn_events,
5272            child,
5273            runtime.drain_timeout,
5274            ModuleState::Restarting,
5275            Some(true),
5276        )
5277        .await?;
5278    } else {
5279        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5280            state.enabled = true;
5281            state.state = ModuleState::Restarting;
5282            clear_current_process_facts(state);
5283        })?;
5284        release_dead_registration(
5285            registry,
5286            runtime.forwarding.as_deref(),
5287            snapshot,
5288            &spec.module_id,
5289        )
5290        .await?;
5291    }
5292
5293    reset_restart_count(snapshot, &spec.module_id)?;
5294    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5295    schedule_respawn(
5296        runtime,
5297        snapshot,
5298        &spec.module_id,
5299        runtime.restart_policy.backoff,
5300        RespawnKind::Reload,
5301    )
5302}
5303
5304async fn finish_reload_child(
5305    spec: &ModuleSpec,
5306    runtime: &SupervisorRuntimeConfig,
5307    registry: &Registry,
5308    process_liveness: &SupervisorProcessLiveness,
5309    snapshot: &SharedSnapshot,
5310    child: &mut Option<SupervisedChild>,
5311) -> Result<(), SuperviseError> {
5312    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5313    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5314        Ok(next_child) => next_child,
5315        Err(err) => {
5316            return handle_reload_spawn_failure(
5317                spec,
5318                runtime,
5319                process_liveness,
5320                snapshot,
5321                child,
5322                format!("new child failed to spawn: {err}"),
5323            )
5324            .await;
5325        }
5326    };
5327    *child = Some(next_child);
5328
5329    let wait_outcome = {
5330        let active_child = child.as_mut().expect("new reload child was just stored");
5331        wait_for_registration_after_reload(
5332            registry,
5333            &spec.module_id,
5334            snapshot,
5335            active_child,
5336            REGISTRY_RELEASE_TIMEOUT,
5337        )
5338        .await?
5339    };
5340
5341    match wait_outcome {
5342        RegistrationWaitOutcome::Registered => {
5343            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
5344            Ok(())
5345        }
5346        RegistrationWaitOutcome::Exited(exit_report) => {
5347            if let Some(active_child) = child.as_mut() {
5348                active_child.drain_stderr(&spec.module_id).await;
5349            }
5350            // Keep the reaped child's roster guard until its terminal is written.
5351            // Shutdown waits on that guard, not on the child Option used for respawn.
5352            let mut exited_child = child.take().expect("exited reload child is still stored");
5353            #[cfg(test)]
5354            if let Some(gate) = &runtime.test_reload_exit_record_gate {
5355                gate.reached.notify_one();
5356                gate.resume.notified().await;
5357            }
5358            let result = handle_reload_child_registration_failure(
5359                spec,
5360                runtime,
5361                registry,
5362                process_liveness,
5363                snapshot,
5364                child,
5365                ReloadRegistrationFailure {
5366                    exit_report: registration_failure_exit_report(exit_report),
5367                    reason: "new child exited before registering".to_string(),
5368                },
5369            )
5370            .await;
5371            exited_child.release_roster();
5372            result
5373        }
5374        RegistrationWaitOutcome::TimedOut => {
5375            let mut timed_out_child = child
5376                .take()
5377                .expect("timed-out reload child is still running");
5378            timed_out_child
5379                .start_kill()
5380                .map_err(|source| SuperviseError::Kill {
5381                    module_id: spec.module_id.clone(),
5382                    source,
5383                })?;
5384            let status = timed_out_child
5385                .wait()
5386                .await
5387                .map_err(|source| SuperviseError::Wait {
5388                    module_id: spec.module_id.clone(),
5389                    source,
5390                })?;
5391            timed_out_child.drain_stderr(&spec.module_id).await;
5392            handle_reload_child_registration_failure(
5393                spec,
5394                runtime,
5395                registry,
5396                process_liveness,
5397                snapshot,
5398                child,
5399                ReloadRegistrationFailure {
5400                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
5401                        snapshot,
5402                        &timed_out_child,
5403                        &status,
5404                    )),
5405                    reason: format!(
5406                        "new child did not register within {:?}",
5407                        REGISTRY_RELEASE_TIMEOUT
5408                    ),
5409                },
5410            )
5411            .await
5412        }
5413    }
5414}
5415
5416async fn set_child_enabled(
5417    spec: &ModuleSpec,
5418    runtime: &SupervisorRuntimeConfig,
5419    registry: &Registry,
5420    process_liveness: &SupervisorProcessLiveness,
5421    snapshot: &SharedSnapshot,
5422    child: &mut Option<SupervisedChild>,
5423    enabled: bool,
5424) -> Result<bool, SuperviseError> {
5425    let (current_enabled, current_state, respawn_pending) = {
5426        let state = lock_snapshot(snapshot)?;
5427        (state.enabled, state.state, state.respawn_pending)
5428    };
5429    // `start` (enable on an already-enabled module) heals TERMINAL states instead
5430    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
5431    // clean (Stopped) has no live process and no other in-band recovery — the
5432    // operator's start is the explicit recovery act and resets the budget. Without
5433    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
5434    // which the 2026-07-14 aft outage proved is a trap when the failed module is
5435    // the one providing every agent's shell.
5436    let revive_terminal = enabled
5437        && current_enabled
5438        && child.is_none()
5439        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
5440            || (current_state == ModuleState::Restarting && !respawn_pending));
5441    if current_enabled == enabled && !revive_terminal {
5442        return Ok(false);
5443    }
5444
5445    if enabled {
5446        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5447            state.enabled = true;
5448            state.state = ModuleState::Starting;
5449            clear_current_process_facts(state);
5450        })?;
5451        #[cfg(test)]
5452        if runtime.test_seed_stale_facts_before_enable_spawn {
5453            update_snapshot(snapshot, Some(&spec.module_id), |state| {
5454                state.process_alive = true;
5455                state.pid = Some(41);
5456                state.spawned_at_ms = Some(42);
5457                state.spawned_from = Some(PathBuf::from("/spawned/module"));
5458                state.spawned_file_identity = Some(SpawnedFileIdentity {
5459                    device: 43,
5460                    inode: 44,
5461                });
5462            })?;
5463        }
5464        release_dead_registration(
5465            registry,
5466            runtime.forwarding.as_deref(),
5467            snapshot,
5468            &spec.module_id,
5469        )
5470        .await?;
5471        reset_restart_count(snapshot, &spec.module_id)?;
5472        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5473        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5474            Ok(next_child) => next_child,
5475            Err(err) => {
5476                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5477                    state.state = ModuleState::Failed;
5478                    clear_current_process_facts(state);
5479                }) {
5480                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
5481                }
5482                process_liveness.untrack_if_current(&spec.module_id, snapshot);
5483                return Err(err);
5484            }
5485        };
5486        *child = Some(next_child);
5487        debug!(module_id = %spec.module_id, "supervised module enabled");
5488        Ok(true)
5489    } else {
5490        let stop_notice = begin_forwarding_drain_if_configured(
5491            spec,
5492            runtime,
5493            registry,
5494            snapshot,
5495            Some(false),
5496            RouteCloseReason::Disable,
5497        )
5498        .await?;
5499        drain_optional_child(
5500            &spec.module_id,
5501            spec.protocol,
5502            stop_notice,
5503            registry,
5504            runtime.forwarding.as_deref(),
5505            snapshot,
5506            &runtime.terminal_ring,
5507            &runtime.spawn_events,
5508            child,
5509            runtime.drain_timeout,
5510            ModuleState::Disabled,
5511            Some(false),
5512        )
5513        .await?;
5514        debug!(module_id = %spec.module_id, "supervised module disabled");
5515        Ok(true)
5516    }
5517}
5518
5519#[allow(clippy::too_many_arguments)]
5520async fn on_child_exit(
5521    spec: &ModuleSpec,
5522    policy: RestartPolicy,
5523    registry: &Registry,
5524    snapshot: &SharedSnapshot,
5525    terminal_ring: &Arc<Mutex<TerminalRing>>,
5526    spawn_events: &SpawnEventFeed,
5527    roster: &ChildRoster,
5528    exit_report: ExitReport,
5529) -> NextAction {
5530    // Once the daemon has begun shutting down, no exit is a crash to recover
5531    // from: the module is exiting because the daemon is going away (EOF on its
5532    // connection, or a service manager signalling the whole cgroup). Record it
5533    // as such and never schedule a respawn, which would only start a process
5534    // for the shutdown to end again.
5535    if roster.is_closed() {
5536        return on_child_exit_during_daemon_shutdown(
5537            spec,
5538            registry,
5539            snapshot,
5540            terminal_ring,
5541            spawn_events,
5542            exit_report,
5543        )
5544        .await;
5545    }
5546    // Every stop the supervisor itself asks for (operator stop, disable,
5547    // restart, reload, swap, a health restart, a drain that runs out of budget)
5548    // takes the child out of the supervise loop and reaps it in
5549    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
5550    // that reaches this point was not requested by the daemon.
5551    //
5552    // For a subc-wire module a clean exit is still a stop: those modules are
5553    // written to re-raise SIGTERM, so a stray outside signal already reads as a
5554    // crash, and exiting 0 is a deliberate choice the module made. A
5555    // `protocol: "none"` module is a stock program we cannot change, and many
5556    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
5557    // stop would leave the module down for good after any stray signal, so it
5558    // goes through the crash path instead: it spends restart budget, respawns
5559    // with the crash backoff, and ends `failed` when the budget runs out.
5560    let unrequested_clean_exit_of_protocol_none =
5561        exit_report.kind == ExitKind::Clean && spec.protocol == ModuleProtocol::None;
5562    match exit_report.kind {
5563        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
5564            info!(
5565                module_id = %spec.module_id,
5566                exit_code = ?exit_report.code,
5567                exit_signal = ?exit_report.signal,
5568                "supervised module exited cleanly"
5569            );
5570            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5571                state.state = ModuleState::Stopped;
5572                clear_current_process_facts(state);
5573                state.last_exit = Some(exit_report.clone());
5574            }) {
5575                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
5576            }
5577            record_terminal(
5578                &spec.module_id,
5579                terminal_ring,
5580                spawn_events,
5581                &exit_report,
5582                TerminalDisposition::Stopped,
5583            );
5584            let registration_released = match wait_for_registration_release(
5585                registry,
5586                &spec.module_id,
5587                REGISTRY_RELEASE_TIMEOUT,
5588            )
5589            .await
5590            {
5591                Ok(()) => true,
5592                Err(err) => {
5593                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
5594                    false
5595                }
5596            };
5597            NextAction::Stop {
5598                registration_released,
5599            }
5600        }
5601        ExitKind::Clean | ExitKind::Crash => {
5602            if unrequested_clean_exit_of_protocol_none {
5603                warn!(
5604                    module_id = %spec.module_id,
5605                    exit_code = ?exit_report.code,
5606                    exit_signal = ?exit_report.signal,
5607                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
5608                );
5609            } else {
5610                warn!(
5611                    module_id = %spec.module_id,
5612                    exit_code = ?exit_report.code,
5613                    exit_signal = ?exit_report.signal,
5614                    "supervised module exited abnormally (crash)"
5615                );
5616            }
5617            let mut restart_schedule = None;
5618            let mut disposition = TerminalDisposition::Disabled;
5619            // Set only when the budget is what stopped the module, so the
5620            // terminal record says which limit was hit rather than leaving
5621            // `failed` to be read as "crashed once, badly".
5622            let mut disposition_detail = None;
5623            let now = Instant::now();
5624            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5625                clear_current_process_facts(state);
5626                state.last_exit = Some(exit_report.clone());
5627                if state.enabled {
5628                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
5629                        state.state = ModuleState::Restarting;
5630                        restart_schedule = Some(schedule);
5631                        disposition = TerminalDisposition::Restarting;
5632                    } else {
5633                        state.state = ModuleState::Failed;
5634                        disposition = TerminalDisposition::Failed;
5635                        disposition_detail = Some(policy.budget_exhausted_detail());
5636                    }
5637                } else {
5638                    state.state = ModuleState::Disabled;
5639                    disposition = TerminalDisposition::Disabled;
5640                }
5641            }) {
5642                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
5643                return NextAction::Stop {
5644                    registration_released: false,
5645                };
5646            }
5647            if disposition_detail.is_some() {
5648                // The window is in the message, not only in the fields: this line
5649                // is read in a scrollback where a bare `max_restarts=3` reads as a
5650                // lifetime cap and sends the operator looking for three crashes
5651                // that never happened together.
5652                error!(
5653                    module_id = %spec.module_id,
5654                    max_restarts = policy.max_restarts,
5655                    window_secs = policy.window.as_secs(),
5656                    "module stopped: {}",
5657                    policy.budget_exhausted_detail()
5658                );
5659            }
5660            record_terminal_with_detail(
5661                &spec.module_id,
5662                terminal_ring,
5663                spawn_events,
5664                &exit_report,
5665                disposition,
5666                disposition_detail,
5667            );
5668
5669            if let Some(schedule) = restart_schedule {
5670                NextAction::Restart {
5671                    schedule: Some(schedule),
5672                }
5673            } else {
5674                let registration_released = match wait_for_registration_release(
5675                    registry,
5676                    &spec.module_id,
5677                    REGISTRY_RELEASE_TIMEOUT,
5678                )
5679                .await
5680                {
5681                    Ok(()) => true,
5682                    Err(err) => {
5683                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
5684                        false
5685                    }
5686                };
5687                NextAction::Stop {
5688                    registration_released,
5689                }
5690            }
5691        }
5692        ExitKind::DeliberateSeverance => {
5693            warn!(
5694                module_id = %spec.module_id,
5695                exit_code = ?exit_report.code,
5696                exit_signal = ?exit_report.signal,
5697                "supervised module exited after deliberate connection severance"
5698            );
5699            let mut should_restart = false;
5700            let mut disposition = TerminalDisposition::Disabled;
5701            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5702                clear_current_process_facts(state);
5703                state.last_exit = Some(exit_report.clone());
5704                state.lifetime_restarts += 1;
5705                if state.enabled {
5706                    state.state = ModuleState::Restarting;
5707                    should_restart = true;
5708                    disposition = TerminalDisposition::Restarting;
5709                } else {
5710                    state.state = ModuleState::Disabled;
5711                }
5712            }) {
5713                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
5714                return NextAction::Stop {
5715                    registration_released: false,
5716                };
5717            }
5718            record_terminal(
5719                &spec.module_id,
5720                terminal_ring,
5721                spawn_events,
5722                &exit_report,
5723                disposition,
5724            );
5725
5726            if should_restart {
5727                NextAction::Restart { schedule: None }
5728            } else {
5729                let registration_released = match wait_for_registration_release(
5730                    registry,
5731                    &spec.module_id,
5732                    REGISTRY_RELEASE_TIMEOUT,
5733                )
5734                .await
5735                {
5736                    Ok(()) => true,
5737                    Err(err) => {
5738                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
5739                        false
5740                    }
5741                };
5742                NextAction::Stop {
5743                    registration_released,
5744                }
5745            }
5746        }
5747    }
5748}
5749
5750async fn on_child_exit_during_daemon_shutdown(
5751    spec: &ModuleSpec,
5752    registry: &Registry,
5753    snapshot: &SharedSnapshot,
5754    terminal_ring: &Arc<Mutex<TerminalRing>>,
5755    spawn_events: &SpawnEventFeed,
5756    exit_report: ExitReport,
5757) -> NextAction {
5758    info!(
5759        module_id = %spec.module_id,
5760        exit_code = ?exit_report.code,
5761        exit_signal = ?exit_report.signal,
5762        exit_kind = ?exit_report.kind,
5763        "supervised module exited during daemon shutdown; not restarting it"
5764    );
5765    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5766        state.state = ModuleState::Stopped;
5767        clear_current_process_facts(state);
5768        state.last_exit = Some(exit_report.clone());
5769    }) {
5770        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
5771    }
5772    record_terminal(
5773        &spec.module_id,
5774        terminal_ring,
5775        spawn_events,
5776        &exit_report,
5777        TerminalDisposition::DaemonShutdown,
5778    );
5779    let registration_released =
5780        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
5781            .await
5782            .is_ok();
5783    NextAction::Stop {
5784        registration_released,
5785    }
5786}
5787
5788fn record_wait_error_terminal(
5789    module_id: &str,
5790    terminal_ring: &Arc<Mutex<TerminalRing>>,
5791    spawn_events: &SpawnEventFeed,
5792) {
5793    record_terminal(
5794        module_id,
5795        terminal_ring,
5796        spawn_events,
5797        &wait_error_exit_report(),
5798        TerminalDisposition::Failed,
5799    );
5800}
5801
5802fn record_terminal(
5803    module_id: &str,
5804    terminal_ring: &Arc<Mutex<TerminalRing>>,
5805    spawn_events: &SpawnEventFeed,
5806    exit_report: &ExitReport,
5807    disposition: TerminalDisposition,
5808) {
5809    record_terminal_with_detail(
5810        module_id,
5811        terminal_ring,
5812        spawn_events,
5813        exit_report,
5814        disposition,
5815        None,
5816    );
5817}
5818
5819/// The ring lock is held only to capture the read (see
5820/// `TerminalJournal::capture_read`), so this module's exits keep recording
5821/// while the journal files are read. Blocking: it reads files.
5822fn durable_terminal_history_of(
5823    terminal_ring: &Mutex<TerminalRing>,
5824    module_id: &str,
5825) -> subc_control::TerminalHistory {
5826    let read = terminal_ring
5827        .lock()
5828        .unwrap_or_else(|p| p.into_inner())
5829        .capture_durable_history();
5830    read.read(module_id)
5831}
5832
5833fn record_terminal_with_detail(
5834    module_id: &str,
5835    terminal_ring: &Arc<Mutex<TerminalRing>>,
5836    spawn_events: &SpawnEventFeed,
5837    exit_report: &ExitReport,
5838    disposition: TerminalDisposition,
5839    disposition_detail: Option<String>,
5840) {
5841    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
5842    let record = TerminalRecord {
5843        exit_code: exit_report.code,
5844        exit_signal: exit_report.signal,
5845        at_ms: exit_report.at_ms,
5846        disposition,
5847        exit_kind: exit_report.kind.into(),
5848        disposition_detail,
5849    };
5850    terminal_ring
5851        .lock()
5852        .unwrap_or_else(|poisoned| poisoned.into_inner())
5853        .record_exit(module_id, record);
5854}
5855
5856fn untrack_if_registration_released(
5857    process_liveness: &SupervisorProcessLiveness,
5858    registry: &Registry,
5859    module_id: &str,
5860    snapshot: &SharedSnapshot,
5861) {
5862    match registry.get_module(module_id) {
5863        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
5864        Ok(Some(_)) => {}
5865        Err(err) => {
5866            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
5867        }
5868    }
5869}
5870
5871/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
5872/// then apply the module's configured entries minus daemon-private capture keys.
5873///
5874/// Separated from `spawn_child` only so it can be asserted without spawning a
5875/// process — a duplicate of this logic in a test would pass while the real one
5876/// drifted, which is the defect class this function exists to avoid.
5877/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
5878/// nonce. A `protocol: "none"` module gets neither, because it cannot use
5879/// either and the argument would stop a stock binary from starting at all.
5880/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
5881///
5882/// The plain-spawn form, kept for the tests that assert its plan; spawns go
5883/// through [`apply_wire_spawn_args_for_role`].
5884#[cfg(test)]
5885fn apply_wire_spawn_args(
5886    command: &mut Command,
5887    spec: &ModuleSpec,
5888    connection_file_path: Option<&std::path::Path>,
5889    handle: Option<&SupervisorHandle>,
5890) -> Result<Option<NonceHandoff>, SuperviseError> {
5891    apply_wire_spawn_args_for_role(
5892        command,
5893        spec,
5894        connection_file_path,
5895        handle,
5896        SpawnRole::Plain,
5897    )
5898}
5899
5900/// The read end of a spawn's launch-nonce pipe, prepared by
5901/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
5902/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
5903/// handoff and keeps only the environment copy.
5904#[cfg(unix)]
5905type NonceHandoff = subc_os::LaunchNonceHandoff;
5906#[cfg(not(unix))]
5907type NonceHandoff = std::convert::Infallible;
5908
5909/// Prepare wire identity for a plain spawn or a swap candidate.
5910///
5911/// A plain spawn replaces the module's recorded nonce. A swap candidate records
5912/// a separate candidate token so the still-serving incumbent and its consumers
5913/// keep their nonce. Both records are installed before the process exists, so
5914/// the child's initial HELLO registration cannot arrive ahead of its nonce.
5915///
5916/// On Unix the nonce is delivered only through a pipe. It is written into
5917/// a pipe whose read end the child gets as descriptor 3, named by
5918/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
5919/// process of the same user cannot read it with `ps eww`. That handoff is
5920/// returned rather than installed here, because installing it replaces
5921/// whatever the child has at descriptor 3 and so must be the last pre-exec
5922/// step, after the Linux cgroup placement that the caller registers later.
5923/// Windows retains the environment handoff until restricted handle inheritance
5924/// can be implemented outside std's process primitives.
5925fn apply_wire_spawn_args_for_role(
5926    command: &mut Command,
5927    spec: &ModuleSpec,
5928    connection_file_path: Option<&std::path::Path>,
5929    handle: Option<&SupervisorHandle>,
5930    role: SpawnRole,
5931) -> Result<Option<NonceHandoff>, SuperviseError> {
5932    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
5933    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
5934    // included: a daemon started from a module's process tree inherits it,
5935    // and passing it on would point the child at a descriptor it does not
5936    // have.
5937    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
5938    // Remove inherited or configured copies too: withholding must mean absent.
5939    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
5940    if spec.protocol == ModuleProtocol::None {
5941        return Ok(None);
5942    }
5943    if let Some(connection_file_path) = connection_file_path {
5944        command.arg(SUBC_ARG).arg(connection_file_path);
5945    }
5946
5947    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
5948    // route.open attestation. Reserved modules additionally use the same nonce
5949    // for HELLO id-squatting protection. A respawn rotates both records.
5950    let nonce = generate_launch_nonce()?;
5951    if let Some(handle) = handle {
5952        match role {
5953            SpawnRole::Plain => {
5954                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
5955                if spec.reserved {
5956                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
5957                }
5958            }
5959            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
5960        }
5961    }
5962    #[cfg(unix)]
5963    let handoff = {
5964        let handoff =
5965            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
5966                program: spec.program.clone(),
5967                source,
5968                cgroup_path: None,
5969            })?;
5970        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
5971        Some(handoff)
5972    };
5973    #[cfg(not(unix))]
5974    let handoff = None;
5975    // Windows keeps the environment copy: std cannot restrict an inherited pipe
5976    // handle to this child without leaking it to concurrently spawned processes.
5977    #[cfg(not(unix))]
5978    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
5979    Ok(handoff)
5980}
5981
5982fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
5983    command.env_remove(CK_LOG_ENV);
5984    // The spawn role is the supervisor's to set, and only on a swap candidate
5985    // (see `apply_spawn_role`). Removing it here, rather than just not setting
5986    // it, is what makes it absent on a plain spawn: the daemon's own
5987    // environment could carry it, and so could a spec built outside daemon
5988    // config (config refuses it as an `env` key). A module reading it on a
5989    // plain restart would pick the long swap budget and leave callers waiting.
5990    command.env_remove(SUBC_SPAWN_ROLE_ENV);
5991    for (key, value) in &spec.env {
5992        // cortexkit-log currently exposes retention only as a Rust struct, not
5993        // environment names. These values are daemon-private sink metadata and
5994        // must never become a public child-process contract by being inherited.
5995        if matches!(
5996            key.as_str(),
5997            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
5998        ) || key == SUBC_SPAWN_ROLE_ENV
5999        {
6000            continue;
6001        }
6002        command.env(key, value);
6003    }
6004}
6005
6006/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
6007/// of a blue/green swap.
6008#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6009enum SpawnRole {
6010    Plain,
6011    SwapCandidate,
6012}
6013
6014/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
6015/// `apply_child_env` has already removed the variable for every spawn.
6016fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
6017    if role == SpawnRole::SwapCandidate {
6018        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
6019    }
6020}
6021
6022fn spawn_child(
6023    spec: &ModuleSpec,
6024    connection_file_path: Option<&std::path::Path>,
6025    handle: Option<&SupervisorHandle>,
6026    ring: &Arc<Mutex<StderrRing>>,
6027    capture_logs_dir: Option<&std::path::Path>,
6028    roster: &ChildRoster,
6029    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
6030) -> Result<SupervisedChild, SuperviseError> {
6031    spawn_child_in_slot(
6032        spec,
6033        connection_file_path,
6034        handle,
6035        ring,
6036        capture_logs_dir,
6037        roster,
6038        #[cfg(target_os = "linux")]
6039        cgroup_placement,
6040        SpawnRole::Plain,
6041        false,
6042    )
6043}
6044
6045/// Spawn one process of `spec` into a slot.
6046///
6047/// `alternate_slot` picks the process's cgroup name (see `swap::cgroup_name`).
6048/// A swap candidate needs a different cgroup from the process it is replacing,
6049/// which is still alive: in the same cgroup the two would be one kill domain,
6050/// and killing a failed candidate could take the incumbent with it.
6051///
6052/// The stderr capture file is `<module_id>.stderr.log` for every process of
6053/// the module, whichever slot it is in, because that is the one file
6054/// `ck module logs` reads. During a swap's overlap both processes append to it;
6055/// the daemon writes whole lines, so the two interleave by line, which is also
6056/// the merged view an operator wants while a swap runs.
6057#[allow(clippy::too_many_arguments)]
6058fn spawn_child_in_slot(
6059    spec: &ModuleSpec,
6060    connection_file_path: Option<&std::path::Path>,
6061    handle: Option<&SupervisorHandle>,
6062    ring: &Arc<Mutex<StderrRing>>,
6063    capture_logs_dir: Option<&std::path::Path>,
6064    roster: &ChildRoster,
6065    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
6066    role: SpawnRole,
6067    alternate_slot: bool,
6068) -> Result<SupervisedChild, SuperviseError> {
6069    if roster.is_closed() {
6070        return Err(SuperviseError::Spawn {
6071            program: spec.program.clone(),
6072            source: io::Error::other("the daemon is shutting down; not starting a new process"),
6073            cgroup_path: None,
6074        });
6075    }
6076    #[cfg(target_os = "linux")]
6077    let cgroup_name = swap::cgroup_name(&spec.module_id, alternate_slot);
6078    #[cfg(not(target_os = "linux"))]
6079    let _ = alternate_slot;
6080    let mut command = Command::new(&spec.program);
6081    command.args(&spec.args);
6082    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
6083    // that is the whole of the intent, so remove that one key rather than the
6084    // environment.
6085    //
6086    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
6087    // and took the POSIX environment with it. Modules spawned that way had no
6088    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
6089    // logging:
6090    //
6091    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
6092    //     both unset it fell back to the temp dir alone and `ck` could not find
6093    //     a daemon running on the same machine from inside any module's process
6094    //     tree — reporting a path the file has never lived at, which reads as
6095    //     "the daemon did not write its file".
6096    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
6097    //     the RELATIVE `.local/share`, so a module deriving its own store path
6098    //     resolved it against its own CWD. That is the store-fragmentation
6099    //     defect the daemon already refuses in config (`parse_doc` rejects a
6100    //     relative `storage.data_home`) arriving by derivation instead.
6101    //   * anything a module spawns inherited it: git without ~/.gitconfig,
6102    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
6103    //     quietly rather than erroring.
6104    //
6105    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
6106    // offered one candidate under /tmp while the file sat in /run/user/1000.
6107    //
6108    // A configured module is unaffected either way: `module_spec()` puts the
6109    // resolved CK_LOG into `spec.env`, which is applied below and therefore
6110    // wins over anything ambient.
6111    apply_child_env(&mut command, spec);
6112    apply_spawn_role(&mut command, role);
6113    let nonce_handoff =
6114        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
6115
6116    #[cfg(target_os = "linux")]
6117    let cgroup_path = cgroup_placement
6118        .map(|placement| placement.module_path(&cgroup_name))
6119        .transpose()
6120        .map_err(|source| SuperviseError::Cgroup {
6121            module_id: spec.module_id.clone(),
6122            source,
6123        })?;
6124    #[cfg(not(target_os = "linux"))]
6125    let cgroup_path: Option<PathBuf> = None;
6126    #[cfg(target_os = "linux")]
6127    if let Some(path) = &cgroup_path {
6128        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
6129            if let Some(placement) = cgroup_placement {
6130                remove_module_cgroup(placement, &cgroup_name);
6131            }
6132            return Err(error);
6133        }
6134    }
6135
6136    let output_sink = if let Some(logs_dir) = capture_logs_dir {
6137        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
6138        match ChildOutputSink::open(&path, capture_retention(spec)) {
6139            Ok(sink) => sink,
6140            Err(error) => {
6141                warn!(
6142                    module_id = %spec.module_id,
6143                    path = %path.display(),
6144                    error = %error,
6145                    "could not open child output capture file; forwarding to stderr"
6146                );
6147                ChildOutputSink::Stderr
6148            }
6149        }
6150    } else {
6151        ChildOutputSink::Stderr
6152    };
6153
6154    command.stdout(Stdio::piped());
6155    command.stderr(Stdio::piped());
6156    command.kill_on_drop(true);
6157    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
6158    // before exec). In the daemon's group, a service manager that kills the
6159    // job's process group when the daemon exits (launchd's default) killed
6160    // every module at the same moment its control connection closed, so no
6161    // module ever ran its EOF teardown on a daemon stop. Outside that group a
6162    // module is reached only by the daemon: the EOF it sees when its
6163    // connection closes, and the bounded stop in `child_roster` for anything
6164    // still running after that. On Linux this composes with the cgroup
6165    // placement above: that is a pre_exec write to cgroup.procs, std performs
6166    // setpgid in the child before running pre_exec callbacks, and the two
6167    // change independent process attributes.
6168    //
6169    // stdin is /dev/null because a process outside the terminal's foreground
6170    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
6171    // by hand would otherwise hand down. Under a service manager stdin is
6172    // already /dev/null.
6173    #[cfg(unix)]
6174    command.process_group(0);
6175    command.stdin(Stdio::null());
6176    // The LAST pre-exec step, after the cgroup placement above: installing the
6177    // nonce at descriptor 3 replaces whatever the child had there, which could
6178    // be the descriptor an earlier step writes through.
6179    #[cfg(unix)]
6180    if let Some(handoff) = nonce_handoff {
6181        handoff.install_last(command.as_std_mut());
6182    }
6183    #[cfg(not(unix))]
6184    let _ = nonce_handoff;
6185
6186    // Containment, step 1 of 3 (issue #109): create the child suspended so it
6187    // cannot run a single instruction -- and therefore cannot spawn a
6188    // grandchild -- before it is in the job. See `contain_spawned_child` for the
6189    // other two steps and why the window matters.
6190    #[cfg(windows)]
6191    subc_jobobject::suspend_on_create_async(&mut command);
6192    let mut child = match command.spawn() {
6193        Ok(child) => child,
6194        Err(source) => {
6195            #[cfg(target_os = "linux")]
6196            if let Some(placement) = cgroup_placement {
6197                remove_module_cgroup(placement, &cgroup_name);
6198            }
6199            return Err(SuperviseError::Spawn {
6200                program: spec.program.clone(),
6201                source,
6202                cgroup_path,
6203            });
6204        }
6205    };
6206
6207    // Containment, steps 2 and 3: assign while suspended, then resume.
6208    #[cfg(windows)]
6209    let job = contain_spawned_child(&child, spec)?;
6210    let spawned_at_ms = unix_ms_now();
6211    let spawned_from = spec.program.clone();
6212    let spawned_file_identity = spawned_file_identity(&spawned_from);
6213    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
6214        program: spec.program.clone(),
6215        source: io::Error::other("spawned child exposed no live pid"),
6216        cgroup_path: cgroup_path.clone(),
6217    })?;
6218    let process_start_time = crate::provenance::process_start_time(pid);
6219    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
6220    // The executable identity is the spawned path's, read above, not the
6221    // running image's: right after spawn the child may not have finished its
6222    // exec yet and would still report this daemon's own image.
6223    #[cfg(target_os = "linux")]
6224    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
6225    #[cfg(not(target_os = "linux"))]
6226    let recorded_cgroup_name = None;
6227    let roster_guard = roster.admit(
6228        spec.module_id.clone(),
6229        pid,
6230        spec.protocol,
6231        process_start_time,
6232        crate::child_roster::RecordedIdentity {
6233            start_time: subc_os::start_time(pid),
6234            executable: spawned_file_identity.map(|identity| {
6235                crate::live_children::ExecutableIdentity {
6236                    device: identity.device,
6237                    inode: identity.inode,
6238                }
6239            }),
6240            cgroup_name: recorded_cgroup_name,
6241            #[cfg(target_os = "linux")]
6242            cgroup_placement: cgroup_placement.cloned(),
6243        },
6244    );
6245    // The check at the top of this function can pass just before daemon
6246    // shutdown begins, and the process is only in the roster from here on.
6247    // The shutdown stop returns as soon as it finds the roster empty, so a
6248    // process admitted after that look would outlive the daemon. The roster
6249    // is closed before the stop first reads it and admission happens under
6250    // the roster's lock, so either the stop sees this process or this check
6251    // sees the roster closed: end the process now rather than start a module
6252    // the daemon is about to stop.
6253    if roster.is_closed() {
6254        // This child was never admitted, so there is no module protocol shutdown to wait for.
6255        #[cfg(target_os = "linux")]
6256        kill_module_cgroup(cgroup_placement, &cgroup_name);
6257        if let Err(error) = child.start_kill() {
6258            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
6259        }
6260        drop(roster_guard);
6261        return Err(SuperviseError::Spawn {
6262            program: spec.program.clone(),
6263            source: io::Error::other(
6264                "the daemon began shutting down while this process was starting; ended it",
6265            ),
6266            cgroup_path,
6267        });
6268    }
6269
6270    let stdout_pump = match child.stdout.take() {
6271        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
6272        None => {
6273            warn!(
6274                module_id = %spec.module_id,
6275                "spawned child exposed no stdout pipe; file capture will be incomplete"
6276            );
6277            None
6278        }
6279    };
6280    let stderr_pump = match child.stderr.take() {
6281        Some(stderr) => {
6282            let generation = ring
6283                .lock()
6284                .unwrap_or_else(|poisoned| poisoned.into_inner())
6285                .begin_process();
6286            Some(StderrPump {
6287                task: tokio::spawn(pump_stderr_to(
6288                    stderr,
6289                    Arc::clone(ring),
6290                    generation,
6291                    output_sink,
6292                )),
6293                generation,
6294            })
6295        }
6296        None => {
6297            // Spawning succeeded but the pipe did not materialise. Recording it as
6298            // uncaptured keeps the tail honest: the alternative is an empty tail
6299            // that reads as a module which printed nothing.
6300            ring.lock()
6301                .unwrap_or_else(|poisoned| poisoned.into_inner())
6302                .mark_not_captured("stderr pipe was not available on spawn");
6303            warn!(
6304                module_id = %spec.module_id,
6305                "spawned child exposed no stderr pipe; tail will be unavailable"
6306            );
6307            None
6308        }
6309    };
6310
6311    Ok(SupervisedChild {
6312        child,
6313        #[cfg(target_os = "linux")]
6314        module_id: cgroup_name,
6315        #[cfg(target_os = "linux")]
6316        cgroup_placement: cgroup_placement.cloned(),
6317        #[cfg(windows)]
6318        job,
6319        stdout_pump,
6320        stderr_pump,
6321        stderr_ring: Arc::clone(ring),
6322        spawned_at_ms,
6323        spawned_from,
6324        spawned_file_identity,
6325        process_start_time,
6326        process_identity,
6327        pid,
6328        roster_guard: Some(roster_guard),
6329    })
6330}
6331
6332#[cfg(target_os = "linux")]
6333pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
6334    use subc_cgroup::KillOutcome;
6335    match subc_cgroup::kill_module(placement, module_id) {
6336        KillOutcome::Killed => {}
6337        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
6338            debug!(
6339                module_id,
6340                "cgroup tree kill unavailable; using direct-child kill"
6341            );
6342        }
6343        KillOutcome::IoError { path, error } => {
6344            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
6345        }
6346    }
6347}
6348
6349/// Contain a freshly spawned Windows child and start it.
6350///
6351/// Steps 2 and 3 of the suspended-create contract: the job is created and the
6352/// child assigned **while it is still suspended** (step 1 is
6353/// `suspend_on_create_async` at the spawn site), then the child is resumed.
6354///
6355/// A child that is never resumed hangs forever holding a pid, so a resume
6356/// failure kills the child and fails the spawn rather than returning a
6357/// `SupervisedChild` that can never run.
6358///
6359/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
6360/// it did before this existed, whereas refusing to start one would be a new
6361/// outage. It is logged at warn because it means a helper process could leak.
6362#[cfg(windows)]
6363fn contain_spawned_child(
6364    child: &Child,
6365    spec: &ModuleSpec,
6366) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
6367    let module_id = spec.module_id.as_str();
6368    let Some(pid) = child.id() else {
6369        // The child exited between spawn and here. Its tree, if it made one,
6370        // needs no containment: nothing is left to contain.
6371        warn!(
6372            module_id,
6373            "spawned child had already exited before containment; no job object attached"
6374        );
6375        return Ok(None);
6376    };
6377
6378    let job = match subc_jobobject::JobObject::new() {
6379        Ok(job) => job,
6380        Err(source) => {
6381            warn!(
6382                module_id,
6383                error = %source,
6384                "could not create a job object; this module's helper processes will not be \
6385                 reaped on teardown"
6386            );
6387            // Resume regardless: leaving the child suspended would turn a
6388            // containment gap into a hung module.
6389            resume_suspended_child(pid, spec)?;
6390            return Ok(None);
6391        }
6392    };
6393
6394    if let Err(source) = job.assign(child) {
6395        warn!(
6396            module_id,
6397            error = %source,
6398            "could not assign the child to its job object; this module's helper processes \
6399             will not be reaped on teardown"
6400        );
6401        resume_suspended_child(pid, spec)?;
6402        return Ok(None);
6403    }
6404
6405    resume_suspended_child(pid, spec)?;
6406    Ok(Some(job))
6407}
6408
6409/// Resume a suspended child, killing it if it cannot be started.
6410///
6411/// A suspended process holds a pid and does nothing, so there is no useful
6412/// state to return: the caller gets an error and the spawn fails.
6413#[cfg(windows)]
6414fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
6415    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
6416        // Kill it here rather than leaving a suspended process for the caller
6417        // to notice; `kill_on_drop` would eventually do this, but the module
6418        // would have been reported as running in between.
6419        let _ = std::process::Command::new("taskkill.exe")
6420            .args(["/PID", &pid.to_string(), "/T", "/F"])
6421            .stdin(Stdio::null())
6422            .stdout(Stdio::null())
6423            .stderr(Stdio::null())
6424            .status();
6425        return Err(SuperviseError::Spawn {
6426            program: spec.program.clone(),
6427            source,
6428            cgroup_path: None,
6429        });
6430    }
6431    Ok(())
6432}
6433
6434#[cfg(target_os = "linux")]
6435fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
6436    match placement.remove_module(module_id) {
6437        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
6438        Err(error) => warn!(
6439            module_id,
6440            error = %error,
6441            "could not remove module cgroup after process exit; continuing teardown"
6442        ),
6443    }
6444}
6445
6446#[cfg(target_os = "linux")]
6447fn apply_cgroup_placement(
6448    command: &mut Command,
6449    spec: &ModuleSpec,
6450    path: &std::path::Path,
6451) -> Result<(), SuperviseError> {
6452    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
6453        module_id: spec.module_id.clone(),
6454        source,
6455    })
6456}
6457
6458fn capture_retention(spec: &ModuleSpec) -> Retention {
6459    let defaults = Retention::default();
6460    let value = |name: &str| {
6461        spec.env
6462            .iter()
6463            .rev()
6464            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
6465    };
6466    Retention {
6467        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
6468            .and_then(|value| value.parse().ok())
6469            .unwrap_or(defaults.max_file_mb),
6470        keep: value(CAPTURE_KEEP_ENV)
6471            .and_then(|value| value.parse().ok())
6472            .unwrap_or(defaults.keep),
6473        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
6474            .and_then(|value| value.parse().ok())
6475            .unwrap_or(defaults.max_age_days),
6476    }
6477}
6478
6479/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
6480/// module's registration to the exact process the supervisor spawned.
6481fn generate_launch_nonce() -> Result<String, SuperviseError> {
6482    let mut bytes = [0u8; 32];
6483    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
6484        reason: source.to_string(),
6485    })?;
6486    let mut hex = String::with_capacity(64);
6487    for b in bytes {
6488        use std::fmt::Write;
6489        let _ = write!(hex, "{b:02x}");
6490    }
6491    Ok(hex)
6492}
6493
6494/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
6495/// signal about how many leading bytes matched.
6496fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
6497    if a.len() != b.len() {
6498        return false;
6499    }
6500    let mut diff = 0u8;
6501    for (x, y) in a.iter().zip(b.iter()) {
6502        diff |= x ^ y;
6503    }
6504    diff == 0
6505}
6506
6507fn spawn_and_mark_running(
6508    spec: &ModuleSpec,
6509    runtime: &SupervisorRuntimeConfig,
6510    snapshot: &SharedSnapshot,
6511) -> Result<SupervisedChild, SuperviseError> {
6512    let child = spawn_child(
6513        spec,
6514        runtime.connection_file_path.as_deref(),
6515        runtime.supervisor_handle.as_ref(),
6516        &runtime.stderr_ring,
6517        runtime.capture_logs_dir.as_deref(),
6518        &runtime.child_roster,
6519        #[cfg(target_os = "linux")]
6520        runtime.cgroup_placement.as_ref(),
6521    )?;
6522    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
6523    Ok(child)
6524}
6525
6526enum RegistrationWaitOutcome {
6527    Registered,
6528    Exited(ExitReport),
6529    TimedOut,
6530}
6531
6532struct ReloadRegistrationFailure {
6533    exit_report: ExitReport,
6534    reason: String,
6535}
6536
6537#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6538enum BusyGaugeObservation {
6539    Quiescent,
6540    Busy,
6541    Omitted,
6542}
6543
6544fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
6545    let Some(metrics) = metrics.and_then(Value::as_object) else {
6546        return BusyGaugeObservation::Omitted;
6547    };
6548    let mut sum = 0u128;
6549    for gauge in gauges {
6550        let Some(value) = metrics.get(gauge) else {
6551            return BusyGaugeObservation::Omitted;
6552        };
6553        let Some(value) = value.as_u64() else {
6554            return BusyGaugeObservation::Busy;
6555        };
6556        sum = sum.saturating_add(u128::from(value));
6557    }
6558    if sum == 0 {
6559        BusyGaugeObservation::Quiescent
6560    } else {
6561        BusyGaugeObservation::Busy
6562    }
6563}
6564
6565fn declared_busy_gauges(
6566    registry: &Registry,
6567    module_id: &str,
6568) -> Result<Vec<String>, SuperviseError> {
6569    busy_gauges_of(
6570        registry
6571            .get_module(module_id)
6572            .map_err(SuperviseError::Registry)?,
6573    )
6574}
6575
6576/// [`declared_busy_gauges`] for the registration a connection holds, in any
6577/// slot: after cutover the incumbent is no longer the id's active
6578/// registration, and its own manifest is the one that names its gauges.
6579fn declared_busy_gauges_for_connection(
6580    registry: &Registry,
6581    connection_id: ConnectionId,
6582) -> Result<Vec<String>, SuperviseError> {
6583    busy_gauges_of(
6584        registry
6585            .get_module_by_connection(connection_id)
6586            .map_err(SuperviseError::Registry)?,
6587    )
6588}
6589
6590fn busy_gauges_of(
6591    registration: Option<crate::registry::ModuleRegistration>,
6592) -> Result<Vec<String>, SuperviseError> {
6593    let Some(registration) = registration else {
6594        return Ok(Vec::new());
6595    };
6596    let Some(self_signals) = registration.manifest.self_signals else {
6597        return Ok(Vec::new());
6598    };
6599
6600    let mut gauges = Vec::new();
6601    for declaration in self_signals {
6602        if declaration.kind != SelfSignalKind::Busy {
6603            continue;
6604        }
6605        match declaration.anchored_to {
6606            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
6607                gauges.extend(declared)
6608            }
6609            _ => {
6610                // An invalid Busy anchor is fail-safe: the empty name cannot be
6611                // present in a conforming health report, so this drain stays busy.
6612                gauges.push(String::new());
6613            }
6614        }
6615    }
6616    Ok(gauges)
6617}
6618
6619/// Wait for `endpoint` to have nothing in flight and, when the module declares
6620/// busy gauges, for a health probe to report them quiet. The probe is addressed
6621/// by `scope`: a swap's superseded incumbent must be asked about its own
6622/// gauges, and by module id the probe would reach the promoted candidate.
6623async fn wait_for_forwarding_quiescence(
6624    forwarding: &ForwardingTable,
6625    module_id: &str,
6626    runtime: &SupervisorRuntimeConfig,
6627    endpoint: crate::ModuleEndpointId,
6628    deadline: Instant,
6629    busy_gauges: &[String],
6630    scope: DrainScope,
6631) -> Result<bool, SuperviseError> {
6632    let mut gauges_quiescent = busy_gauges.is_empty();
6633    let mut next_probe_at = Instant::now();
6634    let mut omission_counted = false;
6635
6636    loop {
6637        let now = Instant::now();
6638        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
6639            let report = match scope {
6640                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
6641                DrainScope::Endpoint(endpoint) => {
6642                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
6643                }
6644            };
6645            gauges_quiescent = match report {
6646                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
6647                    BusyGaugeObservation::Quiescent => true,
6648                    BusyGaugeObservation::Busy => false,
6649                    BusyGaugeObservation::Omitted => {
6650                        if !omission_counted {
6651                            forwarding
6652                                .counters()
6653                                .increment_drains_with_undeclared_gauge();
6654                            omission_counted = true;
6655                        }
6656                        false
6657                    }
6658                },
6659                Err(err) => {
6660                    warn!(
6661                        module_id,
6662                        error = %err,
6663                        "drain health.check did not produce declared busy gauges; treating module as busy"
6664                    );
6665                    false
6666                }
6667            };
6668            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
6669        }
6670
6671        let in_flight = forwarding
6672            .endpoint_in_flight_count(endpoint)
6673            .map_err(SuperviseError::Forwarding)?;
6674        if in_flight == 0 && gauges_quiescent {
6675            return Ok(true);
6676        }
6677
6678        let now = Instant::now();
6679        if now >= deadline {
6680            return Ok(false);
6681        }
6682        let mut wait = deadline
6683            .saturating_duration_since(now)
6684            .min(REGISTRY_RELEASE_POLL);
6685        if !busy_gauges.is_empty() {
6686            wait = wait.min(next_probe_at.saturating_duration_since(now));
6687        }
6688        sleep(wait).await;
6689    }
6690}
6691
6692/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
6693///
6694/// `Ok` is always honest and passed straight through -- the wait actually measured
6695/// in-flight state. `Err` means the wait produced no measurement at all (the
6696/// forwarding table's lock was poisoned), so `false` is reported as the one honest
6697/// constant: the drain did not complete. Never recomputed from route state, never a
6698/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
6699fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
6700    match wait_result {
6701        Ok(drained) => *drained,
6702        Err(_) => false,
6703    }
6704}
6705
6706fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
6707    for released in released_routes {
6708        let frame = match Frame::build_with_version(
6709            released.negotiated_ver,
6710            FrameType::Goodbye,
6711            control_flags(),
6712            released.channel,
6713            released.epoch,
6714            0,
6715            Vec::new(),
6716        ) {
6717            Ok(frame) => frame,
6718            Err(err) => {
6719                warn!(
6720                    route_channel = released.channel,
6721                    error = %err,
6722                    "failed to build supervisor drain route GOODBYE frame"
6723                );
6724                continue;
6725            }
6726        };
6727        if !released.close_on_delivery_failure() {
6728            crate::forwarding::send_module_route_goodbye(
6729                &forwarding.counters(),
6730                &released.sink,
6731                frame,
6732                released.module_id.as_deref(),
6733                "supervisor drain",
6734            );
6735            continue;
6736        }
6737        if let Err(err) = released.sink.try_send(frame) {
6738            warn!(
6739                target_connection_id = released.connection_id.get(),
6740                route_channel = released.channel,
6741                error = %err,
6742                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
6743            );
6744            let _ = forwarding.escalate_client_delivery_failure(
6745                released.connection_id,
6746                released.channel,
6747                released.epoch,
6748                CloseReason::new(
6749                    "route_goodbye_delivery_failed",
6750                    format!(
6751                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
6752                        released.channel
6753                    ),
6754                ),
6755                crate::forwarding::UndeliveredFrame {
6756                    module_id: released.module_id.as_deref(),
6757                    sink: &released.sink,
6758                },
6759            );
6760        }
6761    }
6762}
6763
6764fn send_module_draining(
6765    module_id: &str,
6766    reason: RouteCloseReason,
6767    deadline_ms: u64,
6768    target: &ModuleDrainTarget,
6769) {
6770    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
6771        reason,
6772        deadline_ms,
6773    }) {
6774        Ok(body) => body,
6775        Err(err) => {
6776            warn!(
6777                module_id,
6778                error = %err,
6779                "failed to encode module draining command"
6780            );
6781            return;
6782        }
6783    };
6784    let frame = match Frame::build_with_version(
6785        target.negotiated_ver,
6786        FrameType::Push,
6787        control_flags(),
6788        0,
6789        0,
6790        0,
6791        body,
6792    ) {
6793        Ok(frame) => frame,
6794        Err(err) => {
6795            warn!(
6796                module_id,
6797                error = %err,
6798                "failed to build module draining command frame"
6799            );
6800            return;
6801        }
6802    };
6803    if let Err(err) = target.sink.try_send(frame) {
6804        warn!(
6805            module_id,
6806            target_connection_id = target.endpoint.connection_id.get(),
6807            error = %err,
6808            "module draining command was not delivered to peer"
6809        );
6810    }
6811}
6812
6813/// The channel-0 GOODBYE that tells a module its stop is planned.
6814fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
6815    match Frame::build_with_version(
6816        negotiated_ver,
6817        FrameType::Goodbye,
6818        control_flags(),
6819        0,
6820        0,
6821        0,
6822        Vec::new(),
6823    ) {
6824        Ok(frame) => Some(frame),
6825        Err(err) => {
6826            warn!(
6827                module_id,
6828                error = %err,
6829                "failed to build module GOODBYE frame"
6830            );
6831            None
6832        }
6833    }
6834}
6835
6836/// Send every registered module connection its module GOODBYE at daemon
6837/// shutdown, then request that connection's close.
6838///
6839/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
6840/// before EOF, so the GOODBYE must reach the socket before the close. A close
6841/// request does not wait for the connection's queued frames: its writer gets a
6842/// bounded grace after the close, is aborted if it overruns it, and the daemon
6843/// process may exit before that grace ends. So with `wait_for_flush`, each
6844/// connection is closed only after its writer has acknowledged writing the
6845/// GOODBYE, or once a short shared budget runs out, so one module that is not
6846/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
6847/// are only queued, for a shutdown the operator has told to stop waiting.
6848/// A connection that is already gone is skipped.
6849#[cfg(unix)]
6850async fn send_module_goodbyes_for_daemon_shutdown(
6851    forwarding: &Arc<ForwardingTable>,
6852    reason: &CloseReason,
6853    wait_for_flush: bool,
6854) {
6855    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
6856    let targets = match forwarding.module_connections() {
6857        Ok(targets) => targets,
6858        Err(err) => {
6859            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
6860            return;
6861        }
6862    };
6863    let deadline = Instant::now() + GOODBYE_BUDGET;
6864    let mut sends = tokio::task::JoinSet::new();
6865    for target in targets {
6866        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
6867            continue;
6868        };
6869        if !wait_for_flush {
6870            if let Err(err) = target.sink.try_send(frame) {
6871                debug!(
6872                    module_id = %target.module_id,
6873                    error = %err,
6874                    "shutdown module GOODBYE was not queued"
6875                );
6876            }
6877            continue;
6878        }
6879        let forwarding = Arc::clone(forwarding);
6880        let reason = reason.clone();
6881        sends.spawn(async move {
6882            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
6883                Ok(Ok(())) => {}
6884                Ok(Err(err)) => debug!(
6885                    module_id = %target.module_id,
6886                    error = %err,
6887                    "module connection closed before its shutdown GOODBYE was written"
6888                ),
6889                Err(_) => warn!(
6890                    module_id = %target.module_id,
6891                    budget = ?GOODBYE_BUDGET,
6892                    "shutdown module GOODBYE was not written within its budget; closing anyway"
6893                ),
6894            }
6895            forwarding.request_connection_close(target.endpoint.connection_id, reason);
6896        });
6897    }
6898    // Every task ends by the shared deadline, so this wait is bounded too.
6899    while sends.join_next().await.is_some() {}
6900}
6901
6902fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
6903    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
6904        return;
6905    };
6906    if let Err(err) = target.sink.try_send(frame) {
6907        warn!(
6908            module_id,
6909            target_connection_id = target.endpoint.connection_id.get(),
6910            error = %err,
6911            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
6912        );
6913        forwarding.request_connection_close(
6914            target.endpoint.connection_id,
6915            CloseReason::new(
6916                "module_goodbye_delivery_failed",
6917                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
6918            ),
6919        );
6920    }
6921}
6922
6923#[derive(Clone, Copy)]
6924struct ForwardingDrainContext<'a> {
6925    spec: &'a ModuleSpec,
6926    runtime: &'a SupervisorRuntimeConfig,
6927    registry: &'a Registry,
6928    scope: DrainScope,
6929}
6930
6931/// Which process a forwarding drain addresses.
6932#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6933enum DrainScope {
6934    /// Whatever endpoint is active for the module id: every plain stop,
6935    /// restart and reload. Also moves the module's state to `Draining`.
6936    Active,
6937    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
6938    /// module id would resolve to the promoted candidate and leave neither
6939    /// process routable. The module's state is left alone, since the promoted
6940    /// candidate is what it describes and that process is running.
6941    Endpoint(crate::ModuleEndpointId),
6942}
6943
6944/// Whether a child being drained has already been asked to stop by the time
6945/// its drain wait starts.
6946///
6947/// The drain wait is the same budget whatever this says. What it decides is
6948/// whether the supervisor must ask by signal before that wait begins: a child
6949/// that nobody asked will sit out the whole budget and then be SIGKILLed,
6950/// healthy or not.
6951#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6952enum StopNotice {
6953    /// The module was sent `module.draining` and a module GOODBYE over its own
6954    /// registered connection, and stops itself.
6955    SentOverConnection,
6956    /// The forwarding drain found no registered connection for the module: a
6957    /// subc child spawned moments ago that has not sent HELLO yet, or a
6958    /// `protocol: "none"` child, which never registers.
6959    NoConnection,
6960    /// This path sends nothing over the module's connection: the supervisor has
6961    /// no forwarding table, or the caller stops the child without a forwarding
6962    /// drain.
6963    NotSent,
6964}
6965
6966async fn begin_forwarding_drain(
6967    spec: &ModuleSpec,
6968    runtime: &SupervisorRuntimeConfig,
6969    registry: &Registry,
6970    snapshot: &SharedSnapshot,
6971    enabled: Option<bool>,
6972    reason: RouteCloseReason,
6973) -> Result<StopNotice, SuperviseError> {
6974    let Some(forwarding) = runtime.forwarding.as_ref() else {
6975        return Err(SuperviseError::ReloadUnavailable {
6976            module_id: spec.module_id.clone(),
6977            reason: "supervisor was not configured with a forwarding table".to_string(),
6978        });
6979    };
6980
6981    begin_forwarding_drain_with(
6982        forwarding,
6983        ForwardingDrainContext {
6984            spec,
6985            runtime,
6986            registry,
6987            scope: DrainScope::Active,
6988        },
6989        snapshot,
6990        enabled,
6991        reason,
6992        runtime.drain_timeout,
6993    )
6994    .await
6995}
6996
6997async fn begin_forwarding_drain_if_configured(
6998    spec: &ModuleSpec,
6999    runtime: &SupervisorRuntimeConfig,
7000    registry: &Registry,
7001    snapshot: &SharedSnapshot,
7002    enabled: Option<bool>,
7003    reason: RouteCloseReason,
7004) -> Result<StopNotice, SuperviseError> {
7005    begin_forwarding_drain_with_timeout(
7006        spec,
7007        runtime,
7008        registry,
7009        snapshot,
7010        enabled,
7011        reason,
7012        runtime.drain_timeout,
7013    )
7014    .await
7015}
7016
7017/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
7018/// budget, for paths where the operator overrides the module's configured one
7019/// (`supervisor.restart{drain_timeout_ms}`).
7020async fn begin_forwarding_drain_with_timeout(
7021    spec: &ModuleSpec,
7022    runtime: &SupervisorRuntimeConfig,
7023    registry: &Registry,
7024    snapshot: &SharedSnapshot,
7025    enabled: Option<bool>,
7026    reason: RouteCloseReason,
7027    drain_timeout: Duration,
7028) -> Result<StopNotice, SuperviseError> {
7029    let Some(forwarding) = runtime.forwarding.as_ref() else {
7030        return Ok(StopNotice::NotSent);
7031    };
7032
7033    begin_forwarding_drain_with(
7034        forwarding,
7035        ForwardingDrainContext {
7036            spec,
7037            runtime,
7038            registry,
7039            scope: DrainScope::Active,
7040        },
7041        snapshot,
7042        enabled,
7043        reason,
7044        drain_timeout,
7045    )
7046    .await
7047}
7048
7049async fn begin_forwarding_drain_with(
7050    forwarding: &ForwardingTable,
7051    context: ForwardingDrainContext<'_>,
7052    snapshot: &SharedSnapshot,
7053    enabled: Option<bool>,
7054    reason: RouteCloseReason,
7055    drain_timeout: Duration,
7056) -> Result<StopNotice, SuperviseError> {
7057    let ForwardingDrainContext {
7058        spec,
7059        runtime,
7060        registry,
7061        scope,
7062    } = context;
7063    debug_assert_ne!(reason, RouteCloseReason::Crash);
7064    let terminal = matches!(reason, RouteCloseReason::Disable);
7065    let drain_started_at = Instant::now();
7066    let drain_deadline = drain_started_at + drain_timeout;
7067    let deadline_ms =
7068        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
7069    let busy_gauges = match scope {
7070        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
7071        DrainScope::Endpoint(endpoint) => {
7072            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
7073        }
7074    };
7075
7076    // Admission gate first: route.open/commit and route REQUEST admission are closed
7077    // before the first quiescence check, so the outstanding count can only fall.
7078    let gate_started = Instant::now();
7079    let drain_target = match scope {
7080        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
7081        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
7082    }
7083    .map_err(SuperviseError::Forwarding)?;
7084    // The instant admission closed, and how long taking the forwarding write
7085    // lock to close it took. The timeout line reports only the quiescence
7086    // wait, so without this a drain that started late looked like one that
7087    // started on time.
7088    info!(
7089        module_id = %spec.module_id,
7090        ?reason,
7091        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
7092        connected = drain_target.is_some(),
7093        "module drain began; route admission closed"
7094    );
7095    if scope == DrainScope::Active {
7096        update_snapshot(snapshot, Some(&spec.module_id), |state| {
7097            state.state = ModuleState::Draining;
7098            state.draining_to_replace =
7099                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
7100            if let Some(enabled) = enabled {
7101                state.enabled = enabled;
7102            }
7103        })?;
7104    }
7105
7106    let Some(target) = drain_target.as_ref() else {
7107        // Nothing was sent: the module has no registered connection to carry
7108        // `module.draining` or a GOODBYE. The caller must not assume the child
7109        // was asked to stop.
7110        return Ok(StopNotice::NoConnection);
7111    };
7112    {
7113        send_module_draining(&spec.module_id, reason, deadline_ms, target);
7114        let routes = forwarding
7115            .endpoint_routes(target.endpoint)
7116            .map_err(SuperviseError::Forwarding)?;
7117        let routes_notified = routes.len();
7118        crate::control::send_route_control_pushes(
7119            forwarding,
7120            routes.clone(),
7121            ClientControlPush::RouteClosing {
7122                module_id: spec.module_id.clone(),
7123                channels: Vec::new(),
7124                reason,
7125            },
7126        );
7127        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
7128
7129        // `route.closing` was just sent above: from here on every return path,
7130        // including an early one, MUST send `route.closed` before propagating
7131        // anything else. A client holds `closing` as a promise that a verdict is
7132        // coming; leaving early without `closed` strands it waiting forever, since
7133        // `closing` carries no timeout of its own.
7134        let wait_result = wait_for_forwarding_quiescence(
7135            forwarding,
7136            &spec.module_id,
7137            runtime,
7138            target.endpoint,
7139            drain_deadline,
7140            &busy_gauges,
7141            scope,
7142        )
7143        .await;
7144        let drained = drained_after_quiescence_wait(&wait_result);
7145        if let Err(err) = &wait_result {
7146            error!(
7147                module_id = %spec.module_id,
7148                ?reason,
7149                error = %err,
7150                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
7151            );
7152        } else if !drained {
7153            // Name what the drain waited on. Without it the line says only that
7154            // something did not settle, and "one wedged call" and "every
7155            // session's held stream" read the same; the first is a module bug,
7156            // the second is a module that should end its streams on
7157            // module.draining. Read before teardown releases the routes.
7158            let holdouts = forwarding
7159                .endpoint_drain_holdouts(target.endpoint)
7160                .unwrap_or_default();
7161            warn!(
7162                module_id = %spec.module_id,
7163                waited = ?drain_timeout,
7164                ?reason,
7165                held_requests = holdouts.requests,
7166                held_routes = holdouts.routes,
7167                total_routes = holdouts.total_routes,
7168                top_connections = ?holdouts.top_connections,
7169                // `module_channel:corr`, so the module can find each held request
7170                // in its own log; capped, so `held_requests` is the full count.
7171                held = %holdouts
7172                    .held
7173                    .iter()
7174                    .map(|(channel, corr)| format!("{channel}:{corr}"))
7175                    .collect::<Vec<_>>()
7176                    .join(","),
7177                "route drain timed out before request quiescence; forcing teardown"
7178            );
7179        }
7180        crate::control::send_route_control_pushes(
7181            forwarding,
7182            routes,
7183            ClientControlPush::RouteClosed {
7184                module_id: spec.module_id.clone(),
7185                channels: Vec::new(),
7186                reason,
7187                drained,
7188                abandoned: target.abandoned_bindings.len() as u32,
7189                excluded_subscriptions: target.excluded_subscriptions,
7190                terminal: Some(terminal),
7191            },
7192        );
7193        wait_result?;
7194
7195        // `route.closed` has now been sent unconditionally above. From here the
7196        // remaining steps are cleanup (route + module GOODBYE) rather than a
7197        // promise the client is waiting on, but a lock-poisoned
7198        // `release_module_endpoint_routes` would otherwise skip the module
7199        // GOODBYE silently too -- send it before propagating the error.
7200        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
7201            Ok(routes) => routes,
7202            Err(err) => {
7203                warn!(
7204                    module_id = %spec.module_id,
7205                    ?reason,
7206                    error = %err,
7207                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
7208                );
7209                send_module_goodbye(&spec.module_id, forwarding, target);
7210                return Err(SuperviseError::Forwarding(err));
7211            }
7212        };
7213        let route_goodbye_count = released_routes.len();
7214        send_route_goodbyes(forwarding, released_routes);
7215        send_module_goodbye(&spec.module_id, forwarding, target);
7216
7217        // The drain's happy path was previously silent: every emission above is
7218        // best-effort with only its failure arm logged, so "were consumers told"
7219        // was unprovable from the daemon log (surfaced by a 30-minute consumer
7220        // hang where the open question was exactly whether teardown notice went
7221        // out). One summary line makes that class decidable in one grep.
7222        info!(
7223            module_id = %spec.module_id,
7224            ?reason,
7225            routes_notified,
7226            route_goodbyes = route_goodbye_count,
7227            abandoned_reservations = target.abandoned_bindings.len(),
7228            excluded_subscriptions = target.excluded_subscriptions,
7229            drained,
7230            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
7231        );
7232    }
7233
7234    Ok(StopNotice::SentOverConnection)
7235}
7236
7237/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
7238/// the only slot a plain (non-swap) spawn can register into.
7239async fn wait_for_registration_after_reload(
7240    registry: &Registry,
7241    module_id: &str,
7242    snapshot: &SharedSnapshot,
7243    child: &mut SupervisedChild,
7244    wait: Duration,
7245) -> Result<RegistrationWaitOutcome, SuperviseError> {
7246    wait_for_slot_registration(
7247        registry,
7248        crate::registry::RegistrationSlot::Active(module_id),
7249        module_id,
7250        snapshot,
7251        child,
7252        wait,
7253    )
7254    .await
7255}
7256
7257/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
7258///
7259/// Keyed on the slot rather than the bare module id because during a swap the
7260/// id's active slot is already held by the incumbent: an id-keyed wait would
7261/// report the incumbent's registration as the candidate's and a candidate that
7262/// never registers would look registered. A swap candidate waits on
7263/// `crate::registry::RegistrationSlot::Candidate`.
7264async fn wait_for_slot_registration(
7265    registry: &Registry,
7266    slot: crate::registry::RegistrationSlot<'_>,
7267    module_id: &str,
7268    snapshot: &SharedSnapshot,
7269    child: &mut SupervisedChild,
7270    wait: Duration,
7271) -> Result<RegistrationWaitOutcome, SuperviseError> {
7272    let deadline = Instant::now() + wait;
7273    loop {
7274        if registry
7275            .registration(slot)
7276            .map_err(SuperviseError::Registry)?
7277            .is_some()
7278        {
7279            return Ok(RegistrationWaitOutcome::Registered);
7280        }
7281
7282        let now = Instant::now();
7283        if now >= deadline {
7284            return Ok(RegistrationWaitOutcome::TimedOut);
7285        }
7286        let remaining = deadline.saturating_duration_since(now);
7287        let poll = remaining.min(REGISTRY_RELEASE_POLL);
7288
7289        tokio::select! {
7290            wait_result = child.wait() => {
7291                let status = wait_result.map_err(|source| SuperviseError::Wait {
7292                    module_id: module_id.to_string(),
7293                    source,
7294                })?;
7295                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
7296                    snapshot,
7297                    child,
7298                    &status,
7299                )));
7300            }
7301            _ = sleep(poll) => {}
7302        }
7303    }
7304}
7305
7306fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
7307    // A replacement process that exits before HELLO did not provide service, even
7308    // if it used status 0. Count it against the restart cap as a new-binary failure.
7309    if exit_report.kind != ExitKind::DeliberateSeverance {
7310        exit_report.kind = ExitKind::Crash;
7311    }
7312    exit_report
7313}
7314
7315async fn handle_reload_child_registration_failure(
7316    spec: &ModuleSpec,
7317    runtime: &SupervisorRuntimeConfig,
7318    registry: &Registry,
7319    process_liveness: &SupervisorProcessLiveness,
7320    snapshot: &SharedSnapshot,
7321    _child: &mut Option<SupervisedChild>,
7322    failure: ReloadRegistrationFailure,
7323) -> Result<(), SuperviseError> {
7324    let ReloadRegistrationFailure {
7325        exit_report,
7326        reason,
7327    } = failure;
7328    match on_child_exit(
7329        spec,
7330        runtime.restart_policy,
7331        registry,
7332        snapshot,
7333        &runtime.terminal_ring,
7334        &runtime.spawn_events,
7335        &runtime.child_roster,
7336        exit_report,
7337    )
7338    .await
7339    {
7340        NextAction::Stop {
7341            registration_released,
7342        } => {
7343            if registration_released {
7344                process_liveness.untrack_if_current(&spec.module_id, snapshot);
7345            }
7346        }
7347        NextAction::Restart { schedule } => {
7348            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
7349                schedule.delay
7350            });
7351            if let Some(schedule) = schedule {
7352                log_crash_respawn(&spec.module_id, schedule);
7353            }
7354            schedule_respawn(
7355                runtime,
7356                snapshot,
7357                &spec.module_id,
7358                delay,
7359                RespawnKind::Spawn,
7360            )?;
7361        }
7362    }
7363    Err(SuperviseError::ReloadFailed {
7364        module_id: spec.module_id.clone(),
7365        reason,
7366    })
7367}
7368
7369async fn handle_reload_spawn_failure(
7370    spec: &ModuleSpec,
7371    runtime: &SupervisorRuntimeConfig,
7372    process_liveness: &SupervisorProcessLiveness,
7373    snapshot: &SharedSnapshot,
7374    _child: &mut Option<SupervisedChild>,
7375    reason: String,
7376) -> Result<(), SuperviseError> {
7377    let now = Instant::now();
7378    let mut schedule = None;
7379    update_snapshot(snapshot, Some(&spec.module_id), |state| {
7380        clear_current_process_facts(state);
7381        if state.enabled {
7382            schedule = state.next_crash_restart(&runtime.restart_policy, now);
7383            state.state = if schedule.is_some() {
7384                ModuleState::Restarting
7385            } else {
7386                ModuleState::Failed
7387            };
7388        } else {
7389            state.state = ModuleState::Disabled;
7390        }
7391    })?;
7392    if let Some(schedule) = schedule {
7393        schedule_respawn(
7394            runtime,
7395            snapshot,
7396            &spec.module_id,
7397            schedule.delay,
7398            RespawnKind::Spawn,
7399        )?;
7400    } else {
7401        process_liveness.untrack_if_current(&spec.module_id, snapshot);
7402    }
7403    Err(SuperviseError::ReloadFailed {
7404        module_id: spec.module_id.clone(),
7405        reason,
7406    })
7407}
7408
7409fn control_flags() -> Flags {
7410    Flags::new(false, Priority::Passive, false)
7411}
7412
7413#[allow(clippy::too_many_arguments)]
7414async fn drain_optional_child(
7415    module_id: &str,
7416    protocol: ModuleProtocol,
7417    stop_notice: StopNotice,
7418    registry: &Registry,
7419    forwarding: Option<&ForwardingTable>,
7420    snapshot: &SharedSnapshot,
7421    terminal_ring: &Arc<Mutex<TerminalRing>>,
7422    spawn_events: &SpawnEventFeed,
7423    child: &mut Option<SupervisedChild>,
7424    drain_timeout: Duration,
7425    final_state: ModuleState,
7426    enabled: Option<bool>,
7427) -> Result<(), SuperviseError> {
7428    if let Some(child) = child.take() {
7429        drain_child_to_state(
7430            module_id,
7431            protocol,
7432            stop_notice,
7433            registry,
7434            forwarding,
7435            snapshot,
7436            terminal_ring,
7437            spawn_events,
7438            child,
7439            drain_timeout,
7440            final_state,
7441            enabled,
7442        )
7443        .await
7444    } else {
7445        update_snapshot(snapshot, Some(module_id), |state| {
7446            state.state = final_state;
7447            if let Some(enabled) = enabled {
7448                state.enabled = enabled;
7449            }
7450            clear_current_process_facts(state);
7451        })?;
7452        release_dead_registration(registry, forwarding, snapshot, module_id).await
7453    }
7454}
7455
7456#[allow(clippy::too_many_arguments)]
7457async fn drain_child_to_state(
7458    module_id: &str,
7459    protocol: ModuleProtocol,
7460    stop_notice: StopNotice,
7461    registry: &Registry,
7462    forwarding: Option<&ForwardingTable>,
7463    snapshot: &SharedSnapshot,
7464    terminal_ring: &Arc<Mutex<TerminalRing>>,
7465    spawn_events: &SpawnEventFeed,
7466    mut child: SupervisedChild,
7467    drain_timeout: Duration,
7468    final_state: ModuleState,
7469    enabled: Option<bool>,
7470) -> Result<(), SuperviseError> {
7471    update_snapshot(snapshot, Some(module_id), |state| {
7472        state.state = ModuleState::Draining;
7473        state.draining_to_replace = final_state == ModuleState::Restarting;
7474        if let Some(enabled) = enabled {
7475            state.enabled = enabled;
7476        }
7477    })?;
7478
7479    // The wait below is the same budget in every case; what differs is
7480    // whether anything has ASKED the child to stop before it starts. Only a
7481    // forwarding drain that reached the module's registered connection has
7482    // (`module.draining`, then a module GOODBYE). Every other child was told
7483    // nothing: a `protocol: "none"` module, which never registers; a subc
7484    // module spawned moments ago that has not sent HELLO yet; or a stop that
7485    // runs no forwarding drain. Without a signal the budget is only a delay
7486    // in front of SIGKILL -- and the not-yet-registered child is the worst
7487    // case, because it registers into a module that is already draining,
7488    // is never told, and is killed while healthy.
7489    if stop_notice != StopNotice::SentOverConnection {
7490        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
7491            info!(
7492                module_id,
7493                pid = child.pid,
7494                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7495                "module has no connection yet; requesting stop by signal"
7496            );
7497        }
7498        request_graceful_stop(module_id, &child);
7499    }
7500
7501    let exit_report = match timeout(drain_timeout, child.wait()).await {
7502        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
7503        Ok(Err(source)) => {
7504            fail_snapshot(snapshot, Some(module_id), None);
7505            return Err(SuperviseError::Wait {
7506                module_id: module_id.to_string(),
7507                source,
7508            });
7509        }
7510        Err(_) => {
7511            // Mirror the sibling arm above: state is already `Draining`, and an
7512            // error propagated from here would strand it there -- a state
7513            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
7514            // `Failed | Stopped`), leaving an operator Restart as the only exit.
7515            // `Failed` before `?` keeps the module operator-visible and
7516            // revivable. Trigger is an ESRCH race (process exits between the
7517            // drain timeout firing and the kill) or a post-kill wait failure
7518            // (issue #34).
7519            //
7520            // Logged because the kill is otherwise visible only as signal 9 in
7521            // the terminal ring, and the budget it follows can be long enough
7522            // that consumers see a stretch of refusals with no stated cause.
7523            warn!(
7524                module_id,
7525                pid = child.pid,
7526                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7527                reason = ?final_state,
7528                ?stop_notice,
7529                "drain budget expired before the module exited; killing it"
7530            );
7531            child.start_kill().map_err(|source| {
7532                fail_snapshot(snapshot, Some(module_id), None);
7533                SuperviseError::Kill {
7534                    module_id: module_id.to_string(),
7535                    source,
7536                }
7537            })?;
7538            let status = child.wait().await.map_err(|source| {
7539                fail_snapshot(snapshot, Some(module_id), None);
7540                SuperviseError::Wait {
7541                    module_id: module_id.to_string(),
7542                    source,
7543                }
7544            })?;
7545            classify_reaped_child_exit(snapshot, &child, &status)
7546        }
7547    };
7548
7549    update_snapshot(snapshot, Some(module_id), |state| {
7550        state.state = final_state;
7551        if let Some(enabled) = enabled {
7552            state.enabled = enabled;
7553        }
7554        clear_current_process_facts(state);
7555        state.last_exit = Some(exit_report.clone());
7556        if exit_report.kind == ExitKind::DeliberateSeverance {
7557            state.lifetime_restarts += 1;
7558        }
7559    })?;
7560    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
7561    record_terminal_with_detail(
7562        module_id,
7563        terminal_ring,
7564        spawn_events,
7565        &exit_report,
7566        terminal_disposition(final_state),
7567        detail,
7568    );
7569    child.drain_stderr(module_id).await;
7570
7571    release_dead_registration(registry, forwarding, snapshot, module_id).await
7572}
7573
7574/// Ask a child that nothing else has asked to stop, by signal.
7575///
7576/// A registered subc module is asked over its own connection: the drain sends
7577/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
7578/// module GOODBYE, and the module stops itself. A module that speaks no subc
7579/// wire receives none of that, and neither does a subc module that has not
7580/// registered yet, so for them the drain budget would be pure delay in front of
7581/// a SIGKILL -- and for a process with a store to flush (JetStream is the
7582/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
7583/// into a recovery on the next start.
7584///
7585/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
7586/// rule rather than an optimisation: that module's graceful stop is already
7587/// running by the time its child is drained, and a signal would race it.
7588///
7589/// Best-effort by construction. A child that has already exited is the ordinary
7590/// case rather than an error (the kill lands on a reaped or exiting pid), so a
7591/// failure is logged at debug and the wait-then-kill below still decides the
7592/// outcome.
7593#[cfg(unix)]
7594fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
7595    let Some(pid) = child
7596        .id()
7597        .and_then(|pid| i32::try_from(pid).ok())
7598        .and_then(rustix::process::Pid::from_raw)
7599    else {
7600        debug!(
7601            module_id,
7602            "no pid to signal for teardown; falling through to the drain wait"
7603        );
7604        return;
7605    };
7606    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
7607        Ok(()) => debug!(
7608            module_id,
7609            "sent SIGTERM to a module nothing else asked to stop"
7610        ),
7611        Err(err) => debug!(
7612            module_id,
7613            error = %err,
7614            "SIGTERM to module failed; the drain wait and kill still apply"
7615        ),
7616    }
7617}
7618
7619/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
7620/// Windows does offer need cooperation this supervisor cannot assume: a console
7621/// control event requires sharing a console with the child, and `WM_CLOSE`
7622/// requires the child to pump a message loop. A supervised server process does
7623/// neither, so there is nothing to send and teardown is the wait followed by the
7624/// kill. Emulating a signal here would mean inventing a stop protocol, which is
7625/// the thing `protocol: "none"` exists to avoid.
7626#[cfg(not(unix))]
7627fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
7628    debug!(
7629        module_id,
7630        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
7631    );
7632}
7633
7634fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
7635    match final_state {
7636        ModuleState::Stopped => TerminalDisposition::Stopped,
7637        ModuleState::Disabled => TerminalDisposition::Disabled,
7638        ModuleState::Restarting => TerminalDisposition::Restarting,
7639        ModuleState::Failed => TerminalDisposition::Failed,
7640        ModuleState::Starting
7641        | ModuleState::Running
7642        | ModuleState::Unresponsive
7643        | ModuleState::Draining => {
7644            unreachable!("terminal exits only finish in terminal or restarting states")
7645        }
7646    }
7647}
7648
7649/// Release a reaped child's registration before allowing another spawn.
7650///
7651/// EOF is not a process-lifetime signal: an inherited socket can stay open
7652/// indefinitely, and serial frame dispatch can be waiting on egress instead of
7653/// reading EOF. After the normal release grace, request connection close (which
7654/// cancels both reads and dispatch), then allow one more release grace for the
7655/// connection guard's forwarding cleanup. Never evict a different connection.
7656async fn release_dead_registration(
7657    registry: &Registry,
7658    forwarding: Option<&ForwardingTable>,
7659    snapshot: &SharedSnapshot,
7660    module_id: &str,
7661) -> Result<(), SuperviseError> {
7662    let result = async {
7663        let registration = registry
7664            .get_module(module_id)
7665            .map_err(SuperviseError::Registry)?;
7666        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
7667            Ok(()) => return Ok(()),
7668            Err(SuperviseError::RegistrationStillActive { .. }) => {}
7669            Err(err) => return Err(err),
7670        }
7671        let pid = lock_snapshot(snapshot)?.reaped_pid;
7672        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
7673            warn!(
7674                module_id,
7675                pid,
7676                connection_id = registration.connection_id.get(),
7677                "reaped module registration outlived release grace; closing dead connection"
7678            );
7679            forwarding.request_connection_close(
7680                registration.connection_id,
7681                CloseReason::new(
7682                    "supervised_process_reaped",
7683                    format!("module '{module_id}' pid {pid} exited"),
7684                ),
7685            );
7686            wait_for_slot_registration_release(
7687                registry,
7688                crate::registry::RegistrationSlot::Connection(registration.connection_id),
7689                REGISTRY_RELEASE_TIMEOUT,
7690            )
7691            .await?;
7692        }
7693        wait_for_registration_release(registry, module_id, Duration::ZERO).await
7694    }
7695    .await;
7696    if let Err(err) = &result {
7697        fail_snapshot(snapshot, Some(module_id), None);
7698        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
7699    }
7700    result
7701}
7702
7703/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
7704/// plain stop or restart waits for before it spawns a replacement.
7705async fn wait_for_registration_release(
7706    registry: &Registry,
7707    module_id: &str,
7708    wait: Duration,
7709) -> Result<(), SuperviseError> {
7710    wait_for_slot_registration_release(
7711        registry,
7712        crate::registry::RegistrationSlot::Active(module_id),
7713        wait,
7714    )
7715    .await
7716}
7717
7718/// Wait for the registration in `slot` to go away.
7719///
7720/// Keyed on the slot rather than the bare module id because a successful swap
7721/// never empties the id's active slot (the promoted candidate is in it), so an
7722/// id-keyed wait for the incumbent's release would always time out. Draining a
7723/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
7724/// incumbent's connection instead.
7725async fn wait_for_slot_registration_release(
7726    registry: &Registry,
7727    slot: crate::registry::RegistrationSlot<'_>,
7728    wait: Duration,
7729) -> Result<(), SuperviseError> {
7730    let deadline = Instant::now() + wait;
7731    let mut release_events = registration_release_events().subscribe();
7732    let still_active = |registration: &crate::registry::ModuleRegistration| {
7733        SuperviseError::RegistrationStillActive {
7734            module_id: registration.manifest.module_id.clone(),
7735            waited: wait,
7736        }
7737    };
7738    loop {
7739        let _observed_generation = *release_events.borrow_and_update();
7740        let Some(registration) = registry
7741            .registration(slot)
7742            .map_err(SuperviseError::Registry)?
7743        else {
7744            return Ok(());
7745        };
7746
7747        let now = Instant::now();
7748        if now >= deadline {
7749            return Err(still_active(&registration));
7750        }
7751
7752        let remaining = deadline.saturating_duration_since(now);
7753        match timeout(remaining, release_events.changed()).await {
7754            Ok(Ok(())) | Ok(Err(_)) => {}
7755            Err(_) => return Err(still_active(&registration)),
7756        }
7757    }
7758}
7759
7760#[cfg(test)]
7761mod slot_registration_wait_tests {
7762    use super::*;
7763    use crate::registry::{ConnectionId, RegistrationSlot};
7764    use subc_protocol::manifest::ModuleManifest;
7765
7766    #[tokio::test]
7767    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
7768        let registry = Arc::new(Registry::default());
7769        let supervisor = Supervisor::new(Arc::clone(&registry), RestartPolicy::default());
7770        let runtime = supervisor.runtime_config();
7771        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
7772        let spec = ModuleSpec {
7773            module_id: "enable-stale-registration".to_string(),
7774            program: PathBuf::from("/missing/enable-retry-test"),
7775            args: Vec::new(),
7776            env: Vec::new(),
7777            reserved: false,
7778            reserved_prefixes: Vec::new(),
7779            protocol: ModuleProtocol::Subc,
7780            overlap: Default::default(),
7781        };
7782        let connection = ConnectionId::new(90);
7783        registry
7784            .register_with_control_ops(
7785                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
7786                1,
7787                connection,
7788                Vec::new(),
7789            )
7790            .unwrap();
7791        let mut child = None;
7792        let err = set_child_enabled(
7793            &spec,
7794            &runtime,
7795            &registry,
7796            &supervisor.process_liveness,
7797            &snapshot,
7798            &mut child,
7799            true,
7800        )
7801        .await
7802        .unwrap_err();
7803        assert!(matches!(
7804            err,
7805            SuperviseError::RegistrationStillActive { .. }
7806        ));
7807        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
7808        assert!(child.is_none());
7809        registry.deregister_connection(connection).unwrap();
7810        let err = set_child_enabled(
7811            &spec,
7812            &runtime,
7813            &registry,
7814            &supervisor.process_liveness,
7815            &snapshot,
7816            &mut child,
7817            true,
7818        )
7819        .await
7820        .unwrap_err();
7821        assert!(
7822            matches!(err, SuperviseError::Spawn { .. }),
7823            "second enable must attempt a spawn: {err}"
7824        );
7825        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
7826    }
7827
7828    const INCUMBENT: u64 = 1;
7829    const CANDIDATE: u64 = 2;
7830
7831    fn swapped_registry() -> Arc<Registry> {
7832        let registry = Arc::new(Registry::default());
7833        let manifest = ModuleManifest::builder("m", "0.1.0").build();
7834        registry
7835            .register_with_control_ops(
7836                manifest.clone(),
7837                1,
7838                ConnectionId::new(INCUMBENT),
7839                Vec::new(),
7840            )
7841            .unwrap();
7842        registry
7843            .register_candidate_with_control_ops(
7844                manifest,
7845                1,
7846                ConnectionId::new(CANDIDATE),
7847                Vec::new(),
7848            )
7849            .unwrap();
7850        registry
7851    }
7852
7853    /// After a promotion the id's active slot is held by the new process, so an
7854    /// id-keyed wait for the incumbent's release can never succeed; the
7855    /// connection-keyed wait completes as soon as the incumbent deregisters.
7856    #[tokio::test]
7857    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
7858        let registry = swapped_registry();
7859        registry.promote_candidate("m").unwrap().unwrap();
7860
7861        assert!(matches!(
7862            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
7863            Err(SuperviseError::RegistrationStillActive { .. })
7864        ));
7865
7866        // Still held while the incumbent's connection has not deregistered.
7867        assert!(matches!(
7868            wait_for_slot_registration_release(
7869                &registry,
7870                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7871                Duration::from_millis(50),
7872            )
7873            .await,
7874            Err(SuperviseError::RegistrationStillActive { .. })
7875        ));
7876
7877        let releaser = Arc::clone(&registry);
7878        let release = tokio::spawn(async move {
7879            sleep(Duration::from_millis(20)).await;
7880            releaser
7881                .deregister_connection(ConnectionId::new(INCUMBENT))
7882                .unwrap();
7883            notify_registration_release();
7884        });
7885        wait_for_slot_registration_release(
7886            &registry,
7887            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7888            Duration::from_secs(5),
7889        )
7890        .await
7891        .expect("the incumbent's own registration is released");
7892        release.await.unwrap();
7893        assert!(registry.get_module("m").unwrap().is_some());
7894    }
7895
7896    /// The candidate slot is waited on separately from the active slot: the
7897    /// incumbent's registration neither holds up nor stands in for it.
7898    #[tokio::test]
7899    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
7900        let registry = swapped_registry();
7901        assert!(matches!(
7902            wait_for_slot_registration_release(
7903                &registry,
7904                RegistrationSlot::Candidate("m"),
7905                Duration::from_millis(50),
7906            )
7907            .await,
7908            Err(SuperviseError::RegistrationStillActive { .. })
7909        ));
7910        registry
7911            .deregister_connection(ConnectionId::new(CANDIDATE))
7912            .unwrap();
7913        wait_for_slot_registration_release(
7914            &registry,
7915            RegistrationSlot::Candidate("m"),
7916            Duration::from_millis(50),
7917        )
7918        .await
7919        .expect("a candidate slot with no candidate is released");
7920        assert!(registry
7921            .registration(RegistrationSlot::Active("m"))
7922            .unwrap()
7923            .is_some());
7924    }
7925}
7926
7927fn classify_exit(status: &ExitStatus) -> ExitReport {
7928    ExitReport {
7929        kind: if status.success() {
7930            ExitKind::Clean
7931        } else {
7932            ExitKind::Crash
7933        },
7934        code: status.code(),
7935        signal: exit_signal(status),
7936        at_ms: unix_ms_now(),
7937    }
7938}
7939
7940/// The terminal record for a module whose `wait()` call itself errored (e.g. the
7941/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
7942/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
7943/// disposition still must be `Failed` so the terminal ring is not silently missing
7944/// an entry, matching what `fail_snapshot` records for this same arm.
7945fn wait_error_exit_report() -> ExitReport {
7946    ExitReport {
7947        kind: ExitKind::Crash,
7948        code: None,
7949        signal: None,
7950        at_ms: unix_ms_now(),
7951    }
7952}
7953
7954#[cfg(unix)]
7955fn exit_signal(status: &ExitStatus) -> Option<i32> {
7956    use std::os::unix::process::ExitStatusExt;
7957
7958    status.signal()
7959}
7960
7961#[cfg(not(unix))]
7962fn exit_signal(_status: &ExitStatus) -> Option<i32> {
7963    None
7964}
7965
7966/// Give an operator-touched module its full crash budget back.
7967///
7968/// Named for the counter it used to zero; it now empties the in-window ring,
7969/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
7970/// ledger of what happened survives every operator action.
7971fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
7972    update_snapshot(snapshot, Some(module_id), |state| {
7973        state.clear_crash_restarts();
7974    })
7975}
7976
7977fn set_running(
7978    snapshot: &SharedSnapshot,
7979    child: &SupervisedChild,
7980    module_id: &str,
7981    spawn_events: &SpawnEventFeed,
7982) -> Result<(), SuperviseError> {
7983    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7984        module_id: Some(module_id.to_string()),
7985    })?;
7986    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
7987    if std::mem::take(&mut state.coalesced_restart_pending) {
7988        let generation = state.spawn_generation;
7989        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
7990    }
7991    state.drain_disposition_detail = None;
7992    // Every caller of this is a plain spawn, which always uses the primary key;
7993    // a promoted swap candidate sets the flag itself after this returns.
7994    state.in_alternate_slot = false;
7995    state.configuration_updated_since_spawn = false;
7996    state.state = ModuleState::Running;
7997    state.enabled = true;
7998    state.process_alive = true;
7999    state.pid = child.id();
8000    state.spawned_at_ms = Some(child.spawned_at_ms);
8001    state.spawned_from = Some(child.spawned_from.clone());
8002    state.spawned_file_identity = child.spawned_file_identity;
8003    state.process_start_time = child.process_start_time;
8004    Ok(())
8005}
8006
8007fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
8008    state.process_alive = false;
8009    state.pid = None;
8010    state.spawned_at_ms = None;
8011    state.spawned_from = None;
8012    state.spawned_file_identity = None;
8013    state.process_start_time = None;
8014    state.deliberate_severance = None;
8015}
8016
8017#[cfg(test)]
8018fn record_deliberate_severance(
8019    snapshot: &SharedSnapshot,
8020    identity: ProcessIdentity,
8021) -> Result<(), SuperviseError> {
8022    update_snapshot(snapshot, None, |state| {
8023        state.deliberate_severance = Some(identity);
8024    })
8025}
8026
8027fn apply_deliberate_severance_marker(
8028    snapshot: &SharedSnapshot,
8029    exited_identity: Option<ProcessIdentity>,
8030    mut exit_report: ExitReport,
8031) -> ExitReport {
8032    let marker = lock_snapshot(snapshot)
8033        .ok()
8034        .and_then(|mut state| state.deliberate_severance.take());
8035    if marker.is_some() && marker == exited_identity {
8036        exit_report.kind = ExitKind::DeliberateSeverance;
8037    }
8038    exit_report
8039}
8040
8041fn classify_reaped_child_exit(
8042    snapshot: &SharedSnapshot,
8043    child: &SupervisedChild,
8044    status: &ExitStatus,
8045) -> ExitReport {
8046    let _ = update_snapshot(snapshot, None, |state| state.reaped_pid = Some(child.pid));
8047    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
8048}
8049
8050fn fail_snapshot(
8051    snapshot: &SharedSnapshot,
8052    module_id: Option<&str>,
8053    last_exit: Option<ExitReport>,
8054) {
8055    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
8056        state.state = ModuleState::Failed;
8057        clear_current_process_facts(state);
8058        if let Some(last_exit) = last_exit {
8059            state.last_exit = Some(last_exit);
8060        }
8061    }) {
8062        error!(error = %err, "failed to mark supervisor state failed");
8063    }
8064}
8065
8066fn update_snapshot(
8067    snapshot: &SharedSnapshot,
8068    module_id: Option<&str>,
8069    update: impl FnOnce(&mut SupervisorSnapshot),
8070) -> Result<(), SuperviseError> {
8071    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
8072        module_id: module_id.map(ToOwned::to_owned),
8073    })?;
8074    update(&mut state);
8075    Ok(())
8076}
8077
8078const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
8079
8080fn lock_snapshot_for_control<'a>(
8081    snapshot: &'a SharedSnapshot,
8082    module_id: &str,
8083    caller: &'static str,
8084) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
8085    let started_at = Instant::now();
8086    let guard = lock_snapshot(snapshot)?;
8087    let waited = started_at.elapsed();
8088    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
8089        warn!(
8090            module_id = %module_id,
8091            waited_ms = waited.as_millis() as u64,
8092            caller = %caller,
8093            "slow snapshot lock"
8094        );
8095    }
8096    Ok(guard)
8097}
8098
8099fn lock_snapshot(
8100    snapshot: &SharedSnapshot,
8101) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
8102    snapshot
8103        .lock()
8104        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
8105}
8106
8107#[cfg(test)]
8108mod terminal_history_tests {
8109    use std::{
8110        path::PathBuf,
8111        sync::Arc,
8112        time::{Duration, Instant},
8113    };
8114
8115    use tokio::time::sleep;
8116
8117    use super::{
8118        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
8119        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
8120        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
8121        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
8122        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
8123        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
8124        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
8125    };
8126    // The supervisor's clock, distinct from the `std::time::Instant` these tests
8127    // use for their own wall-clock deadlines: crash-restart instants must be on
8128    // the same clock the production code stamps them with, which is tokio's (and
8129    // is what `start_paused` tests can move).
8130    use super::Instant as ClockInstant;
8131    use crate::{
8132        registry::Registry,
8133        terminal_ring::{TerminalRing, TerminalRingConfig},
8134    };
8135    use std::sync::Mutex;
8136    use subc_control::TerminalDisposition;
8137
8138    /// See the twin in `control.rs` for why this derives the path from
8139    /// `current_exe()` and why the existence check is here: `--lib` alone does
8140    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
8141    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
8142    pub(super) fn fake_aft_stub_path() -> PathBuf {
8143        let mut path = std::env::current_exe().expect("current_exe available in tests");
8144        path.pop();
8145        path.pop();
8146        path.push(if cfg!(windows) {
8147            "fake-aft-stub.exe"
8148        } else {
8149            "fake-aft-stub"
8150        });
8151        assert!(
8152            path.exists(),
8153            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
8154             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
8155            path.display()
8156        );
8157        path
8158    }
8159
8160    #[test]
8161    fn reserved_never_spawned_refuses_every_hello() {
8162        // The canary hole: a reserved id whose module has never spawned had NO
8163        // gate entry and admitted anyone -- the reservation protected the nonce
8164        // holder, not the NAME. Now the entry is present with no legitimate
8165        // holder and refuses all comers.
8166        let supervisor = SupervisorHandle::default();
8167        supervisor.apply_identity_configuration(&ModuleSpec {
8168            module_id: "never-spawned".to_string(),
8169            program: PathBuf::from("/usr/bin/false"),
8170            args: Vec::new(),
8171            env: Vec::new(),
8172            reserved: true,
8173            reserved_prefixes: Vec::new(),
8174            protocol: ModuleProtocol::Subc,
8175            overlap: Default::default(),
8176        });
8177        assert!(
8178            supervisor
8179                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
8180                .is_some(),
8181            "forged nonce must refuse on a reserved never-spawned id"
8182        );
8183        assert!(
8184            supervisor
8185                .reserved_hello_rejection("never-spawned", None)
8186                .is_some(),
8187            "absent nonce must refuse on a reserved never-spawned id"
8188        );
8189        // And a real spawn nonce minted later admits exactly that nonce.
8190        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
8191        supervisor.apply_identity_configuration(&ModuleSpec {
8192            module_id: "never-spawned".to_string(),
8193            program: PathBuf::from("/usr/bin/false"),
8194            args: Vec::new(),
8195            env: Vec::new(),
8196            reserved: true,
8197            reserved_prefixes: Vec::new(),
8198            protocol: ModuleProtocol::Subc,
8199            overlap: Default::default(),
8200        });
8201        assert!(supervisor
8202            .reserved_hello_rejection("never-spawned", Some("minted"))
8203            .is_none());
8204        assert!(supervisor
8205            .reserved_hello_rejection("never-spawned", Some("forged"))
8206            .is_some());
8207    }
8208
8209    /// Put `count` crash restarts on a snapshot's ring as if they had all just
8210    /// happened, which is what "spent budget" looks like to every reader.
8211    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
8212        let now = ClockInstant::now();
8213        for _ in 0..count {
8214            state.crash_restarts.push_back(now);
8215        }
8216    }
8217
8218    /// Age the oldest recorded restart out of `window`, standing in for the hours
8219    /// that would otherwise have to pass. Injecting the instant is the point: a
8220    /// test that slept a real window would take ten minutes and still prove less.
8221    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
8222        let aged = state
8223            .crash_restarts
8224            .front()
8225            .expect("a crash restart must be recorded before it can be aged")
8226            .checked_sub(window + Duration::from_secs(1))
8227            .expect("the test clock is far enough from its origin to age an instant");
8228        state.crash_restarts[0] = aged;
8229    }
8230
8231    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
8232        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
8233        seed_crash_restarts(&mut state, count);
8234        state
8235    }
8236
8237    #[test]
8238    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
8239        let policy = RestartPolicy::new(3, Duration::ZERO);
8240        let now = ClockInstant::now();
8241        assert!(daemon_will_restart(
8242            &mut snapshot_with_restarts(true, 2),
8243            &policy,
8244            now
8245        ));
8246        assert!(!daemon_will_restart(
8247            &mut snapshot_with_restarts(true, 3),
8248            &policy,
8249            now
8250        ));
8251        assert!(!daemon_will_restart(
8252            &mut snapshot_with_restarts(false, 0),
8253            &policy,
8254            now
8255        ));
8256    }
8257
8258    #[test]
8259    fn crash_restart_backoff_escalates_with_in_window_count() {
8260        let policy = RestartPolicy::new(4, Duration::from_millis(100))
8261            .with_max_backoff(Duration::from_secs(30));
8262        let now = ClockInstant::now();
8263        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
8264        let schedules = (0..4)
8265            .map(|_| {
8266                state
8267                    .next_crash_restart(&policy, now)
8268                    .expect("the test policy allows four crash restarts")
8269            })
8270            .collect::<Vec<_>>();
8271
8272        assert_eq!(
8273            schedules
8274                .iter()
8275                .map(|schedule| schedule.restart_in_window)
8276                .collect::<Vec<_>>(),
8277            vec![0, 1, 2, 3]
8278        );
8279        assert_eq!(
8280            schedules
8281                .iter()
8282                .map(|schedule| schedule.delay)
8283                .collect::<Vec<_>>(),
8284            vec![
8285                Duration::from_millis(100),
8286                Duration::from_secs(1),
8287                Duration::from_secs(10),
8288                Duration::from_secs(30),
8289            ]
8290        );
8291    }
8292
8293    #[test]
8294    fn crash_restart_backoff_resets_after_ring_clear() {
8295        let policy = RestartPolicy::new(3, Duration::from_millis(100));
8296        let now = ClockInstant::now();
8297        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
8298        assert_eq!(
8299            state.next_crash_restart(&policy, now).unwrap().delay,
8300            Duration::from_millis(100)
8301        );
8302        assert_eq!(
8303            state.next_crash_restart(&policy, now).unwrap().delay,
8304            Duration::from_secs(1)
8305        );
8306
8307        state.clear_crash_restarts();
8308        let schedule = state
8309            .next_crash_restart(&policy, now)
8310            .expect("a cleared ring must allow another restart");
8311        assert_eq!(schedule.restart_in_window, 0);
8312        assert_eq!(schedule.delay, Duration::from_millis(100));
8313    }
8314
8315    #[test]
8316    fn crash_restart_backoff_ignores_aged_restarts() {
8317        let policy = RestartPolicy::new(3, Duration::from_millis(100));
8318        let now = ClockInstant::now();
8319        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
8320        state
8321            .next_crash_restart(&policy, now)
8322            .expect("the first restart is allowed");
8323        state
8324            .next_crash_restart(&policy, now)
8325            .expect("the second restart is allowed");
8326        state.crash_restarts[0] = now
8327            .checked_sub(policy.window + Duration::from_secs(1))
8328            .expect("the fake clock can age a restart past the window");
8329
8330        let schedule = state
8331            .next_crash_restart(&policy, now)
8332            .expect("an aged restart must release its slot");
8333        assert_eq!(schedule.restart_in_window, 1);
8334        assert_eq!(schedule.delay, Duration::from_secs(1));
8335        assert_eq!(state.crash_restarts.len(), 2);
8336    }
8337
8338    /// The budget is a rate: the same three spent restarts refuse a respawn
8339    /// while they are recent and allow one once they have aged past the window.
8340    /// Nothing about the module changed in between, which is the whole point.
8341    #[test]
8342    fn a_budget_spent_before_the_window_no_longer_refuses() {
8343        let policy = RestartPolicy::new(3, Duration::ZERO);
8344        let mut state = snapshot_with_restarts(true, 3);
8345        let now = ClockInstant::now();
8346        assert!(!daemon_will_restart(&mut state, &policy, now));
8347
8348        assert!(daemon_will_restart(
8349            &mut state,
8350            &policy,
8351            now + policy.window + Duration::from_secs(1)
8352        ));
8353        assert!(
8354            state.crash_restarts.is_empty(),
8355            "reading the budget must drop the instants that left the window"
8356        );
8357    }
8358
8359    fn module_with_recovery_snapshot(
8360        state: ModuleState,
8361        enabled: bool,
8362        restart_count: u32,
8363    ) -> SupervisedModule {
8364        let registry = Arc::new(Registry::default());
8365        let supervisor =
8366            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
8367        let module = supervisor
8368            .spawn(ModuleSpec {
8369                module_id: "recovery-snapshot".to_string(),
8370                program: fake_aft_stub_path(),
8371                args: Vec::new(),
8372                env: Vec::new(),
8373                reserved: false,
8374                reserved_prefixes: Vec::new(),
8375                protocol: ModuleProtocol::Subc,
8376                overlap: Default::default(),
8377            })
8378            .unwrap();
8379        update_snapshot(
8380            &module.inner.snapshot,
8381            Some("recovery-snapshot"),
8382            |snapshot| {
8383                snapshot.state = state;
8384                snapshot.enabled = enabled;
8385                seed_crash_restarts(snapshot, restart_count);
8386            },
8387        )
8388        .unwrap();
8389        module
8390    }
8391
8392    #[cfg(target_os = "linux")]
8393    #[tokio::test]
8394    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
8395        let supervisor = Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
8396            .with_cgroup_placement(None);
8397        let result = supervisor.spawn(ModuleSpec {
8398            module_id: "no-cgroup-placement".to_string(),
8399            program: fake_aft_stub_path(),
8400            args: Vec::new(),
8401            env: Vec::new(),
8402            reserved: false,
8403            reserved_prefixes: Vec::new(),
8404            protocol: ModuleProtocol::Subc,
8405            overlap: Default::default(),
8406        });
8407
8408        assert!(
8409            result.is_ok(),
8410            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
8411        );
8412    }
8413
8414    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8415    async fn undecided_snapshot_uses_shared_restart_predicate() {
8416        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
8417            .will_recover_after_connection_loss()
8418            .unwrap());
8419        assert!(
8420            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
8421                .will_recover_after_connection_loss()
8422                .unwrap()
8423        );
8424    }
8425
8426    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8427    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
8428        assert!(
8429            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
8430                .will_recover_after_connection_loss()
8431                .unwrap()
8432        );
8433    }
8434
8435    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8436    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
8437        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
8438            .will_recover_after_connection_loss()
8439            .unwrap());
8440        assert!(
8441            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
8442                .will_recover_after_connection_loss()
8443                .unwrap()
8444        );
8445    }
8446
8447    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8448    async fn warming_snapshot_is_limited_to_startup_phases() {
8449        for state in [
8450            ModuleState::Starting,
8451            ModuleState::Running,
8452            ModuleState::Restarting,
8453        ] {
8454            assert!(
8455                module_with_recovery_snapshot(state, true, 0)
8456                    .is_warming()
8457                    .unwrap(),
8458                "{state:?} should be warming"
8459            );
8460        }
8461        for state in [
8462            ModuleState::Unresponsive,
8463            ModuleState::Draining,
8464            ModuleState::Stopped,
8465            ModuleState::Failed,
8466            ModuleState::Disabled,
8467        ] {
8468            assert!(
8469                !module_with_recovery_snapshot(state, true, 0)
8470                    .is_warming()
8471                    .unwrap(),
8472                "{state:?} should not be warming"
8473            );
8474        }
8475    }
8476
8477    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8478    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
8479        let registry = Arc::new(Registry::default());
8480        let supervisor =
8481            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
8482        let module = supervisor
8483            .spawn(ModuleSpec {
8484                module_id: "terminal-history".to_string(),
8485                program: fake_aft_stub_path(),
8486                args: Vec::new(),
8487                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8488                reserved: false,
8489                reserved_prefixes: Vec::new(),
8490                protocol: ModuleProtocol::Subc,
8491                overlap: Default::default(),
8492            })
8493            .unwrap();
8494
8495        let deadline = Instant::now() + Duration::from_secs(5);
8496        loop {
8497            let history = module.terminal_history();
8498            if history.entries.len() == 2 {
8499                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
8500                assert_eq!(history.dropped, 0);
8501                assert_eq!(
8502                    history
8503                        .entries
8504                        .iter()
8505                        .map(|entry| entry.exit_code)
8506                        .collect::<Vec<_>>(),
8507                    vec![Some(23), Some(23)]
8508                );
8509                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
8510                return;
8511            }
8512            assert!(
8513                Instant::now() < deadline,
8514                "module did not retain two terminal exits: {history:?}"
8515            );
8516            sleep(Duration::from_millis(10)).await;
8517        }
8518    }
8519
8520    /// A disable issued while a crash respawn is still backing off must preempt
8521    /// that respawn: the operator's stop wins, the disable must not queue behind
8522    /// the backoff, and the module must never come back up afterwards.
8523    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8524    async fn disable_during_crash_backoff_cancels_pending_respawn() {
8525        let backoff = Duration::from_secs(2);
8526        let supervisor = Supervisor::new(
8527            Arc::new(Registry::default()),
8528            RestartPolicy::new(10, backoff),
8529        );
8530        let module = supervisor
8531            .spawn(ModuleSpec {
8532                module_id: "disable-during-backoff".to_string(),
8533                program: fake_aft_stub_path(),
8534                args: Vec::new(),
8535                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8536                reserved: false,
8537                reserved_prefixes: Vec::new(),
8538                protocol: ModuleProtocol::Subc,
8539                overlap: Default::default(),
8540            })
8541            .unwrap();
8542
8543        // Wait for the first crash to put the module into its backoff window.
8544        let deadline = Instant::now() + Duration::from_secs(5);
8545        loop {
8546            if module.status().unwrap().state == ModuleState::Restarting {
8547                break;
8548            }
8549            assert!(
8550                Instant::now() < deadline,
8551                "module never entered the crash backoff"
8552            );
8553            sleep(Duration::from_millis(10)).await;
8554        }
8555
8556        let started = Instant::now();
8557        module.set_enabled(false).await.unwrap();
8558        let waited = started.elapsed();
8559
8560        assert!(
8561            waited < backoff / 2,
8562            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
8563        );
8564        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
8565
8566        // Outlast the backoff: the respawn it was counting down to must never run.
8567        sleep(backoff + Duration::from_millis(500)).await;
8568        let status = module.status().unwrap();
8569        assert_eq!(status.state, ModuleState::Disabled);
8570        assert_eq!(
8571            status.spawn_generation, 1,
8572            "module respawned after the operator disabled it"
8573        );
8574    }
8575
8576    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
8577    /// the shape of nats-server, the program this rule exists for.
8578    #[cfg(unix)]
8579    fn protocol_none_sigterm_exits_clean_spec(
8580        module_id: &str,
8581        dir: &std::path::Path,
8582    ) -> (ModuleSpec, PathBuf, PathBuf) {
8583        let ready = dir.join("ready");
8584        let marker = dir.join("sigterm");
8585        let spec = ModuleSpec {
8586            module_id: module_id.to_string(),
8587            program: fake_aft_stub_path(),
8588            args: Vec::new(),
8589            env: vec![
8590                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
8591                (
8592                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
8593                    marker.display().to_string(),
8594                ),
8595                (
8596                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
8597                    ready.display().to_string(),
8598                ),
8599            ],
8600            reserved: false,
8601            reserved_prefixes: Vec::new(),
8602            protocol: ModuleProtocol::None,
8603            overlap: Default::default(),
8604        };
8605        (spec, ready, marker)
8606    }
8607
8608    /// Wait for a file the child writes, so a signal is never sent before the
8609    /// child's SIGTERM handler is installed (the default disposition would
8610    /// kill it by signal and the exit would not be clean).
8611    #[cfg(unix)]
8612    async fn wait_for_file(path: &std::path::Path) {
8613        let deadline = Instant::now() + Duration::from_secs(10);
8614        while !path.exists() {
8615            assert!(
8616                Instant::now() < deadline,
8617                "{} never appeared",
8618                path.display()
8619            );
8620            sleep(Duration::from_millis(10)).await;
8621        }
8622    }
8623
8624    /// A protocol-none module that exits 0 because something OUTSIDE the
8625    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
8626    /// the crash-path disposition rather than `stopped`.
8627    #[cfg(unix)]
8628    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8629    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
8630        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
8631        let (spec, ready, marker) =
8632            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
8633        let supervisor = Supervisor::new(
8634            Arc::new(Registry::default()),
8635            RestartPolicy::new(3, Duration::ZERO),
8636        );
8637        let module = supervisor.spawn(spec).unwrap();
8638        wait_for_file(&ready).await;
8639        let first_pid = module
8640            .status()
8641            .unwrap()
8642            .pid
8643            .expect("a running module reports its pid");
8644
8645        rustix::process::kill_process(
8646            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
8647            rustix::process::Signal::TERM,
8648        )
8649        .unwrap();
8650
8651        let deadline = Instant::now() + Duration::from_secs(10);
8652        let respawned = loop {
8653            let status = module.status().unwrap();
8654            if status.state == ModuleState::Running
8655                && status.pid.is_some_and(|pid| pid != first_pid)
8656            {
8657                break status;
8658            }
8659            assert!(
8660                Instant::now() < deadline,
8661                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
8662            );
8663            sleep(Duration::from_millis(10)).await;
8664        };
8665        assert_eq!(respawned.spawn_generation, 2);
8666        assert!(
8667            marker.exists(),
8668            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
8669        );
8670
8671        let history = module.terminal_history();
8672        assert_eq!(history.entries.len(), 1, "{history:?}");
8673        let entry = &history.entries[0];
8674        assert_eq!(entry.exit_code, Some(0));
8675        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
8676        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
8677
8678        module.stop().await.unwrap();
8679    }
8680
8681    /// Repeated unrequested clean exits of a protocol-none module spend the
8682    /// restart budget exactly as crashes do, and the module ends `failed` with
8683    /// the budget named.
8684    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8685    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
8686        let supervisor = Supervisor::new(
8687            Arc::new(Registry::default()),
8688            RestartPolicy::new(1, Duration::ZERO),
8689        );
8690        let module = supervisor
8691            .spawn(ModuleSpec {
8692                module_id: "none-clean-exit-budget".to_string(),
8693                program: fake_aft_stub_path(),
8694                args: Vec::new(),
8695                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8696                reserved: false,
8697                reserved_prefixes: Vec::new(),
8698                protocol: ModuleProtocol::None,
8699                overlap: Default::default(),
8700            })
8701            .unwrap();
8702
8703        let deadline = Instant::now() + Duration::from_secs(10);
8704        loop {
8705            let status = module.status().unwrap();
8706            if status.state == ModuleState::Failed {
8707                break;
8708            }
8709            assert!(
8710                Instant::now() < deadline,
8711                "module never exhausted its budget: {status:?} {:?}",
8712                module.terminal_history()
8713            );
8714            sleep(Duration::from_millis(10)).await;
8715        }
8716        let history = module.terminal_history();
8717        assert_eq!(
8718            history
8719                .entries
8720                .iter()
8721                .map(|entry| (entry.exit_code, entry.disposition.clone()))
8722                .collect::<Vec<_>>(),
8723            vec![
8724                (Some(0), TerminalDisposition::Restarting),
8725                (Some(0), TerminalDisposition::Failed),
8726            ]
8727        );
8728        let detail = history.entries[1]
8729            .disposition_detail
8730            .as_deref()
8731            .expect("a budget failure names the budget");
8732        assert!(detail.contains("max_restarts=1"), "{detail}");
8733        assert_eq!(module.status().unwrap().spawn_generation, 2);
8734    }
8735
8736    /// A stop the supervisor itself requests still stops a protocol-none
8737    /// module, even though the child answers the SIGTERM with exit 0.
8738    #[cfg(unix)]
8739    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8740    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
8741        for disable in [false, true] {
8742            let label = if disable {
8743                "none-requested-disable"
8744            } else {
8745                "none-requested-stop"
8746            };
8747            let dir = subc_test_support::TestTempDir::new(label);
8748            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
8749            let supervisor = Supervisor::new(
8750                Arc::new(Registry::default()),
8751                RestartPolicy::new(3, Duration::ZERO),
8752            );
8753            let module = supervisor.spawn(spec).unwrap();
8754            wait_for_file(&ready).await;
8755
8756            if disable {
8757                module.set_enabled(false).await.unwrap();
8758            } else {
8759                module.stop().await.unwrap();
8760            }
8761            assert!(
8762                marker.exists(),
8763                "{label}: the child must have left through its SIGTERM handler with exit 0"
8764            );
8765
8766            // Long enough for a zero-backoff respawn to have happened if the
8767            // exit had been treated as a crash.
8768            sleep(Duration::from_millis(500)).await;
8769            let status = module.status().unwrap();
8770            let expected = if disable {
8771                ModuleState::Disabled
8772            } else {
8773                ModuleState::Stopped
8774            };
8775            assert_eq!(status.state, expected, "{label}");
8776            assert_eq!(
8777                status.spawn_generation, 1,
8778                "{label}: respawned after a requested stop"
8779            );
8780            let history = module.terminal_history();
8781            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
8782            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
8783            assert_ne!(
8784                history.entries[0].disposition,
8785                TerminalDisposition::Restarting,
8786                "{label}"
8787            );
8788        }
8789    }
8790
8791    /// A subc-wire module that exits 0 on its own is still a stop: the
8792    /// protocol-none rule must not reach it.
8793    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8794    async fn subc_wire_clean_exit_is_still_a_stop() {
8795        let supervisor = Supervisor::new(
8796            Arc::new(Registry::default()),
8797            RestartPolicy::new(3, Duration::ZERO),
8798        );
8799        let module = supervisor
8800            .spawn(ModuleSpec {
8801                module_id: "wire-clean-exit".to_string(),
8802                program: fake_aft_stub_path(),
8803                args: Vec::new(),
8804                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8805                reserved: false,
8806                reserved_prefixes: Vec::new(),
8807                protocol: ModuleProtocol::Subc,
8808                overlap: Default::default(),
8809            })
8810            .unwrap();
8811
8812        let deadline = Instant::now() + Duration::from_secs(10);
8813        while module.terminal_history().entries.is_empty() {
8814            assert!(Instant::now() < deadline, "module never exited");
8815            sleep(Duration::from_millis(10)).await;
8816        }
8817        // Long enough for a zero-backoff respawn to have happened.
8818        sleep(Duration::from_millis(500)).await;
8819        let status = module.status().unwrap();
8820        assert_eq!(status.state, ModuleState::Stopped);
8821        assert_eq!(status.spawn_generation, 1);
8822        let history = module.terminal_history();
8823        assert_eq!(history.entries.len(), 1, "{history:?}");
8824        assert_eq!(history.entries[0].exit_code, Some(0));
8825        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
8826    }
8827
8828    #[cfg(unix)]
8829    #[tokio::test]
8830    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
8831        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
8832        let record = dir.join("live-children.json");
8833        let supervisor = Supervisor::new(
8834            Arc::new(Registry::default()),
8835            RestartPolicy::new(0, Duration::ZERO),
8836        );
8837        let mut runtime = supervisor.runtime_config();
8838        runtime.child_roster.record_to(record.clone());
8839        let gate = Arc::new(super::ReloadExitRecordGate::default());
8840        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
8841        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8842        let spec = ModuleSpec {
8843            module_id: "reload-exit-roster".into(),
8844            program: fake_aft_stub_path(),
8845            args: Vec::new(),
8846            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
8847            reserved: false,
8848            reserved_prefixes: Vec::new(),
8849            protocol: ModuleProtocol::Subc,
8850            overlap: Default::default(),
8851        };
8852        let mut child = None;
8853        let reload = super::finish_reload_child(
8854            &spec,
8855            &runtime,
8856            &supervisor.registry,
8857            &supervisor.process_liveness,
8858            &snapshot,
8859            &mut child,
8860        );
8861        tokio::pin!(reload);
8862        tokio::select! {
8863            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
8864            _ = gate.reached.notified() => {}
8865        }
8866        assert!(runtime
8867            .terminal_ring
8868            .lock()
8869            .unwrap()
8870            .snapshot()
8871            .entries
8872            .is_empty());
8873        assert_eq!(
8874            crate::live_children::read_record(&record).unwrap().len(),
8875            1,
8876            "shutdown must still wait for the reaped child until its terminal record exists"
8877        );
8878        runtime.child_roster.close();
8879        gate.resume.notify_one();
8880        assert!(reload.await.is_err());
8881        assert!(crate::live_children::read_record(&record)
8882            .unwrap()
8883            .is_empty());
8884        let history = runtime.terminal_ring.lock().unwrap().snapshot();
8885        assert_eq!(history.entries.len(), 1);
8886        assert_eq!(
8887            history.entries[0].disposition,
8888            TerminalDisposition::DaemonShutdown
8889        );
8890    }
8891
8892    /// Each restart-producing arm has its own state transition. Keeping their
8893    /// lifetime count assertions adjacent prevents a later new arm from silently
8894    /// spending budget without recording the historical restart.
8895    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8896    async fn every_restart_increment_path_advances_lifetime_count() {
8897        let supervisor = Supervisor::new(
8898            Arc::new(Registry::default()),
8899            RestartPolicy::new(1, Duration::ZERO),
8900        );
8901        let runtime = supervisor.runtime_config();
8902        let spec = ModuleSpec {
8903            module_id: "lifetime-increment-path".to_string(),
8904            program: PathBuf::from("/unused/lifetime-increment-path"),
8905            args: Vec::new(),
8906            env: Vec::new(),
8907            reserved: false,
8908            reserved_prefixes: Vec::new(),
8909            protocol: ModuleProtocol::Subc,
8910            overlap: Default::default(),
8911        };
8912
8913        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8914        assert!(matches!(
8915            on_child_exit(
8916                &spec,
8917                runtime.restart_policy,
8918                &supervisor.registry,
8919                &crash_snapshot,
8920                &runtime.terminal_ring,
8921                &runtime.spawn_events,
8922                &runtime.child_roster,
8923                ExitReport {
8924                    kind: ExitKind::Crash,
8925                    code: Some(1),
8926                    signal: None,
8927                    at_ms: 1,
8928                },
8929            )
8930            .await,
8931            NextAction::Restart { schedule: _ }
8932        ));
8933        let (crash_restarts, crash_lifetime) = {
8934            let state = lock_snapshot(&crash_snapshot).unwrap();
8935            (state.crash_restarts.len(), state.lifetime_restarts)
8936        };
8937        assert_eq!(crash_restarts, 1);
8938        assert_eq!(crash_lifetime, 1);
8939
8940        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8941        let mut health_child = None;
8942        assert!(matches!(
8943            health_restart_child(
8944                &spec,
8945                &runtime,
8946                &supervisor.registry,
8947                &supervisor.process_liveness,
8948                &health_snapshot,
8949                &mut health_child,
8950                SupervisorHealthStatus::Failing,
8951                None,
8952                2,
8953            )
8954            .await,
8955            Ok(())
8956        ));
8957        assert!(health_child.is_none());
8958        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
8959        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
8960        let (health_restarts, health_lifetime) = {
8961            let state = lock_snapshot(&health_snapshot).unwrap();
8962            (state.crash_restarts.len(), state.lifetime_restarts)
8963        };
8964        assert_eq!(health_restarts, 1);
8965        assert_eq!(health_lifetime, 1);
8966
8967        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8968        let mut reload_child = None;
8969        assert!(matches!(
8970            handle_reload_spawn_failure(
8971                &spec,
8972                &runtime,
8973                &supervisor.process_liveness,
8974                &reload_snapshot,
8975                &mut reload_child,
8976                "forced reload spawn failure".to_string(),
8977            )
8978            .await,
8979            Err(SuperviseError::ReloadFailed { .. })
8980        ));
8981        let (reload_restarts, reload_lifetime) = {
8982            let state = lock_snapshot(&reload_snapshot).unwrap();
8983            (state.crash_restarts.len(), state.lifetime_restarts)
8984        };
8985        assert_eq!(reload_restarts, 1);
8986        assert_eq!(reload_lifetime, 1);
8987    }
8988
8989    #[tokio::test]
8990    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
8991        let supervisor = Supervisor::new(
8992            Arc::new(Registry::default()),
8993            RestartPolicy::new(3, Duration::ZERO),
8994        );
8995        let runtime = supervisor.runtime_config();
8996        let spec = ModuleSpec {
8997            module_id: "deliberately-severed".to_string(),
8998            program: PathBuf::from("/unused/deliberately-severed"),
8999            args: Vec::new(),
9000            env: Vec::new(),
9001            reserved: false,
9002            reserved_prefixes: Vec::new(),
9003            protocol: ModuleProtocol::Subc,
9004            overlap: Default::default(),
9005        };
9006        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9007        let process = ProcessIdentity {
9008            pid: 41,
9009            start_time: 101,
9010        };
9011        record_deliberate_severance(&snapshot, process).unwrap();
9012        let exit_report = apply_deliberate_severance_marker(
9013            &snapshot,
9014            Some(process),
9015            ExitReport {
9016                kind: ExitKind::Crash,
9017                code: Some(1),
9018                signal: None,
9019                at_ms: 1,
9020            },
9021        );
9022        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
9023
9024        assert!(matches!(
9025            on_child_exit(
9026                &spec,
9027                runtime.restart_policy,
9028                &supervisor.registry,
9029                &snapshot,
9030                &runtime.terminal_ring,
9031                &runtime.spawn_events,
9032                &runtime.child_roster,
9033                exit_report,
9034            )
9035            .await,
9036            NextAction::Restart { schedule: _ }
9037        ));
9038        let state = lock_snapshot(&snapshot).unwrap();
9039        assert_eq!(state.lifetime_restarts, 1);
9040        assert_eq!(state.crash_restarts.len(), 0);
9041    }
9042
9043    #[tokio::test]
9044    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
9045        let supervisor = Supervisor::new(
9046            Arc::new(Registry::default()),
9047            RestartPolicy::new(3, Duration::ZERO),
9048        );
9049        let runtime = supervisor.runtime_config();
9050        let spec = ModuleSpec {
9051            module_id: "genuine-crash".to_string(),
9052            program: PathBuf::from("/unused/genuine-crash"),
9053            args: Vec::new(),
9054            env: Vec::new(),
9055            reserved: false,
9056            reserved_prefixes: Vec::new(),
9057            protocol: ModuleProtocol::Subc,
9058            overlap: Default::default(),
9059        };
9060        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9061
9062        assert!(matches!(
9063            on_child_exit(
9064                &spec,
9065                runtime.restart_policy,
9066                &supervisor.registry,
9067                &snapshot,
9068                &runtime.terminal_ring,
9069                &runtime.spawn_events,
9070                &runtime.child_roster,
9071                ExitReport {
9072                    kind: ExitKind::Crash,
9073                    code: Some(1),
9074                    signal: None,
9075                    at_ms: 1,
9076                },
9077            )
9078            .await,
9079            NextAction::Restart { schedule: _ }
9080        ));
9081        let state = lock_snapshot(&snapshot).unwrap();
9082        assert_eq!(state.lifetime_restarts, 1);
9083        assert_eq!(state.crash_restarts.len(), 1);
9084    }
9085
9086    fn crash_exit_report(at_ms: u64) -> ExitReport {
9087        ExitReport {
9088            kind: ExitKind::Crash,
9089            code: Some(1),
9090            signal: None,
9091            at_ms,
9092        }
9093    }
9094
9095    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
9096        ModuleSpec {
9097            module_id: module_id.to_string(),
9098            program: PathBuf::from("/unused").join(module_id),
9099            args: Vec::new(),
9100            env: Vec::new(),
9101            reserved: false,
9102            reserved_prefixes: Vec::new(),
9103            protocol: ModuleProtocol::Subc,
9104            overlap: Default::default(),
9105        }
9106    }
9107
9108    /// A real crash loop still stops. Three crashes with nothing aging out spend
9109    /// a budget of two and the third respawn is refused, and both surfaces an
9110    /// operator has -- the log line and the retained terminal record -- name the
9111    /// window rather than only the cap, because `max_restarts=2` alone is what
9112    /// this budget used to mean.
9113    #[tokio::test]
9114    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
9115        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
9116        let supervisor = Supervisor::new(
9117            Arc::new(Registry::default()),
9118            RestartPolicy::new(2, Duration::ZERO),
9119        );
9120        let runtime = supervisor.runtime_config();
9121        let spec = windowed_crash_spec("crash-loop-in-window");
9122        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9123
9124        for attempt in 1..=2 {
9125            assert!(
9126                matches!(
9127                    on_child_exit(
9128                        &spec,
9129                        runtime.restart_policy,
9130                        &supervisor.registry,
9131                        &snapshot,
9132                        &runtime.terminal_ring,
9133                        &runtime.spawn_events,
9134                        &runtime.child_roster,
9135                        crash_exit_report(attempt),
9136                    )
9137                    .await,
9138                    NextAction::Restart { schedule: _ }
9139                ),
9140                "crash {attempt} is inside the budget and must respawn"
9141            );
9142        }
9143
9144        assert!(matches!(
9145            on_child_exit(
9146                &spec,
9147                runtime.restart_policy,
9148                &supervisor.registry,
9149                &snapshot,
9150                &runtime.terminal_ring,
9151                &runtime.spawn_events,
9152                &runtime.child_roster,
9153                crash_exit_report(3),
9154            )
9155            .await,
9156            NextAction::Stop { .. }
9157        ));
9158
9159        {
9160            let state = lock_snapshot(&snapshot).unwrap();
9161            assert_eq!(state.state, ModuleState::Failed);
9162            assert_eq!(state.crash_restarts.len(), 2);
9163            assert_eq!(state.lifetime_restarts, 2);
9164        }
9165
9166        let history = runtime
9167            .terminal_ring
9168            .lock()
9169            .expect("terminal ring is not poisoned")
9170            .snapshot();
9171        let last = history
9172            .entries
9173            .last()
9174            .expect("the refused crash is retained");
9175        assert_eq!(last.disposition, TerminalDisposition::Failed);
9176        assert_eq!(
9177            last.disposition_detail.as_deref(),
9178            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
9179        );
9180
9181        let captured = crate::router::test_log::captured_logs(&logs);
9182        assert!(
9183            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
9184            "the stop must be logged with its window: {captured}"
9185        );
9186    }
9187
9188    /// The rate, stated as a test: three crashes where the first has aged past
9189    /// the window are two crashes as far as the budget is concerned, so the
9190    /// third respawn is allowed and the ring holds only the two recent ones.
9191    ///
9192    /// This is the case a lifetime counter got wrong -- and the case the daemon
9193    /// now hits routinely, since a module exits non-zero every time its
9194    /// connection to the daemon drops.
9195    #[tokio::test]
9196    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
9197        let supervisor = Supervisor::new(
9198            Arc::new(Registry::default()),
9199            RestartPolicy::new(2, Duration::ZERO),
9200        );
9201        let runtime = supervisor.runtime_config();
9202        let spec = windowed_crash_spec("crash-across-windows");
9203        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9204
9205        for attempt in 1..=2 {
9206            assert!(matches!(
9207                on_child_exit(
9208                    &spec,
9209                    runtime.restart_policy,
9210                    &supervisor.registry,
9211                    &snapshot,
9212                    &runtime.terminal_ring,
9213                    &runtime.spawn_events,
9214                    &runtime.child_roster,
9215                    crash_exit_report(attempt),
9216                )
9217                .await,
9218                NextAction::Restart { schedule: _ }
9219            ));
9220        }
9221
9222        // The oldest crash moves out of the window; nothing else about the
9223        // module changes.
9224        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
9225            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
9226        })
9227        .unwrap();
9228
9229        assert!(
9230            matches!(
9231                on_child_exit(
9232                    &spec,
9233                    runtime.restart_policy,
9234                    &supervisor.registry,
9235                    &snapshot,
9236                    &runtime.terminal_ring,
9237                    &runtime.spawn_events,
9238                    &runtime.child_roster,
9239                    crash_exit_report(3),
9240                )
9241                .await,
9242                NextAction::Restart { schedule: _ }
9243            ),
9244            "a crash older than the window must not hold a budget slot"
9245        );
9246
9247        let state = lock_snapshot(&snapshot).unwrap();
9248        assert_eq!(state.state, ModuleState::Restarting);
9249        assert_eq!(
9250            state.crash_restarts.len(),
9251            2,
9252            "the aged instant is dropped and the new one takes its place"
9253        );
9254        assert_eq!(
9255            state.lifetime_restarts, 3,
9256            "the ledger counts every restart, including the ones the window forgot"
9257        );
9258    }
9259
9260    /// An operator restart hands the budget back whole, and the ledger keeps
9261    /// counting. Those are different questions -- "how close is this module to
9262    /// being stopped" and "how many times has it been replaced" -- and the
9263    /// operator action answers only the first.
9264    #[tokio::test]
9265    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
9266        let supervisor = Supervisor::new(
9267            Arc::new(Registry::default()),
9268            RestartPolicy::new(2, Duration::ZERO),
9269        );
9270        let runtime = supervisor.runtime_config();
9271        let spec = windowed_crash_spec("operator-cleared-budget");
9272        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9273
9274        for attempt in 1..=2 {
9275            assert!(matches!(
9276                on_child_exit(
9277                    &spec,
9278                    runtime.restart_policy,
9279                    &supervisor.registry,
9280                    &snapshot,
9281                    &runtime.terminal_ring,
9282                    &runtime.spawn_events,
9283                    &runtime.child_roster,
9284                    crash_exit_report(attempt),
9285                )
9286                .await,
9287                NextAction::Restart { schedule: _ }
9288            ));
9289        }
9290
9291        reset_restart_count(&snapshot, &spec.module_id).unwrap();
9292        {
9293            let state = lock_snapshot(&snapshot).unwrap();
9294            assert!(
9295                state.crash_restarts.is_empty(),
9296                "an operator restart returns the full budget"
9297            );
9298            assert_eq!(
9299                state.lifetime_restarts, 2,
9300                "clearing the budget must not unmake the crashes"
9301            );
9302        }
9303
9304        assert!(
9305            matches!(
9306                on_child_exit(
9307                    &spec,
9308                    runtime.restart_policy,
9309                    &supervisor.registry,
9310                    &snapshot,
9311                    &runtime.terminal_ring,
9312                    &runtime.spawn_events,
9313                    &runtime.child_roster,
9314                    crash_exit_report(3),
9315                )
9316                .await,
9317                NextAction::Restart { schedule: _ }
9318            ),
9319            "the cleared budget must be spendable again"
9320        );
9321        let state = lock_snapshot(&snapshot).unwrap();
9322        assert_eq!(state.crash_restarts.len(), 1);
9323        assert_eq!(state.lifetime_restarts, 3);
9324    }
9325
9326    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9327    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
9328        let severed = ProcessIdentity {
9329            pid: 41,
9330            start_time: 101,
9331        };
9332        let successor = ProcessIdentity {
9333            pid: 41,
9334            start_time: 202,
9335        };
9336        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
9337        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
9338            state.pid = Some(successor.pid);
9339            state.process_start_time = Some(successor.start_time);
9340        })
9341        .unwrap();
9342        assert!(!module.record_deliberate_severance(severed).unwrap());
9343
9344        let exit_report = apply_deliberate_severance_marker(
9345            &module.inner.snapshot,
9346            Some(successor),
9347            ExitReport {
9348                kind: ExitKind::Crash,
9349                code: Some(1),
9350                signal: None,
9351                at_ms: 1,
9352            },
9353        );
9354
9355        assert_eq!(exit_report.kind, ExitKind::Crash);
9356    }
9357
9358    #[tokio::test]
9359    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
9360        let registry = Registry::default();
9361        let supervisor = Supervisor::new(
9362            Arc::new(Registry::default()),
9363            RestartPolicy::new(3, Duration::ZERO),
9364        );
9365        let runtime = supervisor.runtime_config();
9366        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9367        let spec = ModuleSpec {
9368            module_id: "drain-deliberate-severance".to_string(),
9369            program: fake_aft_stub_path(),
9370            args: Vec::new(),
9371            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9372            reserved: false,
9373            reserved_prefixes: Vec::new(),
9374            protocol: ModuleProtocol::Subc,
9375            overlap: Default::default(),
9376        };
9377        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9378        let process = ProcessIdentity {
9379            pid: 41,
9380            start_time: 101,
9381        };
9382        child.process_identity = Some(process);
9383        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
9384            state.pid = Some(process.pid);
9385            state.process_start_time = Some(process.start_time);
9386        })
9387        .unwrap();
9388        record_deliberate_severance(&snapshot, process).unwrap();
9389
9390        drain_child_to_state(
9391            &spec.module_id,
9392            spec.protocol,
9393            // The child exits on its own; no signal may change the exit this
9394            // test classifies.
9395            StopNotice::SentOverConnection,
9396            &registry,
9397            None,
9398            &snapshot,
9399            &runtime.terminal_ring,
9400            &runtime.spawn_events,
9401            child,
9402            Duration::from_secs(1),
9403            ModuleState::Stopped,
9404            Some(false),
9405        )
9406        .await
9407        .unwrap();
9408
9409        let state = lock_snapshot(&snapshot).unwrap();
9410        assert_eq!(
9411            state.last_exit.as_ref().map(|exit| exit.kind),
9412            Some(ExitKind::DeliberateSeverance)
9413        );
9414        assert_eq!(state.lifetime_restarts, 1);
9415        assert_eq!(state.crash_restarts.len(), 0);
9416        drop(state);
9417        let history = runtime.terminal_ring.lock().unwrap().snapshot();
9418        assert_eq!(
9419            history.entries[0].exit_kind,
9420            subc_control::TerminalExitKind::DeliberateSeverance
9421        );
9422    }
9423
9424    #[tokio::test]
9425    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
9426        let registry = Registry::default();
9427        let supervisor = Supervisor::new(
9428            Arc::new(Registry::default()),
9429            RestartPolicy::new(3, Duration::ZERO),
9430        );
9431        let runtime = supervisor.runtime_config();
9432        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9433        let spec = ModuleSpec {
9434            module_id: "ordinary-drain".to_string(),
9435            program: fake_aft_stub_path(),
9436            args: Vec::new(),
9437            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9438            reserved: false,
9439            reserved_prefixes: Vec::new(),
9440            protocol: ModuleProtocol::Subc,
9441            overlap: Default::default(),
9442        };
9443        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9444
9445        drain_child_to_state(
9446            &spec.module_id,
9447            spec.protocol,
9448            // The child exits on its own; no signal may change the exit this
9449            // test classifies.
9450            StopNotice::SentOverConnection,
9451            &registry,
9452            None,
9453            &snapshot,
9454            &runtime.terminal_ring,
9455            &runtime.spawn_events,
9456            child,
9457            Duration::from_secs(1),
9458            ModuleState::Stopped,
9459            Some(false),
9460        )
9461        .await
9462        .unwrap();
9463
9464        let state = lock_snapshot(&snapshot).unwrap();
9465        assert_eq!(
9466            state.last_exit.as_ref().map(|exit| exit.kind),
9467            Some(ExitKind::Crash)
9468        );
9469        assert_eq!(state.lifetime_restarts, 0);
9470        assert_eq!(state.crash_restarts.len(), 0);
9471    }
9472
9473    #[test]
9474    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
9475        // The server's generic fatal-routing branch only knows that the
9476        // connection failed; it does not know that the daemon deliberately
9477        // initiated a process-killing severance. Keep this seam explicit so a
9478        // future connection error path cannot silently reintroduce the stale
9479        // exemption that mislabels a later genuine crash.
9480        assert!(!include_str!("server.rs")
9481            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
9482    }
9483
9484    /// The `route.closed` `drained` value must be the quiescence wait's own
9485    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
9486    /// measurement at all and `false` is the one honest constant. This is the exact
9487    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
9488    /// on every return path, including the one that used to return early via `?`
9489    /// with `route.closing` already sent and no `route.closed` ever following.
9490    #[test]
9491    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
9492        assert!(drained_after_quiescence_wait(&Ok(true)));
9493        assert!(!drained_after_quiescence_wait(&Ok(false)));
9494        assert!(!drained_after_quiescence_wait(&Err(
9495            SuperviseError::StatePoisoned { module_id: None }
9496        )));
9497    }
9498
9499    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
9500    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
9501    /// already reaped out-of-band) still leaves a terminal record rather than none
9502    /// at all. Triggering the real `wait()` I/O error from an integration test would
9503    /// need a genuine already-reaped-child race, which is OS-specific and not
9504    /// something this suite attempts elsewhere; this test instead verifies the
9505    /// record produced for that arm end-to-end through the real `TerminalRing`, and
9506    /// the call site itself is verified by inspection to sit in that exact arm.
9507    #[test]
9508    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
9509        let ring = Arc::new(Mutex::new(TerminalRing::new(
9510            TerminalRingConfig::default(),
9511            0,
9512        )));
9513        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
9514
9515        let snapshot = ring.lock().unwrap().snapshot();
9516        assert_eq!(snapshot.entries.len(), 1);
9517        let entry = &snapshot.entries[0];
9518        assert_eq!(entry.exit_code, None);
9519        assert_eq!(entry.exit_signal, None);
9520        assert_eq!(entry.disposition, TerminalDisposition::Failed);
9521    }
9522
9523    #[test]
9524    fn wait_error_exit_path_preserves_spawn_event_density() {
9525        let feed = super::SpawnEventFeed::default();
9526        feed.configure_incarnation("wait-error-density".to_string());
9527        feed.emit_spawned("wait-error", 41, 1);
9528        let ring = Arc::new(Mutex::new(TerminalRing::new(
9529            TerminalRingConfig::default(),
9530            0,
9531        )));
9532
9533        record_wait_error_terminal("wait-error", &ring, &feed);
9534        feed.emit_spawned("after-wait-error", 42, 2);
9535
9536        let state = feed.0.lock().unwrap();
9537        let sequences = state
9538            .events
9539            .iter()
9540            .map(|event| event.cursor.seq)
9541            .collect::<Vec<_>>();
9542        assert_eq!(sequences, vec![1, 2, 3]);
9543        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
9544        assert_eq!(state.events[1].exit_code, None);
9545        assert_eq!(state.events[1].exit_signal, None);
9546    }
9547
9548    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
9549    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
9550    /// not a clean exit it never actually observed.
9551    #[test]
9552    fn wait_error_exit_report_is_classified_as_a_crash() {
9553        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
9554    }
9555}
9556
9557#[cfg(test)]
9558mod health_evidence_tests {
9559    use super::{HealthProbeError, HealthProbeEvidence};
9560    use std::collections::HashSet;
9561
9562    /// The evidential asymmetry, asserted rather than described.
9563    ///
9564    /// Exactly ONE observation is proof a module cannot serve, and the one that
9565    /// fires under CPU starvation is not it. Before the split, all fifteen
9566    /// construction sites collapsed into a single String, so a timeout carried the
9567    /// same weight as a dead lane -- which is how a healthy module was restarted
9568    /// three times in one day.
9569    #[test]
9570    fn only_a_dead_lane_is_proof_of_death() {
9571        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
9572        // Three non-proof classes, each for a different reason: silence is
9573        // consistent with health, a bad answer proves the module ALIVE, and a
9574        // daemon-side fault never reached the module at all.
9575        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
9576        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
9577        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
9578    }
9579
9580    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
9581    ///
9582    /// A shared label renders two different observations identically in the line an
9583    /// operator reads after an unexplained restart -- the exact confusion this
9584    /// change removes.
9585    #[test]
9586    fn every_evidence_class_has_a_distinct_label() {
9587        let labels = [
9588            HealthProbeError::lane_dead("").label(),
9589            HealthProbeError::no_answer("").label(),
9590            HealthProbeError::bad_answer("").label(),
9591            HealthProbeError::misconfigured("").label(),
9592        ];
9593        let unique: HashSet<_> = labels.iter().collect();
9594        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
9595    }
9596
9597    /// The class is additional information, not a replacement.
9598    ///
9599    /// An operator needs both "this was silence" and the specific text saying how
9600    /// long we waited; a classification that swallowed the message would trade one
9601    /// missing distinction for another.
9602    #[test]
9603    fn classification_preserves_the_original_message() {
9604        let err = HealthProbeError::no_answer("module did not answer within 5s");
9605        assert_eq!(err.to_string(), "module did not answer within 5s");
9606        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9607    }
9608}
9609
9610#[cfg(test)]
9611mod health_tombstone_tests {
9612    use std::{path::PathBuf, sync::Arc, time::Duration};
9613
9614    use subc_protocol::{
9615        manifest::Concurrency,
9616        session::{HealthStatus, ModuleControlResponse},
9617    };
9618    use tokio::sync::mpsc;
9619
9620    use super::{
9621        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
9622        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
9623    };
9624    use crate::{
9625        control::ControlHandler,
9626        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
9627        registry::{ConnectionId, Registry},
9628        router::FrameSink,
9629    };
9630
9631    struct ProbeHarness {
9632        spec: ModuleSpec,
9633        runtime: SupervisorRuntimeConfig,
9634        forwarding: Arc<ForwardingTable>,
9635        module_connection: ConnectionId,
9636        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
9637        handler: ControlHandler,
9638        module: super::SupervisedModule,
9639    }
9640
9641    fn probe_harness() -> ProbeHarness {
9642        let registry = Arc::new(Registry::default());
9643        let forwarding = Arc::new(ForwardingTable::default());
9644        let supervisor_handle = super::SupervisorHandle::new();
9645        let health = HealthConfig {
9646            cadence: Duration::from_secs(30),
9647            deadline: Duration::from_secs(5),
9648            failure_threshold: 3,
9649            on_degraded: HealthAction::Report,
9650            on_failing: HealthAction::Report,
9651            critical: false,
9652        };
9653        let supervisor = Supervisor::new(Arc::clone(&registry), RestartPolicy::default())
9654            .with_forwarding(Arc::clone(&forwarding))
9655            .with_handle(supervisor_handle.clone())
9656            .with_health_config(health);
9657        let spec = ModuleSpec {
9658            module_id: "late-health-module".to_string(),
9659            program: PathBuf::from("disabled-module"),
9660            args: Vec::new(),
9661            env: Vec::new(),
9662            reserved: false,
9663            reserved_prefixes: Vec::new(),
9664            protocol: ModuleProtocol::Subc,
9665            overlap: Default::default(),
9666        };
9667        let module = supervisor
9668            .supervise_configured(spec.clone(), false)
9669            .unwrap();
9670        let runtime = supervisor.runtime_config();
9671        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
9672            .with_supervisor(supervisor_handle);
9673        let module_connection = ConnectionId::new(700);
9674        let (module_tx, module_rx) = mpsc::channel(8);
9675        forwarding
9676            .register_module_connection(
9677                module_connection,
9678                spec.module_id.clone(),
9679                subc_protocol::PROTOCOL_VERSION,
9680                Concurrency::ModuleManaged,
9681                FrameSink::new(module_tx),
9682            )
9683            .unwrap();
9684
9685        ProbeHarness {
9686            spec,
9687            runtime,
9688            forwarding,
9689            module_connection,
9690            module_rx,
9691            handler,
9692            module,
9693        }
9694    }
9695
9696    async fn finish_after(
9697        harness: &mut ProbeHarness,
9698        stall: Duration,
9699    ) -> ModuleControlRpcCompletion {
9700        assert!(stall > harness.runtime.health.deadline);
9701        let deadline = harness.runtime.health.deadline;
9702        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9703        let answer = async {
9704            let frame = harness.module_rx.recv().await.expect("health.check frame");
9705            tokio::time::advance(deadline).await;
9706            tokio::task::yield_now().await;
9707            tokio::time::advance(stall - deadline).await;
9708            harness
9709                .forwarding
9710                .complete_module_control_rpc(
9711                    harness.module_connection,
9712                    frame.header.corr,
9713                    Some("health.check"),
9714                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
9715                        status: HealthStatus::Ok,
9716                        detail: None,
9717                        metrics: None,
9718                    }),
9719                )
9720                .unwrap()
9721        };
9722        let (probe_result, completion) = tokio::join!(probe, answer);
9723        let err = probe_result.expect_err("probe must miss its deadline");
9724        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9725        completion
9726    }
9727
9728    async fn time_out_without_answer(harness: &mut ProbeHarness) {
9729        let deadline = harness.runtime.health.deadline;
9730        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9731        let exhaust_deadline = async {
9732            let _frame = harness.module_rx.recv().await.expect("health.check frame");
9733            tokio::time::advance(deadline).await;
9734            tokio::task::yield_now().await;
9735        };
9736        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
9737        let err = probe_result.expect_err("probe must miss its deadline");
9738        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9739    }
9740
9741    #[tokio::test(start_paused = true)]
9742    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
9743        let mut harness = probe_harness();
9744
9745        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
9746        let first_latency = match &first {
9747            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9748            other => panic!("late answer was not retained: {other:?}"),
9749        };
9750        assert!(harness.handler.observe_module_control_completion(first));
9751
9752        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
9753        let second_latency = match &second {
9754            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9755            other => panic!("late answer was not retained: {other:?}"),
9756        };
9757        assert!(harness.handler.observe_module_control_completion(second));
9758
9759        assert_eq!(first_latency, Duration::from_secs(8));
9760        assert_eq!(
9761            second_latency - first_latency,
9762            Duration::from_secs(3),
9763            "latency must grow linearly with the additional stall"
9764        );
9765        let health = harness.module.status().unwrap().health;
9766        assert_eq!(health.late_answer_count, 2);
9767        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
9768    }
9769
9770    /// A module that answers every probe late must never march to the kill
9771    /// threshold: the late answer proves it is alive, so it must clear the miss
9772    /// streak the timeout recorded. Without the reset, a CPU-starved module
9773    /// that serves every probe seconds past the deadline accumulates
9774    /// `consecutive_failures` to the threshold and is killed — the exact
9775    /// sequence from the 2026-08-14 aft disable, where the daemon logged
9776    /// "proves the module is alive" five times while counting five misses.
9777    #[tokio::test(start_paused = true)]
9778    async fn late_answer_clears_the_consecutive_failure_streak() {
9779        let mut harness = probe_harness();
9780
9781        // Timeout recorded first: the probe path saw no answer in time.
9782        time_out_without_answer(&mut harness).await;
9783        harness
9784            .module
9785            .record_health_probe_failure_for_test("[no-answer] test miss")
9786            .unwrap();
9787        assert_eq!(
9788            harness.module.status().unwrap().health.consecutive_failures,
9789            1,
9790            "precondition: the miss must be on the streak before the late answer"
9791        );
9792
9793        // The stalled reply then lands: proof of life.
9794        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
9795        assert!(matches!(
9796            late,
9797            ModuleControlRpcCompletion::LateHealthAnswer { .. }
9798        ));
9799        assert!(harness.handler.observe_module_control_completion(late));
9800
9801        let health = harness.module.status().unwrap().health;
9802        assert_eq!(
9803            health.consecutive_failures, 0,
9804            "a late answer is an answer: the streak must reset"
9805        );
9806        assert_eq!(health.late_answer_count, 1);
9807    }
9808
9809    #[tokio::test(start_paused = true)]
9810    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
9811        let mut harness = probe_harness();
9812
9813        for _ in 0..20 {
9814            time_out_without_answer(&mut harness).await;
9815            assert_eq!(
9816                harness.forwarding.health_probe_tombstone_count().unwrap(),
9817                1
9818            );
9819        }
9820    }
9821}
9822
9823#[cfg(test)]
9824mod child_env_tests {
9825    use super::{
9826        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
9827        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
9828        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
9829    };
9830    use std::{ffi::OsStr, path::PathBuf};
9831    use tokio::process::Command;
9832
9833    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
9834        ModuleSpec {
9835            module_id: "env-plan".to_string(),
9836            program: PathBuf::from("/nonexistent"),
9837            args: Vec::new(),
9838            env,
9839            reserved: false,
9840            reserved_prefixes: Vec::new(),
9841            protocol: ModuleProtocol::Subc,
9842            overlap: Default::default(),
9843        }
9844    }
9845
9846    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
9847    /// one still gets its own.
9848    ///
9849    /// This is the narrow goal `env_clear()` was reached for, and the reason the
9850    /// fix is `env_remove` rather than deleting the line: an operator's ambient
9851    /// filter silently becoming an unconfigured module's log level is a real
9852    /// defect, just a much smaller one than clearing the environment.
9853    ///
9854    /// Asserted on the command plan rather than a spawned child because proving
9855    /// the ABSENCE of an inherited variable needs the parent's environment
9856    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
9857    /// removal as `(key, None)`, which is exactly the distinction wanted: not
9858    /// "absent because nobody set it" but "explicitly unset for the child".
9859    #[test]
9860    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
9861        let mut command = Command::new("/nonexistent");
9862        apply_child_env(&mut command, &spec(Vec::new()));
9863        let removed = command
9864            .as_std()
9865            .get_envs()
9866            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
9867        assert!(
9868            removed,
9869            "ambient CK_LOG must be explicitly removed for an unconfigured module"
9870        );
9871
9872        let mut configured = Command::new("/nonexistent");
9873        apply_child_env(
9874            &mut configured,
9875            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
9876        );
9877        let effective = configured
9878            .as_std()
9879            .get_envs()
9880            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
9881            .last()
9882            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9883        assert_eq!(
9884            effective,
9885            Some(Some("debug".to_string())),
9886            "a module's configured CK_LOG must survive the ambient removal"
9887        );
9888    }
9889
9890    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
9891    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
9892    /// the same reason as the CK_LOG test above.
9893    ///
9894    /// The argument is the load-bearing half: a stock binary exits on an
9895    /// unknown flag before it listens, so with `--subc` appended the mode
9896    /// could not supervise the one process it exists for. Found by the first
9897    /// conformance run (nats-server: `flag provided but not defined: -subc`).
9898    #[test]
9899    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
9900        let connection_file = std::path::Path::new("/run/subc-connection.json");
9901        let handle = SupervisorHandle::new();
9902
9903        let mut none_spec = spec(Vec::new());
9904        none_spec.protocol = ModuleProtocol::None;
9905        let mut none = Command::new("/nonexistent");
9906        let none_handoff =
9907            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
9908                .expect("protocol-none spawn args apply");
9909        assert!(
9910            none_handoff.is_none(),
9911            "protocol:none spawn must not receive a nonce descriptor"
9912        );
9913        assert!(
9914            !none.as_std().get_envs().any(|(key, value)| key
9915                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
9916                && value.is_some()),
9917            "protocol:none spawn must not name a nonce descriptor"
9918        );
9919        let none_args: Vec<String> = none
9920            .as_std()
9921            .get_args()
9922            .map(|a| a.to_string_lossy().into_owned())
9923            .collect();
9924        assert!(
9925            !none_args.iter().any(|a| a == SUBC_ARG),
9926            "protocol:none argv must not carry --subc; got {none_args:?}"
9927        );
9928        let none_has_nonce = none
9929            .as_std()
9930            .get_envs()
9931            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
9932        assert!(
9933            !none_has_nonce,
9934            "protocol:none spawn must not receive a launch nonce"
9935        );
9936        let none_has_module_id = none
9937            .as_std()
9938            .get_envs()
9939            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
9940        assert!(
9941            none_has_module_id,
9942            "SUBC_MODULE_ID is inert and stays on every path"
9943        );
9944        assert!(
9945            handle.spawn_nonce(&none_spec.module_id).is_none(),
9946            "no nonce record for a process that will never present one"
9947        );
9948
9949        // Control: the subc-wire path is unchanged by the branch above.
9950        let wire_spec = spec(Vec::new());
9951        let mut wire = Command::new("/nonexistent");
9952        let wire_handoff =
9953            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
9954                .expect("subc-wire spawn args apply");
9955        let wire_fd_env = wire
9956            .as_std()
9957            .get_envs()
9958            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
9959            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9960        #[cfg(unix)]
9961        assert_eq!(
9962            wire_fd_env,
9963            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
9964            "a subc-wire spawn names the pipe it will receive at descriptor 3"
9965        );
9966        #[cfg(not(unix))]
9967        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
9968        let wire_args: Vec<String> = wire
9969            .as_std()
9970            .get_args()
9971            .map(|a| a.to_string_lossy().into_owned())
9972            .collect();
9973        assert_eq!(
9974            wire_args,
9975            vec![
9976                SUBC_ARG.to_string(),
9977                connection_file.to_string_lossy().into_owned()
9978            ],
9979            "a subc-wire spawn still carries --subc <path>"
9980        );
9981        assert_eq!(
9982            wire.as_std()
9983                .get_envs()
9984                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
9985            !cfg!(unix),
9986            "only Windows supplies the environment nonce"
9987        );
9988        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
9989    }
9990
9991    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
9992    /// spec tries to set it; only a swap candidate carries it.
9993    ///
9994    /// "Set it only on candidates" is not enough, because spawn applies the
9995    /// spec's env verbatim and the daemon's own environment is inherited: either
9996    /// could hand a plain restart the swap role, and a module reading it would
9997    /// warm on its long swap budget while callers wait. Asserted as an explicit
9998    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
9999    /// test above gives.
10000    #[test]
10001    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
10002        let role = |command: &Command| {
10003            command
10004                .as_std()
10005                .get_envs()
10006                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
10007                .last()
10008                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
10009        };
10010        let forged = spec(vec![(
10011            SUBC_SPAWN_ROLE_ENV.to_string(),
10012            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
10013        )]);
10014
10015        let mut plain = Command::new("/nonexistent");
10016        apply_child_env(&mut plain, &forged);
10017        apply_spawn_role(&mut plain, SpawnRole::Plain);
10018        assert_eq!(
10019            role(&plain),
10020            Some(None),
10021            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
10022        );
10023
10024        let mut candidate = Command::new("/nonexistent");
10025        apply_child_env(&mut candidate, &spec(Vec::new()));
10026        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
10027        assert_eq!(
10028            role(&candidate),
10029            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
10030        );
10031    }
10032
10033    /// Daemon-private capture retention keys never reach the child.
10034    ///
10035    /// cortexkit-log exposes retention as a Rust struct with no environment
10036    /// names, so these entries are supervisor metadata. Passing them through
10037    /// would invent a public child-process contract by accident.
10038    #[test]
10039    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
10040        let mut command = Command::new("/nonexistent");
10041        apply_child_env(
10042            &mut command,
10043            &spec(vec![
10044                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
10045                ("KEPT".to_string(), "yes".to_string()),
10046            ]),
10047        );
10048        let keys: Vec<String> = command
10049            .as_std()
10050            .get_envs()
10051            .filter(|(_, value)| value.is_some())
10052            .map(|(key, _)| key.to_string_lossy().into_owned())
10053            .collect();
10054        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
10055        assert!(
10056            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
10057            "daemon-private capture key leaked to the child: {keys:?}"
10058        );
10059    }
10060}
10061
10062#[cfg(test)]
10063mod jitter_tests {
10064    use super::jittered_health_delay;
10065    use std::{collections::HashSet, time::Duration};
10066
10067    /// Module ids drawn from a real fleet, so the dispersal claim is about names
10068    /// that actually occur rather than invented ones.
10069    ///
10070    /// This is a SAMPLE, not a registry: the property under test is that distinct
10071    /// ids disperse, which holds for any set of distinct strings. Several entries
10072    /// are already historical (modules get renamed), and that costs nothing here --
10073    /// but it means a reader must not mistake this for the live module set, and a
10074    /// rename sweep will match it without there being anything to change.
10075    const FLEET: [&str; 14] = [
10076        "aft",
10077        "alfonso-core",
10078        "magic-context",
10079        "broca",
10080        "thalamus",
10081        "quota",
10082        "engram",
10083        "plexus",
10084        "cerebellum",
10085        "astrocyte",
10086        "synapse",
10087        "subc-mcp",
10088        "cortexkit-credentials",
10089        "subc-federation",
10090    ];
10091
10092    /// Probes must not converge after a fleet-wide restart.
10093    ///
10094    /// This is the property the jitter exists for: every module reconnects at
10095    /// once, and without dispersal all fourteen would then probe on the same
10096    /// tick forever. Nothing failed visibly when this went untested -- a
10097    /// convergent fleet still probes correctly, just in a burst, so the symptom
10098    /// is a periodic load spike that looks like whatever else is running.
10099    #[test]
10100    fn probe_delays_disperse_across_the_fleet() {
10101        let cadence = Duration::from_secs(30);
10102        let delays: HashSet<Duration> = FLEET
10103            .iter()
10104            .map(|id| jittered_health_delay(id, 0, cadence))
10105            .collect();
10106        assert_eq!(
10107            delays.len(),
10108            FLEET.len(),
10109            "every supervised module must land on its own probe offset"
10110        );
10111    }
10112
10113    /// The offset may only ever DELAY a probe, never bring it forward.
10114    ///
10115    /// A delay below the cadence would probe a module more often than
10116    /// configured, which is the opposite of what an operator asked for and
10117    /// would tighten the failure budget without anyone changing it.
10118    #[test]
10119    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
10120        let cadence = Duration::from_secs(30);
10121        let span = cadence / 10;
10122        for id in FLEET {
10123            for probe_index in 0..8 {
10124                let delay = jittered_health_delay(id, probe_index, cadence);
10125                assert!(
10126                    delay >= cadence,
10127                    "{id}#{probe_index}: jitter must not shorten the cadence"
10128                );
10129                assert!(
10130                    delay < cadence + span,
10131                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
10132                );
10133            }
10134        }
10135    }
10136
10137    /// A module keeps its offset across daemon restarts.
10138    ///
10139    /// The delay is derived rather than randomised precisely so a restart does
10140    /// not re-roll every module into a fresh chance of collision. A random
10141    /// source would satisfy the dispersal test above and quietly lose this.
10142    #[test]
10143    fn a_module_offset_is_stable_across_restarts() {
10144        let cadence = Duration::from_secs(30);
10145        for id in FLEET {
10146            assert_eq!(
10147                jittered_health_delay(id, 0, cadence),
10148                jittered_health_delay(id, 0, cadence),
10149                "{id}: the same module and probe index must produce the same offset"
10150            );
10151        }
10152    }
10153
10154    /// A zero cadence disables probing rather than producing a busy loop.
10155    #[test]
10156    fn zero_cadence_yields_zero_delay() {
10157        assert_eq!(
10158            jittered_health_delay("aft", 0, Duration::ZERO),
10159            Duration::ZERO
10160        );
10161    }
10162}
10163
10164#[cfg(all(test, target_os = "linux"))]
10165mod cgroup_placement_tests {
10166    use super::{
10167        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
10168        SupervisedChild,
10169    };
10170    use crate::stderr_tail::{StderrRing, StderrTailConfig};
10171    use std::{
10172        fs, io,
10173        path::{Path, PathBuf},
10174        sync::{Arc, Mutex},
10175    };
10176    use subc_test_support::TestTempDir;
10177    use tokio::process::Command;
10178
10179    #[test]
10180    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
10181        let path = Path::new("/definitely-missing-subc-cgroup");
10182        let mut command = Command::new("true");
10183        let error = apply_cgroup_placement(
10184            &mut command,
10185            &ModuleSpec {
10186                module_id: "broken-cgroup".to_string(),
10187                program: PathBuf::from("true"),
10188                args: Vec::new(),
10189                env: Vec::new(),
10190                reserved: false,
10191                reserved_prefixes: Vec::new(),
10192                protocol: ModuleProtocol::Subc,
10193                overlap: Default::default(),
10194            },
10195            path,
10196        )
10197        .expect_err("a parent cgroup open failure must reject the supervised spawn");
10198        let reason = error.to_string();
10199
10200        assert!(
10201            matches!(error, SuperviseError::Cgroup { .. }),
10202            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
10203        );
10204        assert!(
10205            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
10206            "parent cgroup open failure must name cgroup.procs: {reason}"
10207        );
10208    }
10209
10210    #[tokio::test]
10211    async fn reaping_a_child_removes_its_empty_module_cgroup() {
10212        let root = TestTempDir::new("supervisor-reap-cgroup");
10213        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
10214        let placement = subc_cgroup::prepare_at(&root)
10215            .expect("prepare scratch cgroup root")
10216            .expect("scratch root has a cgroup.procs marker");
10217        let module_id = "reaped-module";
10218        let module = placement
10219            .module_path(module_id)
10220            .expect("create scratch module cgroup");
10221        let child = Command::new("true")
10222            .spawn()
10223            .expect("spawn short-lived child");
10224        let pid = child.id().expect("spawned child has pid");
10225        let mut child = SupervisedChild {
10226            child,
10227            module_id: module_id.to_string(),
10228            cgroup_placement: Some(placement),
10229            stdout_pump: None,
10230            stderr_pump: None,
10231            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
10232            spawned_at_ms: 0,
10233            spawned_from: PathBuf::from("true"),
10234            spawned_file_identity: None,
10235            process_start_time: None,
10236            process_identity: None,
10237            pid,
10238            roster_guard: None,
10239        };
10240
10241        child.wait().await.expect("reap short-lived child");
10242
10243        assert!(
10244            !module.exists(),
10245            "reaping the supervised child must remove its empty cgroup"
10246        );
10247    }
10248
10249    #[test]
10250    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
10251        let root = TestTempDir::new("supervisor-non-empty-cgroup");
10252        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
10253        let placement = subc_cgroup::prepare_at(&root)
10254            .expect("prepare scratch cgroup root")
10255            .expect("scratch root has a cgroup.procs marker");
10256        let module = placement
10257            .module_path("surviving-module")
10258            .expect("create scratch module cgroup");
10259        fs::write(module.join("surviving-process"), b"still present")
10260            .expect("make scratch cgroup non-empty");
10261        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
10262
10263        remove_module_cgroup(&placement, "surviving-module");
10264
10265        let logs = crate::router::test_log::captured_logs(&logs);
10266        assert!(
10267            module.exists(),
10268            "failed removal must leave the cgroup intact"
10269        );
10270        assert!(
10271            logs.contains("could not remove module cgroup after process exit; continuing teardown")
10272                && logs.contains("surviving-module"),
10273            "best-effort removal must report the failure without returning it: {logs}"
10274        );
10275    }
10276
10277    #[test]
10278    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
10279        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
10280        let reason = SuperviseError::Spawn {
10281            program: PathBuf::from("/bin/true"),
10282            source: io::Error::from_raw_os_error(13),
10283            cgroup_path: Some(cgroup_path.clone()),
10284        }
10285        .to_string();
10286
10287        assert!(
10288            reason.contains(&cgroup_path.display().to_string()),
10289            "a pre_exec spawn failure must name the cgroup path: {reason}"
10290        );
10291    }
10292}
10293
10294#[cfg(test)]
10295mod spawn_subscriber_lag_tests {
10296    use super::*;
10297
10298    /// A subscriber whose connection stops draining is dropped once its frame
10299    /// channel fills. The client must learn that from a terminal Error frame
10300    /// after the frames already queued for it, not from a stream that simply
10301    /// goes quiet.
10302    #[tokio::test]
10303    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
10304        let feed = SpawnEventFeed::default();
10305        feed.configure_incarnation("lag-incarnation".to_string());
10306        // A one-slot connection queue that nobody reads until the emits are
10307        // done: the forwarder parks on it and the subscriber channel fills.
10308        let (tx, mut rx) = mpsc::channel(1);
10309        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
10310            .expect("subscribe");
10311        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
10312        for index in 0..emitted {
10313            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
10314            // Let the forwarder take what it can so the fill point is the
10315            // subscriber channel, not a scheduling accident.
10316            tokio::task::yield_now().await;
10317        }
10318        assert_eq!(
10319            feed.subscriber_count(),
10320            0,
10321            "the lagged subscriber must be removed"
10322        );
10323
10324        let mut data = Vec::new();
10325        let mut last = None;
10326        loop {
10327            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
10328                .await
10329                .expect("the forwarder must finish once the subscriber is dropped");
10330            let Some(outbound) = next else { break };
10331            let frame = outbound.frame;
10332            if frame.header.ty == FrameType::StreamData {
10333                assert!(last.is_none(), "no data may follow the terminal frame");
10334                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
10335                data.push(event.cursor.seq);
10336            } else {
10337                assert!(last.is_none(), "exactly one terminal frame");
10338                last = Some(frame);
10339            }
10340        }
10341        assert!(!data.is_empty(), "queued frames drain before the terminal");
10342        for pair in data.windows(2) {
10343            assert_eq!(
10344                pair[1],
10345                pair[0] + 1,
10346                "queued frames arrive dense and in order"
10347            );
10348        }
10349        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
10350        assert_eq!(terminal.header.ty, FrameType::Error);
10351        assert_eq!(terminal.header.corr, 7);
10352        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
10353        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
10354        let detail = body.detail.expect("lagged error carries detail");
10355        assert_eq!(
10356            detail["first_undelivered_cursor"]["seq"],
10357            data.last().unwrap() + 1,
10358            "the named cursor is the first event the subscriber did not receive"
10359        );
10360        assert_eq!(
10361            detail["first_undelivered_cursor"]["daemon_incarnation"],
10362            "lag-incarnation"
10363        );
10364    }
10365}
10366
10367#[cfg(test)]
10368mod terminal_history_read_concurrency_tests {
10369    use super::*;
10370    use crate::terminal_journal::read_pause;
10371    use std::sync::mpsc as std_mpsc;
10372    use subc_test_support::TestTempDir;
10373
10374    fn journaled_ring(
10375        journal: &Arc<crate::terminal_journal::TerminalJournal>,
10376    ) -> Arc<Mutex<TerminalRing>> {
10377        Arc::new(Mutex::new(
10378            TerminalRing::new(TerminalRingConfig::default(), 1)
10379                .with_journal(Some(Arc::clone(journal))),
10380        ))
10381    }
10382
10383    fn crash(at_ms: u64) -> ExitReport {
10384        ExitReport {
10385            kind: ExitKind::Crash,
10386            code: Some(1),
10387            signal: None,
10388            at_ms,
10389        }
10390    }
10391
10392    /// Record an exit on another thread and report whether it finished within
10393    /// `bound`. The recorder thread is left running if it did not.
10394    fn record_within(
10395        module_id: &'static str,
10396        ring: &Arc<Mutex<TerminalRing>>,
10397        at_ms: u64,
10398        bound: Duration,
10399    ) -> bool {
10400        let ring = Arc::clone(ring);
10401        let (done, done_rx) = std_mpsc::channel();
10402        std::thread::spawn(move || {
10403            record_terminal(
10404                module_id,
10405                &ring,
10406                &SpawnEventFeed::default(),
10407                &crash(at_ms),
10408                TerminalDisposition::Restarting,
10409            );
10410            let _ = done.send(());
10411        });
10412        done_rx.recv_timeout(bound).is_ok()
10413    }
10414
10415    /// A history read in progress must not hold the journal writer (which every
10416    /// module's exit recording needs) or the module's own ring. Exits recorded
10417    /// while the read is paused complete promptly; the paused read answers as of
10418    /// the moment it started, and the next read has each exit exactly once.
10419    #[test]
10420    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
10421        let dir = TestTempDir::new("terminal-history-concurrent-read");
10422        let path = dir.join("terminals.jsonl");
10423        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
10424            path.clone(),
10425            "daemon".into(),
10426        ));
10427        let reader_ring = journaled_ring(&journal);
10428        let other_ring = journaled_ring(&journal);
10429        assert!(record_within(
10430            "reader-module",
10431            &reader_ring,
10432            10,
10433            Duration::from_secs(5)
10434        ));
10435
10436        let (started, release) = read_pause::install(&path);
10437        let reading = {
10438            let ring = Arc::clone(&reader_ring);
10439            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
10440        };
10441        started
10442            .recv_timeout(Duration::from_secs(5))
10443            .expect("the history read reached its pause");
10444
10445        let bound = Duration::from_secs(1);
10446        assert!(
10447            record_within("other-module", &other_ring, 20, bound),
10448            "another module's exit waited on a history read (journal writer held)"
10449        );
10450        assert!(
10451            record_within("reader-module", &reader_ring, 30, bound),
10452            "the read module's own exit waited on its history read (ring held)"
10453        );
10454
10455        drop(release);
10456        let paused = reading.join().unwrap();
10457        assert_eq!(
10458            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10459            vec![10],
10460            "an exit recorded after the read began lands in neither half of it"
10461        );
10462        assert_eq!(paused.journal_skipped_lines, 0);
10463        assert_eq!(paused.journal_read_errors, 0);
10464
10465        let after = durable_terminal_history_of(&reader_ring, "reader-module");
10466        assert_eq!(
10467            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10468            vec![10, 30],
10469            "the next read merges ring and journal with no duplicate"
10470        );
10471        assert_eq!(after.journal_skipped_lines, 0);
10472    }
10473}
10474
10475/// What a restart does with the exited process's stderr reader. These drive
10476/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
10477/// holds, so a reader that has not been scheduled by the bound is a controlled
10478/// input rather than something only a loaded machine produces.
10479#[cfg(test)]
10480mod stderr_settle_tests {
10481    use std::{
10482        future::Future,
10483        io,
10484        pin::Pin,
10485        sync::{Arc, Mutex},
10486        task::{Context, Poll},
10487        time::Duration,
10488    };
10489
10490    use tokio::{
10491        io::{AsyncRead, ReadBuf},
10492        sync::oneshot,
10493        time::Instant,
10494    };
10495
10496    use super::{settle_stderr_pump, StderrPump};
10497    use crate::stderr_tail::{
10498        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
10499    };
10500
10501    const BOUND: Duration = Duration::from_millis(250);
10502
10503    /// Yields `before`, then stays pending until the gate is released, then
10504    /// yields `after` and reaches EOF. The bytes after the gate were written
10505    /// by a process that has already exited; only the reader is behind.
10506    struct HeldReader {
10507        before: Option<Vec<u8>>,
10508        gate: Option<oneshot::Receiver<()>>,
10509        after: io::Cursor<Vec<u8>>,
10510    }
10511
10512    impl AsyncRead for HeldReader {
10513        fn poll_read(
10514            mut self: Pin<&mut Self>,
10515            cx: &mut Context<'_>,
10516            buf: &mut ReadBuf<'_>,
10517        ) -> Poll<io::Result<()>> {
10518            if let Some(bytes) = self.before.take() {
10519                buf.put_slice(&bytes);
10520                return Poll::Ready(Ok(()));
10521            }
10522            if let Some(gate) = self.gate.as_mut() {
10523                match Pin::new(gate).poll(cx) {
10524                    Poll::Pending => return Poll::Pending,
10525                    Poll::Ready(_) => self.gate = None,
10526                }
10527            }
10528            Pin::new(&mut self.after).poll_read(cx, buf)
10529        }
10530    }
10531
10532    struct DiscardSink;
10533
10534    impl OutputSink for DiscardSink {
10535        fn write_line(&mut self, _line: &[u8]) {}
10536    }
10537
10538    fn line(text: &str) -> TailEntry {
10539        TailEntry::Line {
10540            text: text.to_string(),
10541            truncated: false,
10542            at_ms: None,
10543        }
10544    }
10545
10546    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
10547        ring.lock().unwrap()
10548    }
10549
10550    /// Start a reader for a new process generation that delivers `before`
10551    /// immediately and `after` only once the returned sender fires (or is
10552    /// dropped).
10553    fn held_pump(
10554        ring: &Arc<Mutex<StderrRing>>,
10555        before: &str,
10556        after: &str,
10557    ) -> (StderrPump, oneshot::Sender<()>) {
10558        let generation = lock(ring).begin_process();
10559        let (release, gate) = oneshot::channel();
10560        let reader = HeldReader {
10561            before: Some(before.as_bytes().to_vec()),
10562            gate: Some(gate),
10563            after: io::Cursor::new(after.as_bytes().to_vec()),
10564        };
10565        let task = tokio::spawn(pump_stderr_to(
10566            reader,
10567            Arc::clone(ring),
10568            generation,
10569            DiscardSink,
10570        ));
10571        (StderrPump { task, generation }, release)
10572    }
10573
10574    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
10575        for _ in 0..1000 {
10576            if done(&lock(ring)) {
10577                return;
10578            }
10579            tokio::time::sleep(Duration::from_millis(1)).await;
10580        }
10581        panic!(
10582            "ring never reached the expected state: {:?}",
10583            lock(ring).snapshot(None, None)
10584        );
10585    }
10586
10587    #[tokio::test(start_paused = true)]
10588    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
10589        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10590        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
10591
10592        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
10593        let before_release = lock(&ring).snapshot(None, None);
10594        assert!(
10595            matches!(before_release.capture, CaptureState::Incomplete { .. }),
10596            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
10597        );
10598
10599        // The restart: the next process starts and writes before the old
10600        // reader catches up.
10601        let next = lock(&ring).begin_process();
10602        lock(&ring).push_line_from(next, "next process booting");
10603        release.send(()).unwrap();
10604        wait_until(&ring, |ring| {
10605            ring.snapshot(None, None).capture == CaptureState::Captured
10606        })
10607        .await;
10608
10609        assert_eq!(
10610            untimed(lock(&ring).snapshot(None, None).entries),
10611            vec![
10612                line("booting"),
10613                line("config error: missing storage"),
10614                TailEntry::ProcessStart,
10615                line("next process booting"),
10616            ],
10617            "the crash's last line must survive a slow reader and stay in the crashed process's section"
10618        );
10619    }
10620
10621    #[tokio::test(start_paused = true)]
10622    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
10623    ) {
10624        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10625        // `_held` is never fired: a descendant keeps the pipe open for the
10626        // whole test.
10627        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
10628
10629        let started = Instant::now();
10630        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
10631        assert_eq!(
10632            started.elapsed(),
10633            BOUND,
10634            "the restart must wait exactly the bound for a pipe that stays open, no longer"
10635        );
10636
10637        let next = lock(&ring).begin_process();
10638        lock(&ring).push_line_from(next, "next process booting");
10639        tokio::time::sleep(Duration::from_secs(60)).await;
10640
10641        let snapshot = lock(&ring).snapshot(None, None);
10642        match &snapshot.capture {
10643            CaptureState::Incomplete { reason } => assert!(
10644                reason.contains("had not reached EOF") && reason.contains("250ms"),
10645                "the reason must say what is missing and after how long: {reason}"
10646            ),
10647            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
10648        }
10649        assert_eq!(
10650            untimed(snapshot.entries),
10651            vec![
10652                line("parent exiting"),
10653                TailEntry::ProcessStart,
10654                line("next process booting"),
10655            ]
10656        );
10657    }
10658
10659    #[tokio::test(start_paused = true)]
10660    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
10661        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10662        let (pump, release) = held_pump(&ring, "one\n", "two\n");
10663        release.send(()).unwrap();
10664
10665        settle_stderr_pump("clean", &ring, pump, BOUND).await;
10666
10667        let snapshot = lock(&ring).snapshot(None, None);
10668        assert_eq!(snapshot.capture, CaptureState::Captured);
10669        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
10670    }
10671}
10672
10673/// Containment of a module's process tree (issue #109).
10674///
10675/// The behaviour these defend against is a module helper surviving its module:
10676/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
10677/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
10678/// compounds it.
10679///
10680/// They run against the SUPERVISOR rather than the job-object crate because the
10681/// claim is about teardown: a crate-level test proves a job can reap a tree, not
10682/// that the daemon's drain path reaches it.
10683///
10684/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
10685/// lane there is a separate containment path with its own tests.
10686#[cfg(all(test, windows))]
10687mod job_containment_tests {
10688    use super::*;
10689    use std::{
10690        path::{Path, PathBuf},
10691        sync::{Arc, Mutex},
10692        time::{Duration, Instant},
10693    };
10694    use subc_test_support::TestTempDir;
10695
10696    /// The stub, expected beside this test executable.
10697    ///
10698    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
10699    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
10700    /// failure then reads as a broken test rather than an unbuilt dependency.
10701    fn stub_path() -> PathBuf {
10702        let mut path = std::env::current_exe().expect("current_exe available in tests");
10703        path.pop();
10704        path.pop();
10705        path.push("fake-aft-stub.exe");
10706        assert!(
10707            path.exists(),
10708            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
10709             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
10710            path.display()
10711        );
10712        path
10713    }
10714
10715    /// Poll for the grandchild pid the stub records, and parse it.
10716    fn read_grandchild_pid(path: &Path) -> u32 {
10717        let deadline = Instant::now() + Duration::from_secs(10);
10718        loop {
10719            if let Ok(contents) = std::fs::read_to_string(path) {
10720                if let Ok(pid) = contents.trim().parse() {
10721                    return pid;
10722                }
10723            }
10724            assert!(
10725                Instant::now() < deadline,
10726                "the stub never recorded a grandchild pid at {}",
10727                path.display()
10728            );
10729            std::thread::sleep(Duration::from_millis(10));
10730        }
10731    }
10732
10733    /// Everything one fixture run needs, so the two tests below differ in exactly
10734    /// one place: whether the child is contained.
10735    struct Fixture {
10736        _dir: TestTempDir,
10737        module_id: String,
10738        grandchild: u32,
10739        child: Option<SupervisedChild>,
10740        registry: Arc<Registry>,
10741        snapshot: Arc<Mutex<SupervisorSnapshot>>,
10742        terminal_ring: Arc<Mutex<TerminalRing>>,
10743        spawn_events: SpawnEventFeed,
10744    }
10745
10746    fn fixture(label: &str, module_id: &str) -> Fixture {
10747        let dir = TestTempDir::new(label);
10748        let pid_file = dir.join("grandchild.pid");
10749        let supervisor = Supervisor::new(
10750            Arc::new(Registry::default()),
10751            RestartPolicy::new(3, Duration::ZERO),
10752        );
10753        let runtime = supervisor.runtime_config();
10754        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10755        let spec = ModuleSpec {
10756            module_id: module_id.to_string(),
10757            program: stub_path(),
10758            // Zero args deliberately: a `--subc` argument would make the stub dial
10759            // a daemon that is not there, and the failure would land in the same
10760            // stderr ring this fixture exists to keep quiet.
10761            args: Vec::new(),
10762            env: vec![
10763                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10764                (
10765                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
10766                    pid_file.display().to_string(),
10767                ),
10768            ],
10769            reserved: false,
10770            reserved_prefixes: Vec::new(),
10771            protocol: ModuleProtocol::Subc,
10772            overlap: Default::default(),
10773        };
10774        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
10775            .expect("spawn the supervised fixture");
10776        let grandchild = read_grandchild_pid(&pid_file);
10777        Fixture {
10778            _dir: dir,
10779            module_id: module_id.to_string(),
10780            grandchild,
10781            child: Some(child),
10782            registry: Arc::new(Registry::default()),
10783            snapshot,
10784            terminal_ring: Arc::clone(&runtime.terminal_ring),
10785            spawn_events: SpawnEventFeed::default(),
10786        }
10787    }
10788
10789    impl Fixture {
10790        /// Drain through the supervisor's own teardown path.
10791        async fn drain(&mut self) {
10792            let child = self
10793                .child
10794                .take()
10795                .expect("the fixture child is still present");
10796            drain_child_to_state(
10797                &self.module_id,
10798                ModuleProtocol::Subc,
10799                // No forwarding table in this fixture, so nothing reaches the
10800                // child over a connection.
10801                StopNotice::NotSent,
10802                &self.registry,
10803                None,
10804                &self.snapshot,
10805                &self.terminal_ring,
10806                &self.spawn_events,
10807                child,
10808                Duration::from_millis(500),
10809                ModuleState::Stopped,
10810                Some(false),
10811            )
10812            .await
10813            .expect("drain the supervised fixture");
10814        }
10815    }
10816
10817    /// Teardown reaps the grandchild, not merely the direct child.
10818    ///
10819    /// This is the assertion the change exists for. Before containment the
10820    /// grandchild survived: it is a separate process, and `start_kill` is
10821    /// `TerminateProcess` scoped to one pid.
10822    #[tokio::test]
10823    async fn teardown_reaps_the_grandchild() {
10824        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
10825        let grandchild = fixture.grandchild;
10826
10827        assert!(
10828            subc_jobobject::process_exists(grandchild),
10829            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
10830        );
10831
10832        fixture.drain().await;
10833
10834        assert!(
10835            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10836            "grandchild {grandchild} outlived module teardown: the tree was not contained"
10837        );
10838    }
10839
10840    /// The mutation control: with containment withheld, the grandchild survives
10841    /// the same kill.
10842    ///
10843    /// This is the defect reproduction from #109 — a direct-child kill reaches
10844    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
10845    /// supervisor because `spawn_and_mark_running` now always contains on
10846    /// Windows, which is the point: there is no longer a path that spawns
10847    /// uncontained, so the control has to construct one.
10848    ///
10849    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
10850    /// grandchild ever dies here, that test is passing for a reason unrelated to
10851    /// the job object and the containment claim is unproven.
10852    #[test]
10853    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
10854        let dir = TestTempDir::new("teardown-uncontained");
10855        let pid_file = dir.join("grandchild.pid");
10856        let mut child = std::process::Command::new(stub_path())
10857            .env("FAKE_AFT_NEVER_CONNECT", "1")
10858            .env(
10859                "FAKE_AFT_GRANDCHILD_PID_FILE",
10860                pid_file.display().to_string(),
10861            )
10862            .stdin(std::process::Stdio::null())
10863            .stdout(std::process::Stdio::null())
10864            .stderr(std::process::Stdio::null())
10865            .spawn()
10866            .expect("spawn the uncontained fixture");
10867        let grandchild = read_grandchild_pid(&pid_file);
10868
10869        // Exactly what the pre-fix teardown did: kill the direct child.
10870        child.kill().expect("kill the direct child");
10871        let _ = child.wait();
10872
10873        assert!(
10874            subc_jobobject::process_exists(grandchild),
10875            "grandchild {grandchild} died with the direct child, so this control no longer \
10876             distinguishes contained from uncontained teardown and the regression test is \
10877             passing vacuously"
10878        );
10879
10880        // The orphan this control demonstrates is the leak the fix prevents, so
10881        // the control must not leave one behind.
10882        kill_tree(grandchild);
10883    }
10884
10885    /// Crash durability: closing the containment handle reaps the tree with no
10886    /// teardown code running at all.
10887    ///
10888    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
10889    /// call anything — and it is why containment is a kernel property of the
10890    /// handle rather than a step in the drain. Discovered by getting the
10891    /// mutation control wrong: clearing `job` to "disable" containment instead
10892    /// killed the tree, which is the guarantee, not a mistake.
10893    #[tokio::test]
10894    async fn dropping_containment_reaps_the_grandchild() {
10895        let mut fixture = fixture("drop-containment", "tree-drop");
10896        let grandchild = fixture.grandchild;
10897
10898        assert!(subc_jobobject::process_exists(grandchild));
10899
10900        // No `drain` call, no kill: dropping the handle is the entire mechanism.
10901        fixture.child.as_mut().expect("child present").job = None;
10902
10903        assert!(
10904            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10905            "grandchild {grandchild} survived the containment handle closing, so a daemon \
10906             crash would leave the tree behind"
10907        );
10908    }
10909
10910    /// Kill a pid and its tree, then confirm it is gone.
10911    fn kill_tree(pid: u32) {
10912        let _ = std::process::Command::new("taskkill.exe")
10913            .args(["/PID", &pid.to_string(), "/T", "/F"])
10914            .stdin(std::process::Stdio::null())
10915            .stdout(std::process::Stdio::null())
10916            .stderr(std::process::Stdio::null())
10917            .status();
10918        assert!(
10919            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
10920            "could not clean up grandchild {pid}"
10921        );
10922    }
10923}
10924
10925/// The daemon's real spawn path hands a subc-wire child its launch nonce on
10926/// descriptor 3, without an environment copy. The shell records the nonce
10927/// and its environment after exec so these tests observe the real handover.
10928#[cfg(all(test, unix))]
10929mod launch_nonce_descriptor_tests {
10930    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
10931    use crate::stderr_tail::{StderrRing, StderrTailConfig};
10932    use std::{
10933        path::PathBuf,
10934        sync::{Arc, Mutex},
10935        time::{Duration, Instant},
10936    };
10937    use subc_test_support::TestTempDir;
10938
10939    async fn probe(role: super::SpawnRole) {
10940        let scratch = TestTempDir::new("launch-nonce-descriptor");
10941        let fd_copy = scratch.join("from-descriptor");
10942        let env_copy = scratch.join("environment");
10943        let script = format!(
10944            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
10945            fd = fd_copy.display(), env = env_copy.display(),
10946        );
10947        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
10948        let spec = ModuleSpec {
10949            module_id: "nonce-descriptor-probe".to_string(),
10950            program: PathBuf::from("/bin/sh"),
10951            args: vec!["-c".to_string(), script],
10952            env: vec![
10953                xdg("XDG_DATA_HOME"),
10954                xdg("XDG_RUNTIME_DIR"),
10955                xdg("XDG_CONFIG_HOME"),
10956                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
10957            ],
10958            reserved: true,
10959            reserved_prefixes: Vec::new(),
10960            protocol: ModuleProtocol::Subc,
10961            overlap: Default::default(),
10962        };
10963        let handle = SupervisorHandle::new();
10964        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10965        let roster = ChildRoster::default();
10966        let child = super::spawn_child_in_slot(
10967            &spec,
10968            None,
10969            Some(&handle),
10970            &ring,
10971            None,
10972            &roster,
10973            #[cfg(target_os = "linux")]
10974            None,
10975            role,
10976            matches!(role, super::SpawnRole::SwapCandidate),
10977        )
10978        .expect("spawn probe");
10979        let deadline = Instant::now() + Duration::from_secs(10);
10980        while !(fd_copy.exists() && env_copy.exists()) {
10981            assert!(Instant::now() < deadline, "probe never wrote its copies");
10982            tokio::time::sleep(Duration::from_millis(20)).await;
10983        }
10984        let nonce = std::fs::read_to_string(fd_copy).unwrap();
10985        assert!(!nonce.is_empty());
10986        let environment = std::fs::read_to_string(env_copy).unwrap();
10987        assert!(environment
10988            .lines()
10989            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
10990        let copy = environment
10991            .lines()
10992            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
10993        assert_eq!(
10994            copy, None,
10995            "Unix children must never receive the environment nonce"
10996        );
10997        if matches!(role, super::SpawnRole::Plain) {
10998            assert_eq!(
10999                handle.spawn_nonce(&spec.module_id).as_deref(),
11000                Some(nonce.as_str())
11001            );
11002        }
11003        drop(child);
11004    }
11005
11006    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
11007    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
11008        probe(super::SpawnRole::Plain).await;
11009    }
11010
11011    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
11012    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
11013        probe(super::SpawnRole::SwapCandidate).await;
11014    }
11015}
11016
11017#[cfg(all(test, target_os = "linux"))]
11018mod cgroup_containment_tests {
11019    use super::*;
11020    use subc_test_support::TestTempDir;
11021
11022    fn running(pid: u32) -> bool {
11023        // An orphan can remain a zombie until the container init reaps it.
11024        std::fs::read_to_string(format!("/proc/{pid}/stat"))
11025            .ok()
11026            .and_then(|stat| {
11027                stat.rsplit_once(") ")
11028                    .map(|(_, rest)| rest.starts_with('Z'))
11029            })
11030            .is_some_and(|zombie| !zombie)
11031    }
11032
11033    #[tokio::test]
11034    async fn linux_teardown_reaps_the_grandchild() {
11035        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
11036    }
11037
11038    #[tokio::test]
11039    async fn linux_shutdown_straggler_reaps_the_grandchild() {
11040        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
11041    }
11042
11043    async fn teardown_tree(test_name: &str, shutdown: bool) {
11044        let dir = TestTempDir::new(test_name);
11045        let root = PathBuf::from(format!(
11046            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
11047            std::process::id(),
11048            unix_ms_now()
11049        ));
11050        if let Err(error) = std::fs::create_dir(&root) {
11051            assert!(
11052                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11053                "required cgroup test cannot execute: {error}"
11054            );
11055            eprintln!(
11056                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
11057                root.display()
11058            );
11059            return;
11060        }
11061        let placement = subc_cgroup::prepare_at(&root)
11062            .expect("prepare isolated kernel cgroup")
11063            .expect("isolated cgroup is delegated");
11064        let module_id = "tree-teardown";
11065        let module = placement
11066            .module_path(module_id)
11067            .expect("create isolated module cgroup");
11068        if !module.join("cgroup.kill").exists() {
11069            std::fs::remove_dir(&module).unwrap();
11070            std::fs::remove_dir(root.join("subc-modules")).unwrap();
11071            std::fs::remove_dir(&root).unwrap();
11072            assert!(
11073                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11074                "required cgroup.kill interface unavailable"
11075            );
11076            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
11077            return;
11078        }
11079        let supervisor = Supervisor::new(
11080            Arc::new(Registry::default()),
11081            RestartPolicy::new(3, Duration::ZERO),
11082        )
11083        .with_cgroup_placement(Some(placement));
11084        let mut runtime = supervisor.runtime_config();
11085        runtime.child_roster = runtime
11086            .child_roster
11087            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
11088        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11089        let pid_file = dir.join("grandchild.pid");
11090        let spec = ModuleSpec {
11091            module_id: module_id.to_string(),
11092            program: PathBuf::from("/bin/sh"),
11093            args: vec![
11094                "-c".into(),
11095                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
11096                "fixture".into(),
11097                pid_file.display().to_string(),
11098            ],
11099            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11100                .into_iter()
11101                .map(|key| (key.to_string(), dir.display().to_string()))
11102                .collect(),
11103            reserved: false,
11104            reserved_prefixes: Vec::new(),
11105            protocol: ModuleProtocol::None,
11106            overlap: Default::default(),
11107        };
11108        let child =
11109            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
11110        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
11111        let grandchild: u32 = loop {
11112            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
11113                if let Ok(pid) = contents.trim().parse() {
11114                    break pid;
11115                }
11116            }
11117            assert!(
11118                tokio::time::Instant::now() < deadline,
11119                "grandchild pid was not recorded"
11120            );
11121            tokio::time::sleep(Duration::from_millis(10)).await;
11122        };
11123        assert!(
11124            running(grandchild),
11125            "grandchild must be alive before teardown"
11126        );
11127        if shutdown {
11128            let mut child = child;
11129            crate::child_roster::end_children_for_daemon_shutdown(
11130                &runtime.child_roster,
11131                false,
11132                std::future::pending(),
11133            )
11134            .await;
11135            child.wait().await.expect("reap shutdown straggler");
11136        } else {
11137            drain_child_to_state(
11138                module_id,
11139                ModuleProtocol::None,
11140                StopNotice::NotSent,
11141                &Registry::default(),
11142                None,
11143                &snapshot,
11144                &runtime.terminal_ring,
11145                &SpawnEventFeed::default(),
11146                child,
11147                Duration::from_millis(100),
11148                ModuleState::Stopped,
11149                Some(false),
11150            )
11151            .await
11152            .expect("real supervisor teardown");
11153        }
11154        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
11155        while running(grandchild) && tokio::time::Instant::now() < deadline {
11156            tokio::time::sleep(Duration::from_millis(10)).await;
11157        }
11158        let survived = running(grandchild);
11159        // Kill a surviving grandchild so a failed test does not leave it behind.
11160        if survived {
11161            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
11162            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
11163            tokio::time::sleep(Duration::from_millis(100)).await;
11164        }
11165        if module.exists() {
11166            std::fs::remove_dir(&module).expect("remove empty module cgroup");
11167        }
11168        std::fs::remove_dir(root.join("subc-modules")).unwrap();
11169        std::fs::remove_dir(&root).unwrap();
11170        assert!(
11171            !survived,
11172            "grandchild {grandchild} outlived module teardown"
11173        );
11174        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
11175    }
11176}