Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113struct SupervisedChild {
114    child: Child,
115    /// The name of this process's cgroup: the module id, or for a swap
116    /// candidate the alternate name (see `swap::cgroup_name`).
117    #[cfg(target_os = "linux")]
118    module_id: String,
119    #[cfg(target_os = "linux")]
120    cgroup_placement: Option<subc_cgroup::Placement>,
121    /// The job that contains this child and every process it spawns (issue #109).
122    ///
123    /// Dropping this handle is what reaps a surviving tree when no supervisor
124    /// code runs — a daemon crash — because the job carries
125    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
126    ///
127    /// That limit is not crash-only, and the difference is worth knowing: a
128    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
129    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
130    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
131    /// module at once. Before this change they survived that, saw EOF on the
132    /// control socket, and ran their own teardown; Unix keeps that path
133    /// deliberately, so a module can seal a WAL or close a capture rather than
134    /// be killed mid-write. So this trades graceful teardown on every Windows
135    /// daemon stop for containment on a crash, which is the right way round
136    /// today: orphaned GPU workers are a reported, recurring problem, and the
137    /// modules that write most heavily do not run on Windows.
138    ///
139    /// The fix is a real Windows stop path — the daemon draining before it
140    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
141    /// reaches only what the drain left behind, which is what it should reach.
142    #[cfg(windows)]
143    job: Option<subc_jobobject::JobObject>,
144    stdout_pump: Option<JoinHandle<()>>,
145    stderr_pump: Option<StderrPump>,
146    stderr_ring: Arc<Mutex<StderrRing>>,
147    spawned_at_ms: u64,
148    spawned_from: PathBuf,
149    spawned_file_identity: Option<SpawnedFileIdentity>,
150    process_start_time: Option<u64>,
151    process_identity: Option<ProcessIdentity>,
152    pid: u32,
153    /// This process's entry in the daemon's child roster, released when the
154    /// process is reaped or this handle is dropped.
155    roster_guard: Option<crate::child_roster::RosterGuard>,
156}
157
158impl SupervisedChild {
159    fn id(&self) -> Option<u32> {
160        Some(self.pid)
161    }
162
163    fn process_identity(&self) -> Option<ProcessIdentity> {
164        self.process_identity
165    }
166
167    async fn wait(&mut self) -> io::Result<ExitStatus> {
168        // The roster entry is NOT released here. A daemon shutdown waits for the
169        // roster to empty and then exits the process, so releasing at the reap
170        // let it exit before the exit handler wrote this child's terminal record
171        // (the stderr drain and snapshot update sit in between), and the
172        // shutdown's own `daemon_shutdown` record was intermittently lost. The
173        // caller releases it after recording the exit (`release_roster`), and
174        // dropping the handle releases it too.
175        let result = self.child.wait().await;
176        #[cfg(target_os = "linux")]
177        if result.is_ok() {
178            if let Some(placement) = self.cgroup_placement.take() {
179                remove_module_cgroup(&placement, &self.module_id);
180            }
181        }
182        result
183    }
184
185    /// Releases this child's daemon-shutdown roster entry once its exit has
186    /// been recorded. The pid is already reaped and free for reuse, so the
187    /// entry must not outlive the record any longer than that.
188    fn release_roster(&mut self) {
189        self.roster_guard = None;
190    }
191
192    /// Kill the child and, where containment is available, its process tree.
193    ///
194    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
195    /// helper process leaked the helper — the Synapse embedding module's CUDA
196    /// worker holds the GPU allocation, so the leak cost VRAM until the next
197    /// restart of something else. Terminating the job reaches grandchildren that
198    /// a tree walk cannot, including one whose parent has already exited and
199    /// been reparented away.
200    ///
201    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
202    /// direct-child kill still decides the outcome, so containment can never
203    /// change whether a module is reported as stopped.
204    fn start_kill(&mut self) -> io::Result<()> {
205        #[cfg(windows)]
206        if let Some(job) = &self.job {
207            if let Err(error) = job.terminate() {
208                debug!(
209                    error = %error,
210                    "job termination failed; the direct-child kill still owns the outcome"
211                );
212            }
213        }
214        #[cfg(target_os = "linux")]
215        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
216        self.child.start_kill()
217    }
218
219    async fn drain_stderr(&mut self, module_id: &str) {
220        if let Some(mut pump) = self.stdout_pump.take() {
221            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
222                Ok(Ok(())) => {}
223                Ok(Err(error)) => {
224                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
225                }
226                Err(_) => {
227                    pump.abort();
228                    warn!(
229                        module_id,
230                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
231                        "stdout pump did not drain before restart; stopped it before the next process"
232                    );
233                }
234            }
235        }
236
237        let Some(pump) = self.stderr_pump.take() else {
238            return;
239        };
240        settle_stderr_pump(
241            module_id,
242            &self.stderr_ring,
243            pump,
244            STDERR_PUMP_DRAIN_TIMEOUT,
245        )
246        .await;
247    }
248}
249
250/// The reader task for one process's stderr, with the ring generation its
251/// lines are attributed to.
252struct StderrPump {
253    task: JoinHandle<()>,
254    generation: u64,
255}
256
257/// Retire an exited process's stderr reader and wait up to `bound` for it to
258/// reach EOF. A reader still running at the bound is detached, not stopped: it
259/// keeps filling the exited process's section of the ring until its pipe
260/// closes, and the tail reads `Incomplete` until then. See
261/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
262async fn settle_stderr_pump(
263    module_id: &str,
264    ring: &Arc<Mutex<StderrRing>>,
265    pump: StderrPump,
266    bound: Duration,
267) {
268    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
269    let StderrPump {
270        mut task,
271        generation,
272    } = pump;
273    lock().retire_pump(generation);
274    match timeout(bound, &mut task).await {
275        Ok(Ok(())) => {}
276        Ok(Err(err)) => {
277            let mut ring = lock();
278            ring.mark_incomplete(format!("stderr pump ended unexpectedly: {err}"));
279            ring.finish_pump(generation);
280            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
281        }
282        Err(_) => {
283            // Dropping the handle detaches the task; it ends at EOF on its pipe.
284            drop(task);
285            lock().mark_pump_late(
286                generation,
287                format!(
288                    "stderr of the exited process had not reached EOF {bound:?} after it was \
289                     retired (a descendant may still hold the pipe open); lines it still \
290                     writes are kept in that process's section"
291                ),
292            );
293            warn!(
294                module_id,
295                waited = ?bound,
296                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
297            );
298        }
299    }
300}
301
302fn registration_release_events() -> &'static watch::Sender<u64> {
303    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
304    EVENTS.get_or_init(|| {
305        let (sender, _receiver) = watch::channel(0);
306        sender
307    })
308}
309
310pub(crate) fn notify_registration_release() {
311    let events = registration_release_events();
312    let next_generation = (*events.borrow()).wrapping_add(1);
313    events.send_replace(next_generation);
314}
315
316/// How to launch one singleton module process.
317#[derive(Debug, Clone, PartialEq, Eq)]
318pub struct ModuleSpec {
319    pub module_id: String,
320    pub program: PathBuf,
321    pub args: Vec<String>,
322    pub env: Vec<(String, String)>,
323    /// When true this is a reserved module: each spawn gets a fresh one-time launch
324    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
325    /// process can register this module_id (a security-boundary module like the
326    /// credential vault must not be impersonable while it is down/restarting).
327    pub reserved: bool,
328    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
329    /// Prefixes come from daemon config and must end in `:` before they reach the
330    /// supervisor; the owner module's current spawn nonce authorizes claims under
331    /// each prefix.
332    pub reserved_prefixes: Vec<String>,
333    /// The wire protocol this module speaks, as DECLARED in daemon config.
334    ///
335    /// [`ModuleProtocol::None`] changes five things and nothing else: health
336    /// probing is suppressed, teardown sends SIGTERM before waiting,
337    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
338    /// and NO launch nonce, and a clean exit the daemon did not request is
339    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
340    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
341    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
342    /// because a process ignores an environment variable it does not read.
343    ///
344    /// The argument is the part that cannot be "harmless to a process that
345    /// ignores it": a stock binary exits on an unknown flag before it listens
346    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
347    /// first conformance run against this mode found it. The nonce is withheld
348    /// because a process that will never present it gains nothing from holding
349    /// it, and a secret in the environment of a process that does not need it is
350    /// a leak surface for no benefit.
351    pub protocol: ModuleProtocol,
352    /// Whether two processes of this module may run at once, which is what a
353    /// blue/green swap does for the length of its overlap. Declared in daemon
354    /// config because the daemon must be able to answer it while the module is
355    /// down, and so a module cannot talk itself into it after registering.
356    pub overlap: ModuleOverlap,
357}
358
359/// Whether a module tolerates a second process of itself running alongside.
360///
361/// Most modules are single-writer on their store (a WAL, a capture log, a
362/// resident index behind a writer barrier), and two processes on one store
363/// corrupt it. So a swap, which overlaps the old and new process by design,
364/// is refused unless the module's config opts in with `overlap: "safe"`.
365#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
366pub enum ModuleOverlap {
367    /// Never run two processes of this module at once. The default.
368    #[default]
369    Exclusive,
370    /// The module has said a second process of itself is harmless for the
371    /// length of a swap.
372    ///
373    /// Declare it only if a second instance can run for a few seconds without
374    /// touching ANY single-writer store: every database, WAL, index, projector
375    /// and scheduled job the module owns. A lease on part of that state is not
376    /// enough. broca's session lease guards WAL appends while its run index, its
377    /// store projector and its archive fold timer (which unlinks live WAL files)
378    /// stay single-writer, so broca is exclusive despite holding a lease. The
379    /// refusal only fires after this has been decided, so the decision is the
380    /// check.
381    Safe,
382}
383
384impl ModuleOverlap {
385    pub fn as_str(self) -> &'static str {
386        match self {
387            Self::Exclusive => "exclusive",
388            Self::Safe => "safe",
389        }
390    }
391}
392
393/// Environment variable telling a spawned module which case it was started
394/// for, before it sends HELLO. Only a swap candidate carries it, as
395/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
396///
397/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
398/// longer because nobody waits on it, while a plain restart must flip ready
399/// quickly because callers see `module_warming` until it does. Absence means
400/// plain restart, the safe reading. The daemon trusts nothing about it; the
401/// candidate is proven by its launch nonce at HELLO.
402pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
403/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
404pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
405/// How long a swap waits for its candidate to register and declare itself
406/// ready when the operator does not say. A module warming as a swap candidate
407/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
408/// daemon allows that plus time to start the process and send HELLO.
409pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
410
411/// Bounded restart policy for crash exits.
412///
413/// `max_restarts` is the number of replacement processes allowed after the
414/// initial spawn WITHIN `window`. After that many crash restarts inside one
415/// window the module enters [`ModuleState::Failed`] and the supervisor stops
416/// the crash loop.
417///
418/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
419/// and that only survived because crashes were rare: a module that crashed
420/// three times across a week was disabled forever by crashes that had nothing
421/// to do with each other. That stopped being survivable once modules began
422/// exiting non-zero whenever the daemon's connection to them drops, because
423/// then every daemon-side connection drop spends a unit of the same budget and
424/// one flappy hour permanently stops a healthy module. Restarts older than
425/// `window` release their slot, so a module that crashed twice yesterday has a
426/// full budget today, while a genuine crash loop -- which is fast by
427/// definition -- still reaches the cap and stops.
428#[derive(Debug, Clone, Copy, PartialEq, Eq)]
429pub struct RestartPolicy {
430    pub max_restarts: u32,
431    /// Base delay before a crash replacement. The actual delay escalates with
432    /// the number of recent crash replacements and is capped by `max_backoff`.
433    pub backoff: Duration,
434    /// Maximum delay before a crash replacement.
435    pub max_backoff: Duration,
436    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
437    /// budget effectively infinite (nothing is ever in-window), which is why
438    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
439    pub window: Duration,
440}
441
442impl RestartPolicy {
443    /// A policy with the default crash window. Callers that care about the
444    /// window say so with [`Self::with_window`]; the ones that do not are
445    /// asking for the standard rate limit, not for no limit.
446    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
447        Self {
448            max_restarts,
449            backoff,
450            max_backoff: DEFAULT_MAX_BACKOFF,
451            window: DEFAULT_RESTART_WINDOW,
452        }
453    }
454
455    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
456        self.max_backoff = max_backoff;
457        self
458    }
459
460    pub fn with_window(mut self, window: Duration) -> Self {
461        self.window = window;
462        self
463    }
464
465    /// Calculate the capped exponential delay for the next crash replacement.
466    /// `restart_in_window` is zero for the first replacement after an operator
467    /// action (restart, reload, re-enable) cleared the crash ring, or after all
468    /// older crash replacements have aged out of the window.
469    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
470        if self.backoff.is_zero() || self.max_backoff.is_zero() {
471            return Duration::ZERO;
472        }
473
474        let mut delay = self.backoff;
475        for _ in 0..restart_in_window {
476            if delay >= self.max_backoff {
477                return self.max_backoff;
478            }
479            delay = delay
480                .checked_mul(10)
481                .unwrap_or(self.max_backoff)
482                .min(self.max_backoff);
483        }
484        delay.min(self.max_backoff)
485    }
486
487    /// The one sentence that explains a budget-exhausted stop, used for both the
488    /// log line and the terminal record so the two cannot drift. It names the
489    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
490    /// exactly what this budget is not.
491    fn budget_exhausted_detail(&self) -> String {
492        format!(
493            "crash budget exhausted: max_restarts={} within window_secs={}",
494            self.max_restarts,
495            self.window.as_secs()
496        )
497    }
498}
499
500impl Default for RestartPolicy {
501    fn default() -> Self {
502        Self {
503            max_restarts: DEFAULT_MAX_RESTARTS,
504            backoff: DEFAULT_BACKOFF,
505            max_backoff: DEFAULT_MAX_BACKOFF,
506            window: DEFAULT_RESTART_WINDOW,
507        }
508    }
509}
510
511#[derive(Debug, Clone, Copy, PartialEq, Eq)]
512struct CrashRestartSchedule {
513    restart_in_window: u32,
514    delay: Duration,
515}
516
517/// Whether the daemon itself will bring this module back after the exit being
518/// handled: it is enabled AND its in-window crash restarts are below the cap.
519///
520/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
521/// the window are dropped here rather than by a timer, so the count is right
522/// the moment somebody asks and no bookkeeping runs for idle modules.
523fn daemon_will_restart(
524    state: &mut SupervisorSnapshot,
525    policy: &RestartPolicy,
526    now: Instant,
527) -> bool {
528    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
529}
530
531const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
532const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
533const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
534const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
535
536#[derive(Debug, Clone, Copy, PartialEq, Eq)]
537pub enum HealthAction {
538    Report,
539    Restart,
540    Alert,
541}
542
543impl fmt::Display for HealthAction {
544    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
545        f.write_str(match self {
546            Self::Report => "report",
547            Self::Restart => "restart",
548            Self::Alert => "alert",
549        })
550    }
551}
552
553#[derive(Debug, Clone, Copy, PartialEq, Eq)]
554pub struct HealthConfig {
555    pub cadence: Duration,
556    pub deadline: Duration,
557    pub failure_threshold: u32,
558    pub on_degraded: HealthAction,
559    pub on_failing: HealthAction,
560    pub critical: bool,
561}
562
563impl Default for HealthConfig {
564    fn default() -> Self {
565        Self {
566            cadence: DEFAULT_HEALTH_CADENCE,
567            deadline: DEFAULT_HEALTH_DEADLINE,
568            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
569            on_degraded: HealthAction::Report,
570            on_failing: HealthAction::Report,
571            critical: false,
572        }
573    }
574}
575
576/// The supervisor's view of one module's health, relayed to clients over
577/// channel-0 and rendered by `ck health`.
578///
579/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
580/// stated here rather than only at the wire type a consumer reads. A reader can
581/// look up what `None` means; only a writer can silently change it, and the
582/// writer has no reason to go looking at a downstream contract before editing.
583///
584/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
585/// back to `None` on re-registration precisely so a respawned module does not
586/// carry its predecessor's timestamp — so an old value and an absent one call for
587/// opposite readings, and anything that defaulted this to a number would make a
588/// never-probed module indistinguishable from one probed at the epoch.
589///
590/// `detail` and `metrics` are `None` when the module published none on this
591/// probe, which does not mean it reported nothing wrong — it is also the shape
592/// when the probe never reached it. `last_probe_ms` is what separates those.
593#[derive(Debug, Clone, PartialEq)]
594pub struct ModuleHealthStatus {
595    pub status: SupervisorHealthStatus,
596    pub last_probe_ms: Option<u64>,
597    pub detail: Option<String>,
598    pub metrics: Option<Value>,
599    pub consecutive_failures: u32,
600    /// Number of replies received after a recurring health probe's deadline.
601    /// Unlike a timeout, every increment proves the module was alive.
602    pub late_answer_count: u64,
603    /// End-to-end latency of the newest late reply, measured from probe start.
604    pub last_late_answer_latency_ms: Option<u64>,
605    pub last_action: Option<String>,
606    /// Set together with `last_action`; the pair moves as one, and both being
607    /// absent means no escalation has ever been taken rather than that the last
608    /// one succeeded.
609    pub last_action_ms: Option<u64>,
610}
611
612impl Default for ModuleHealthStatus {
613    fn default() -> Self {
614        Self {
615            status: SupervisorHealthStatus::Unknown,
616            last_probe_ms: None,
617            detail: None,
618            metrics: None,
619            consecutive_failures: 0,
620            late_answer_count: 0,
621            last_late_answer_latency_ms: None,
622            last_action: None,
623            last_action_ms: None,
624        }
625    }
626}
627
628/// Typed lifecycle state for a supervised module.
629#[derive(Debug, Clone, Copy, PartialEq, Eq)]
630pub enum ModuleState {
631    Starting,
632    Running,
633    Unresponsive,
634    Restarting,
635    Draining,
636    Stopped,
637    Failed,
638    Disabled,
639}
640
641impl fmt::Display for ModuleState {
642    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
643        f.write_str(match self {
644            Self::Starting => "starting",
645            Self::Running => "running",
646            Self::Unresponsive => "unresponsive",
647            Self::Restarting => "restarting",
648            Self::Draining => "draining",
649            Self::Stopped => "stopped",
650            Self::Failed => "failed",
651            Self::Disabled => "disabled",
652        })
653    }
654}
655
656/// Supervisor classification of a child-process exit.
657#[derive(Debug, Clone, Copy, PartialEq, Eq)]
658pub enum ExitKind {
659    Clean,
660    Crash,
661    DeliberateSeverance,
662}
663
664impl From<ExitKind> for TerminalExitKind {
665    fn from(kind: ExitKind) -> Self {
666        match kind {
667            ExitKind::Clean => Self::Clean,
668            ExitKind::Crash => Self::Crash,
669            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
670        }
671    }
672}
673
674/// Exact process identity retained when a supervised module registers its
675/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
676#[derive(Debug, Clone, Copy, PartialEq, Eq)]
677pub(crate) struct ProcessIdentity {
678    pub(crate) pid: u32,
679    pub(crate) start_time: u64,
680}
681
682/// Last observed child exit, if any.
683#[derive(Debug, Clone, PartialEq, Eq)]
684pub struct ExitReport {
685    pub kind: ExitKind,
686    pub code: Option<i32>,
687    pub signal: Option<i32>,
688    pub at_ms: u64,
689}
690
691/// Point-in-time module status answerable by subc without forwarding to the
692/// module process.
693#[derive(Debug, Clone, PartialEq)]
694pub struct ModuleStatus {
695    pub module_id: String,
696    pub state: ModuleState,
697    pub enabled: bool,
698    pub process_alive: bool,
699    pub registration_active: bool,
700    /// The module's declared wire protocol, carried beside `live` because it is
701    /// what makes `live` readable: the two fields answer one question together.
702    pub protocol: ModuleProtocol,
703    /// Whether the module is serving, under the strongest definition the daemon
704    /// can assert for its protocol.
705    ///
706    /// A subc module must also be REGISTERED: its process being alive says
707    /// nothing about whether it can take a request. A `protocol: "none"` module
708    /// never registers, so that term is dropped and this falls back to "enabled,
709    /// running, and the process the daemon launched is alive" -- which is all
710    /// the daemon observes about a process that speaks no subc wire. It stays a
711    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
712    /// rather than printing it bare.
713    pub live: bool,
714    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
715    /// restarts have already released their slot, so this count can go down
716    /// without anybody touching the module.
717    pub restart_count: u32,
718    /// Replacement processes spawned over this module's entire supervisor lifetime;
719    /// unlike `restart_count`, this value is never reset by an operator action
720    /// and never falls out of a window.
721    pub lifetime_restarts: u32,
722    pub spawn_generation: u64,
723    /// The budget `restart_count` is spent against. Carried alongside the count
724    /// because the count alone does not say how close the module is to being
725    /// disabled, and reporting one without the other is what makes an
726    /// about-to-be-retired module look ordinary.
727    pub max_restarts: u32,
728    /// The span `restart_count` is counted over. Carried with the pair above for
729    /// the same reason they are carried together: "2 of 3" means one thing for a
730    /// ten-minute window and something else entirely for a lifetime.
731    pub restart_window: Duration,
732    /// Effective drain and restart timing policy used by this running module.
733    /// These values are carried together with the restart budget so status
734    /// readers can compare configured intent with what the supervisor applied.
735    pub drain_timeout: Duration,
736    pub restart_backoff: Duration,
737    pub restart_max_backoff: Duration,
738    pub pid: Option<u32>,
739    pub spawned_at_ms: Option<u64>,
740    pub spawned_from: Option<PathBuf>,
741    pub process_start_time: Option<u64>,
742    pub last_exit: Option<ExitReport>,
743    pub health: ModuleHealthStatus,
744}
745
746#[derive(Debug, Clone, PartialEq)]
747struct SupervisorSnapshot {
748    state: ModuleState,
749    enabled: bool,
750    process_alive: bool,
751    /// When each crash restart was spent, oldest first. This IS the crash
752    /// budget: its in-window length is the count an operator sees and the count
753    /// the restart decision is made against, so there is no second counter that
754    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
755    /// operator actions that used to zero the old lifetime counter.
756    crash_restarts: VecDeque<Instant>,
757    lifetime_restarts: u32,
758    /// Successful child spawns in this daemon incarnation.
759    ///
760    /// `lifetime_restarts` was considered and rejected: it starts at zero
761    /// (line 640), successful initial/operator spawns in `set_running` do not
762    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
763    /// increments before a successful replacement exists (lines 604, 3846,
764    /// and 3921), so a failed spawn can consume it. This counter moves only
765    /// when a live PID is accepted below.
766    spawn_generation: u64,
767    pid: Option<u32>,
768    spawned_at_ms: Option<u64>,
769    spawned_from: Option<PathBuf>,
770    spawned_file_identity: Option<SpawnedFileIdentity>,
771    process_start_time: Option<u64>,
772    deliberate_severance: Option<ProcessIdentity>,
773    last_exit: Option<ExitReport>,
774    health: ModuleHealthStatus,
775    /// Whether the current process was started as a swap candidate and so
776    /// lives in the module's alternate cgroup. The next swap's candidate takes
777    /// the other one, so the two processes of a swap never share a cgroup. A
778    /// plain spawn always uses the primary cgroup.
779    in_alternate_slot: bool,
780    /// Whether the current `Draining` state ends in a replacement process
781    /// (restart, reload, health restart) rather than a stop. Only meaningful
782    /// while `state` is `Draining`; every entry into that state rewrites it.
783    /// It is what lets route.open answer the retryable `module_reloading` to a
784    /// consumer that reaches a still-registered process mid-restart, instead of
785    /// the `supervisor_not_live` a stop or disable deserves.
786    draining_to_replace: bool,
787    /// Whether a configuration update has been applied since the current
788    /// process was spawned, so that process runs an older spec than the one
789    /// the supervisor now holds. A queued restart is only coalesced into a
790    /// fresher process when this is false: a restart requested to pick up a
791    /// new configuration must not be satisfied by a process that predates it.
792    configuration_updated_since_spawn: bool,
793}
794
795impl SupervisorSnapshot {
796    fn starting() -> Self {
797        Self::new(ModuleState::Starting, true)
798    }
799
800    fn disabled() -> Self {
801        Self::new(ModuleState::Disabled, false)
802    }
803
804    fn failed() -> Self {
805        Self::new(ModuleState::Failed, true)
806    }
807
808    /// Crash restarts still inside `window`, having dropped the ones that are
809    /// not. Pruning on read is what makes the budget a rate: an instant older
810    /// than the window stops holding a slot the moment anybody counts.
811    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
812        while let Some(oldest) = self.crash_restarts.front() {
813            if now.duration_since(*oldest) > window {
814                self.crash_restarts.pop_front();
815            } else {
816                break;
817            }
818        }
819        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
820    }
821
822    /// Spend one unit of the crash budget and record the restart in the ledger.
823    ///
824    /// The ring is bounded by the cap because more than `max_restarts` in-window
825    /// instants can never be reached (the caller refuses the restart first), so
826    /// anything beyond that is an unbounded queue waiting to happen.
827    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
828        self.crash_restarts.push_back(now);
829        while self.crash_restarts.len() > policy.max_restarts as usize {
830            self.crash_restarts.pop_front();
831        }
832        self.lifetime_restarts += 1;
833    }
834
835    /// Reserve one crash-restart slot and calculate the delay before respawning.
836    /// The count is captured before recording this restart, so the first retry
837    /// uses the base delay and each later in-window retry escalates once.
838    fn next_crash_restart(
839        &mut self,
840        policy: &RestartPolicy,
841        now: Instant,
842    ) -> Option<CrashRestartSchedule> {
843        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
844        if restart_in_window >= policy.max_restarts {
845            return None;
846        }
847        self.record_crash_restart(policy, now);
848        Some(CrashRestartSchedule {
849            restart_in_window,
850            delay: policy.delay_for_restart(restart_in_window),
851        })
852    }
853
854    /// Give the module its full budget back, as an operator restart, reload, or
855    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
856    /// ledger of what actually happened, and an operator action does not unmake
857    /// the crashes.
858    fn clear_crash_restarts(&mut self) {
859        self.crash_restarts.clear();
860    }
861
862    fn new(state: ModuleState, enabled: bool) -> Self {
863        Self {
864            state,
865            enabled,
866            process_alive: false,
867            crash_restarts: VecDeque::new(),
868            lifetime_restarts: 0,
869            spawn_generation: 0,
870            pid: None,
871            spawned_at_ms: None,
872            spawned_from: None,
873            spawned_file_identity: None,
874            process_start_time: None,
875            deliberate_severance: None,
876            last_exit: None,
877            health: ModuleHealthStatus::default(),
878            in_alternate_slot: false,
879            draining_to_replace: false,
880            configuration_updated_since_spawn: false,
881        }
882    }
883}
884
885type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
886
887type SpawnSubscriberKey = (ConnectionId, u64);
888
889#[derive(Debug)]
890struct SpawnSubscriber {
891    version: u8,
892    frames: mpsc::Sender<Frame>,
893    /// Tells this subscriber's forwarder that it was dropped for lagging, and
894    /// from which event. The full frame channel cannot carry that news, so it
895    /// travels beside it; see `SpawnEventFeed::subscribe`.
896    lagged: Option<oneshot::Sender<SpawnCursor>>,
897}
898
899#[derive(Debug)]
900struct SpawnEventState {
901    daemon_incarnation: String,
902    seq: u64,
903    capacity: usize,
904    live: HashMap<String, LiveSpawn>,
905    generations: HashMap<String, u64>,
906    events: VecDeque<SpawnEvent>,
907    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
908}
909
910impl Default for SpawnEventState {
911    fn default() -> Self {
912        Self {
913            daemon_incarnation: "unconfigured".to_string(),
914            seq: 0,
915            capacity: SPAWN_EVENT_RING_CAPACITY,
916            live: HashMap::new(),
917            generations: HashMap::new(),
918            events: VecDeque::new(),
919            subscribers: HashMap::new(),
920        }
921    }
922}
923
924#[derive(Debug, Clone, Default)]
925struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
926
927#[derive(Debug, Clone, PartialEq, Eq)]
928pub(crate) enum SpawnSubscribeRefusal {
929    ForeignIncarnation { current: String },
930    TooOld { oldest: SpawnCursor },
931    Frame(String),
932}
933
934impl SpawnEventFeed {
935    fn configure_incarnation(&self, daemon_incarnation: String) {
936        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
937        state.daemon_incarnation = daemon_incarnation;
938        state.seq = 0;
939        state.live.clear();
940        state.generations.clear();
941        state.events.clear();
942        state.subscribers.clear();
943    }
944
945    fn cursor(state: &SpawnEventState) -> SpawnCursor {
946        SpawnCursor {
947            daemon_incarnation: state.daemon_incarnation.clone(),
948            seq: state.seq,
949        }
950    }
951
952    fn snapshot(&self) -> SpawnSnapshot {
953        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
954        let mut live = state.live.values().cloned().collect::<Vec<_>>();
955        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
956        SpawnSnapshot {
957            cursor: Self::cursor(&state),
958            ring_bound: state.capacity as u64,
959            live,
960        }
961    }
962
963    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
964        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
965        let generation = state
966            .generations
967            .get(module_id)
968            .copied()
969            .unwrap_or(0)
970            .checked_add(1)
971            .expect("spawn generation exhausted");
972        state.generations.insert(module_id.to_string(), generation);
973        let live = LiveSpawn {
974            module_id: module_id.to_string(),
975            spawn_generation: generation,
976            pid,
977            spawned_at_ms,
978        };
979        state.live.insert(module_id.to_string(), live);
980        Self::emit_locked(
981            &mut state,
982            SpawnEventKind::Spawned,
983            module_id.to_string(),
984            generation,
985            pid,
986            None,
987            None,
988        );
989        generation
990    }
991
992    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
993        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
994        let Some(live) = state.live.remove(module_id) else {
995            warn!(
996                module_id,
997                "terminal record had no live spawn event identity"
998            );
999            return;
1000        };
1001        Self::emit_locked(
1002            &mut state,
1003            SpawnEventKind::Exited,
1004            module_id.to_string(),
1005            live.spawn_generation,
1006            live.pid,
1007            exit_code,
1008            exit_signal,
1009        );
1010    }
1011
1012    /// Report the exit of a process that a swap has already replaced.
1013    ///
1014    /// `emit_exited` removes the module's live entry, which after a swap's
1015    /// cutover describes the promoted candidate, not the old process now
1016    /// exiting. This emits the old generation's exit and leaves the live entry
1017    /// alone unless it still names that generation.
1018    fn emit_superseded_exited(
1019        &self,
1020        module_id: &str,
1021        spawn_generation: u64,
1022        pid: u32,
1023        exit_code: Option<i32>,
1024        exit_signal: Option<i32>,
1025    ) {
1026        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1027        if state
1028            .live
1029            .get(module_id)
1030            .is_some_and(|live| live.spawn_generation == spawn_generation)
1031        {
1032            state.live.remove(module_id);
1033        }
1034        Self::emit_locked(
1035            &mut state,
1036            SpawnEventKind::Exited,
1037            module_id.to_string(),
1038            spawn_generation,
1039            pid,
1040            exit_code,
1041            exit_signal,
1042        );
1043    }
1044
1045    #[allow(clippy::too_many_arguments)]
1046    fn emit_locked(
1047        state: &mut SpawnEventState,
1048        kind: SpawnEventKind,
1049        module_id: String,
1050        spawn_generation: u64,
1051        pid: u32,
1052        exit_code: Option<i32>,
1053        exit_signal: Option<i32>,
1054    ) {
1055        state.seq = state
1056            .seq
1057            .checked_add(1)
1058            .expect("spawn event sequence exhausted");
1059        let event = SpawnEvent {
1060            cursor: Self::cursor(state),
1061            kind,
1062            module_id,
1063            spawn_generation,
1064            pid,
1065            exit_code,
1066            exit_signal,
1067        };
1068        state.events.push_back(event.clone());
1069        while state.events.len() > state.capacity {
1070            state.events.pop_front();
1071        }
1072        let body = match serde_json::to_vec(&event) {
1073            Ok(body) => body,
1074            Err(error) => {
1075                error!(%error, "failed to serialize supervisor spawn event");
1076                return;
1077            }
1078        };
1079        state.subscribers.retain(|(connection_id, corr), subscriber| {
1080            let frame = Frame::build_with_version(
1081                subscriber.version,
1082                FrameType::StreamData,
1083                control_flags(),
1084                0,
1085                0,
1086                *corr,
1087                body.clone(),
1088            );
1089            match frame {
1090                Ok(frame) => {
1091                    if subscriber.frames.try_send(frame).is_ok() {
1092                        true
1093                    } else {
1094                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1095                        if let Some(lagged) = subscriber.lagged.take() {
1096                            let _ = lagged.send(event.cursor.clone());
1097                        }
1098                        false
1099                    }
1100                }
1101                Err(error) => {
1102                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1103                    false
1104                }
1105            }
1106        });
1107    }
1108
1109    fn subscribe(
1110        &self,
1111        connection_id: ConnectionId,
1112        corr: u64,
1113        version: u8,
1114        since: Option<SpawnCursor>,
1115        sink: FrameSink,
1116    ) -> Result<(), SpawnSubscribeRefusal> {
1117        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1118        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1119        {
1120            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1121            let replay = if let Some(since) = since {
1122                if since.daemon_incarnation != state.daemon_incarnation {
1123                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1124                        current: state.daemon_incarnation.clone(),
1125                    });
1126                }
1127                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1128                    if since.seq < oldest.seq.saturating_sub(1) {
1129                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1130                    }
1131                }
1132                state
1133                    .events
1134                    .iter()
1135                    .filter(|event| event.cursor.seq > since.seq)
1136                    .cloned()
1137                    .collect::<Vec<_>>()
1138            } else {
1139                Vec::new()
1140            };
1141            for event in replay {
1142                let body = serde_json::to_vec(&event)
1143                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1144                let frame = Frame::build_with_version(
1145                    version,
1146                    FrameType::StreamData,
1147                    control_flags(),
1148                    0,
1149                    0,
1150                    corr,
1151                    body,
1152                )
1153                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1154                frames
1155                    .try_send(frame)
1156                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1157            }
1158            state.subscribers.insert(
1159                (connection_id, corr),
1160                SpawnSubscriber {
1161                    version,
1162                    frames,
1163                    lagged: Some(lagged),
1164                },
1165            );
1166        }
1167        // The lagged terminal is sent here, by the forwarder, rather than by
1168        // the emitter: at the moment of the drop the subscriber's own channel
1169        // is full, and writing to the connection sink directly from the emitter
1170        // would put the Error AHEAD of the events still queued in that channel
1171        // (and the emitter holds the feed lock, so it cannot await the sink).
1172        // Dropping the subscriber drops the only sender, so `recv` drains every
1173        // queued event and then returns `None`; only then is the Error sent, so
1174        // the client sees each event it can keep, then the reason it was cut.
1175        // Cancel and connection removal drop the oneshot unsent, so they end
1176        // the stream with no Error.
1177        tokio::spawn(async move {
1178            while let Some(frame) = receiver.recv().await {
1179                if sink.send(frame).await.is_err() {
1180                    return;
1181                }
1182            }
1183            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1184                return;
1185            };
1186            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1187                Ok(frame) => {
1188                    let _ = sink.send(frame).await;
1189                }
1190                Err(error) => {
1191                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1192                }
1193            }
1194        });
1195        Ok(())
1196    }
1197
1198    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1199        let Some(subscriber) = self
1200            .0
1201            .lock()
1202            .unwrap_or_else(|p| p.into_inner())
1203            .subscribers
1204            .remove(&(connection_id, corr))
1205        else {
1206            return false;
1207        };
1208        if let Ok(frame) = Frame::build_with_version(
1209            subscriber.version,
1210            FrameType::StreamEnd,
1211            control_flags(),
1212            0,
1213            0,
1214            corr,
1215            Vec::new(),
1216        ) {
1217            tokio::spawn(async move {
1218                let _ = subscriber.frames.send(frame).await;
1219            });
1220        }
1221        true
1222    }
1223
1224    fn remove_connection(&self, connection_id: ConnectionId) {
1225        self.0
1226            .lock()
1227            .unwrap_or_else(|p| p.into_inner())
1228            .subscribers
1229            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1230    }
1231
1232    #[cfg(any(test, feature = "test-support"))]
1233    fn set_capacity(&self, capacity: usize) {
1234        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1235    }
1236
1237    #[cfg(any(test, feature = "test-support"))]
1238    fn subscriber_count(&self) -> usize {
1239        self.0
1240            .lock()
1241            .unwrap_or_else(|p| p.into_inner())
1242            .subscribers
1243            .len()
1244    }
1245}
1246
1247/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1248/// The terminal Error a lagged spawn subscriber receives after its queued events.
1249fn spawn_subscriber_lagged_frame(
1250    version: u8,
1251    corr: u64,
1252    first_undelivered: SpawnCursor,
1253) -> Result<Frame, String> {
1254    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1255        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1256        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1257            .to_string(),
1258        detail: Some(serde_json::json!({
1259            "first_undelivered_cursor": first_undelivered
1260        })),
1261    })
1262    .map_err(|error| error.to_string())?;
1263    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1264        .map_err(|error| error.to_string())
1265}
1266
1267pub trait ModuleProcessLiveness: Send + Sync {
1268    fn process_live(&self, module_id: &str) -> Option<bool>;
1269
1270    /// Whether the supervisor is replacing this module's process right now: an
1271    /// operator restart or reload, a health restart, or a crash respawn whose
1272    /// backoff is running. A module in that state is not live, but a consumer
1273    /// refused now should retry shortly rather than treat the target as gone.
1274    /// Stopped, failed, and disabled modules are not replacing.
1275    fn process_replacing(&self, _module_id: &str) -> bool {
1276        false
1277    }
1278}
1279
1280/// Shared process-liveness registry keyed by supervised `module_id`.
1281#[derive(Debug, Clone, Default)]
1282pub struct SupervisorProcessLiveness {
1283    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1284}
1285
1286impl SupervisorProcessLiveness {
1287    pub fn new() -> Self {
1288        Self::default()
1289    }
1290
1291    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1292        let mut snapshots = self
1293            .snapshots
1294            .lock()
1295            .unwrap_or_else(|poisoned| poisoned.into_inner());
1296        snapshots.insert(module_id, snapshot);
1297    }
1298
1299    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1300        let mut snapshots = self
1301            .snapshots
1302            .lock()
1303            .unwrap_or_else(|poisoned| poisoned.into_inner());
1304        let is_current = snapshots
1305            .get(module_id)
1306            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1307            .unwrap_or(false);
1308        if is_current {
1309            snapshots.remove(module_id);
1310        }
1311    }
1312}
1313
1314impl ModuleProcessLiveness for SupervisorProcessLiveness {
1315    fn process_live(&self, module_id: &str) -> Option<bool> {
1316        let snapshot = {
1317            let snapshots = self
1318                .snapshots
1319                .lock()
1320                .unwrap_or_else(|poisoned| poisoned.into_inner());
1321            snapshots.get(module_id).cloned()
1322        }?;
1323        let snapshot = snapshot
1324            .lock()
1325            .unwrap_or_else(|poisoned| poisoned.into_inner());
1326        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1327    }
1328
1329    fn process_replacing(&self, module_id: &str) -> bool {
1330        let Some(snapshot) = self
1331            .snapshots
1332            .lock()
1333            .unwrap_or_else(|poisoned| poisoned.into_inner())
1334            .get(module_id)
1335            .cloned()
1336        else {
1337            return false;
1338        };
1339        let snapshot = snapshot
1340            .lock()
1341            .unwrap_or_else(|poisoned| poisoned.into_inner());
1342        snapshot.enabled
1343            && match snapshot.state {
1344                ModuleState::Restarting => true,
1345                ModuleState::Draining => snapshot.draining_to_replace,
1346                ModuleState::Starting
1347                | ModuleState::Running
1348                | ModuleState::Unresponsive
1349                | ModuleState::Stopped
1350                | ModuleState::Failed
1351                | ModuleState::Disabled => false,
1352            }
1353    }
1354}
1355
1356#[derive(Debug, Clone)]
1357struct SupervisorRuntimeConfig {
1358    restart_policy: RestartPolicy,
1359    /// This module's RESOLVED drain budget: per-module config when present,
1360    /// else `default_drain_timeout`.
1361    drain_timeout: Duration,
1362    /// Shared with the status handle so the attested value changes atomically
1363    /// when a rescan updates the running drain policy.
1364    effective_drain_timeout: Arc<Mutex<Duration>>,
1365    /// The supervisor-wide fallback, kept so a configuration update that
1366    /// REMOVES the per-module override can re-resolve to it.
1367    default_drain_timeout: Duration,
1368    health: HealthConfig,
1369    connection_file_path: Option<PathBuf>,
1370    capture_logs_dir: Option<PathBuf>,
1371    forwarding: Option<Arc<ForwardingTable>>,
1372    /// The shared handle, so every spawn path (initial, restart, reload) records the
1373    /// reserved-module launch nonce the HELLO verifier checks against.
1374    supervisor_handle: Option<SupervisorHandle>,
1375    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1376    /// status queries.
1377    ///
1378    /// One ring per module, held across every respawn. The lines explaining an exit
1379    /// are written BEFORE that exit, so a ring recreated per process would be empty
1380    /// exactly when it is asked for.
1381    stderr_ring: Arc<Mutex<StderrRing>>,
1382    terminal_ring: Arc<Mutex<TerminalRing>>,
1383    spawn_events: SpawnEventFeed,
1384    child_roster: ChildRoster,
1385    #[cfg(target_os = "linux")]
1386    cgroup_placement: Option<subc_cgroup::Placement>,
1387    #[cfg(test)]
1388    test_seed_stale_facts_before_enable_spawn: bool,
1389}
1390
1391#[derive(Debug, Clone, PartialEq, Eq)]
1392struct SupervisedConfiguration {
1393    spec: ModuleSpec,
1394    health: HealthConfig,
1395}
1396
1397/// Shared daemon lookup table for supervised module handles.
1398///
1399/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1400/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1401/// launch nonces recorded at spawn are checked by the same daemon instance.
1402#[derive(Debug, Clone, Default)]
1403pub struct SupervisorHandle {
1404    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1405    spawn_events: SpawnEventFeed,
1406    /// The current expected launch nonce for each reserved module_id. Set when the
1407    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1408    /// non-reserved module never has an entry here and is never nonce-checked.
1409    /// Reserved module ids and the nonce that authorizes their next HELLO.
1410    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1411    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1412    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1413    /// had NO entry and admitted anyone: the reservation protected the nonce
1414    /// holder, not the NAME (found live by CKCRED's canary probe registering
1415    /// against a reserved scratch id).
1416    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1417    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1418    ///
1419    /// This is deliberately in-memory only: subc is state-free across daemon
1420    /// restarts, and the tombstone only explains the hours-after-removal window
1421    /// while this executing daemon is still alive. Do not persist it in a store.
1422    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1423    /// The current launch nonce for every supervised spawn. This is separate from
1424    /// reserved_nonces because consumer route.open attestation applies to all spawned
1425    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1426    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1427    /// Reserved namespace prefixes mapped to the supervised owner module whose
1428    /// current spawn nonce authorizes HELLO claims below the prefix.
1429    ///
1430    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1431    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1432    /// accidental collisions and lower-trust processes from squatting protected
1433    /// namespaces.
1434    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1435    /// Blue/green swaps in progress, by module id. An entry exists from just
1436    /// before the candidate process is spawned until the swap has failed, or
1437    /// has cut over and the old process is gone. While it exists, HELLO for the
1438    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1439    /// consumer attestation accepts both processes' nonces.
1440    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1441    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1442    promotion_observer: PromotionObserverSlot,
1443    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1444    /// this daemon-wide ordering, a rescan could retire or update a module while a
1445    /// concurrent reload still held its old handle and launch specification.
1446    operation_lock: Arc<AsyncMutex<()>>,
1447}
1448
1449/// Told when a swap has promoted its candidate to be the module's active
1450/// registration.
1451///
1452/// An ordinary HELLO runs the control plane's registration side effects (the
1453/// capability cache, the deny census, the requirement recompute) as it
1454/// registers. A swap candidate's HELLO does not, because it is not routable;
1455/// promotion is when those must run instead, and promotion happens in the
1456/// supervisor, which has no other way into the control handler.
1457pub(crate) trait SwapPromotionObserver: Send + Sync {
1458    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1459}
1460
1461/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1462/// control handler) owns this handle, so a strong reference back would be a
1463/// cycle that keeps both alive.
1464#[derive(Clone, Default)]
1465struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1466
1467impl fmt::Debug for PromotionObserverSlot {
1468    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1469        f.write_str("PromotionObserverSlot")
1470    }
1471}
1472
1473/// The nonces of one open swap.
1474#[derive(Debug, Clone)]
1475struct OpenSwap {
1476    /// The launch nonce minted for the candidate process. It is the swap
1477    /// token: the only thing that admits a HELLO into the candidate slot.
1478    candidate_nonce: String,
1479    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1480    /// here because cutover moves the module's recorded spawn nonce to the
1481    /// candidate while the incumbent is still draining and its consumers are
1482    /// still attesting with this one.
1483    incumbent_nonce: Option<String>,
1484    /// Set once a HELLO has been admitted with the swap token, so the token
1485    /// admits one registration and cannot be replayed after cutover empties
1486    /// the candidate slot.
1487    candidate_admitted: bool,
1488}
1489
1490/// What the swap gate says about a HELLO. See
1491/// [`SupervisorHandle::swap_hello_admission`].
1492#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1493pub(crate) enum SwapHelloAdmission {
1494    /// No swap is open for the id (or the HELLO carries the incumbent's own
1495    /// nonce); the ordinary gates decide.
1496    NotSwapping,
1497    /// The HELLO carries the swap token: register it into the candidate slot.
1498    Candidate,
1499    /// A swap is open and the HELLO carries a nonce the supervisor did not
1500    /// mint for this id, no nonce, or a token already used.
1501    Refused,
1502}
1503
1504#[derive(Debug, Clone, PartialEq, Eq)]
1505pub(crate) enum ReservedHelloRejection {
1506    Exact {
1507        module_id: String,
1508    },
1509    Prefix {
1510        prefix: String,
1511        owner_module_id: String,
1512    },
1513}
1514
1515impl SupervisorHandle {
1516    pub fn new() -> Self {
1517        Self::default()
1518    }
1519
1520    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1521        self.spawn_events.snapshot()
1522    }
1523
1524    pub(crate) fn subscribe_spawns(
1525        &self,
1526        connection_id: ConnectionId,
1527        corr: u64,
1528        version: u8,
1529        since: Option<SpawnCursor>,
1530        sink: FrameSink,
1531    ) -> Result<(), SpawnSubscribeRefusal> {
1532        self.spawn_events
1533            .subscribe(connection_id, corr, version, since, sink)
1534    }
1535
1536    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1537        self.spawn_events.cancel(connection_id, corr)
1538    }
1539
1540    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1541        self.spawn_events.remove_connection(connection_id);
1542    }
1543
1544    #[cfg(any(test, feature = "test-support"))]
1545    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1546        assert!(capacity > 0, "spawn event capacity must be non-zero");
1547        self.spawn_events.set_capacity(capacity);
1548    }
1549
1550    #[cfg(any(test, feature = "test-support"))]
1551    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1552        self.spawn_events.subscriber_count()
1553    }
1554
1555    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1556    /// a respawn invalidates stale consumer identities.
1557    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1558        self.spawn_nonces
1559            .lock()
1560            .unwrap_or_else(|poisoned| poisoned.into_inner())
1561            .insert(module_id.to_string(), nonce);
1562    }
1563
1564    /// Record the launch nonce expected from the next HELLO for a reserved module,
1565    /// replacing any prior nonce (a respawn invalidates the previous one).
1566    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1567        self.reserved_nonces
1568            .lock()
1569            .unwrap_or_else(|poisoned| poisoned.into_inner())
1570            .insert(module_id.to_string(), Some(nonce));
1571    }
1572
1573    /// Record namespace prefixes owned by a supervised module.
1574    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1575        let mut owners = self
1576            .reserved_prefix_owners
1577            .lock()
1578            .unwrap_or_else(|poisoned| poisoned.into_inner());
1579        owners.retain(|_, owner| owner != owner_module_id);
1580        for prefix in prefixes {
1581            owners.insert(prefix.clone(), owner_module_id.to_string());
1582        }
1583    }
1584
1585    /// The launch nonce most recently minted for a module's spawn, if any.
1586    #[cfg(test)]
1587    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1588        self.spawn_nonces
1589            .lock()
1590            .unwrap_or_else(|poisoned| poisoned.into_inner())
1591            .get(module_id)
1592            .cloned()
1593    }
1594
1595    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1596        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1597        let spawn_nonce = self
1598            .spawn_nonces
1599            .lock()
1600            .unwrap_or_else(|poisoned| poisoned.into_inner())
1601            .get(&spec.module_id)
1602            .cloned();
1603        let mut reserved_nonces = self
1604            .reserved_nonces
1605            .lock()
1606            .unwrap_or_else(|poisoned| poisoned.into_inner());
1607        if spec.reserved {
1608            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1609            // reserved name whose module has never spawned has no legitimate
1610            // holder, and the entry's absence is what used to leave the name
1611            // open to the first claimant.
1612            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1613        }
1614        drop(reserved_nonces);
1615        // A later unreserved declaration must not silently unreserve an id that
1616        // was retained after its reserved configuration was removed. The explicit
1617        // release ceremony is the only operation that retires that gate.
1618        self.removal_tombstones
1619            .lock()
1620            .unwrap_or_else(|poisoned| poisoned.into_inner())
1621            .remove(&spec.module_id);
1622    }
1623
1624    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
1625    /// authorized only by its expected nonce; otherwise a matching reserved prefix
1626    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
1627    /// with no matching prefix are always authorized.
1628    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1629        self.reserved_hello_rejection(module_id, presented)
1630            .is_none()
1631    }
1632
1633    pub(crate) fn reserved_hello_rejection(
1634        &self,
1635        module_id: &str,
1636        presented: Option<&str>,
1637    ) -> Option<ReservedHelloRejection> {
1638        let nonces = self
1639            .reserved_nonces
1640            .lock()
1641            .unwrap_or_else(|poisoned| poisoned.into_inner());
1642        if let Some(expected) = nonces.get(module_id) {
1643            // `None` = reserved with no legitimate holder: refuse every
1644            // presentation, because no process can hold a nonce that was never
1645            // minted. Only a real minted nonce admits, in constant time.
1646            let authorized = match expected {
1647                Some(expected) => {
1648                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1649                }
1650                None => false,
1651            };
1652            if authorized {
1653                return None;
1654            }
1655            return Some(ReservedHelloRejection::Exact {
1656                module_id: module_id.to_string(),
1657            });
1658        }
1659        drop(nonces);
1660
1661        let matched_prefix = self
1662            .reserved_prefix_owners
1663            .lock()
1664            .unwrap_or_else(|poisoned| poisoned.into_inner())
1665            .iter()
1666            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
1667            .max_by_key(|(prefix, _)| prefix.len())
1668            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
1669        let (prefix, owner_module_id) = matched_prefix?;
1670
1671        let authorized = presented.is_some_and(|presented| {
1672            self.spawn_nonces
1673                .lock()
1674                .unwrap_or_else(|poisoned| poisoned.into_inner())
1675                .get(&owner_module_id)
1676                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
1677                // While the owner is being swapped, children started by
1678                // either of its two processes hold that process's nonce.
1679                || self.swap_nonce_matches(&owner_module_id, presented)
1680        });
1681        if authorized {
1682            None
1683        } else {
1684            Some(ReservedHelloRejection::Prefix {
1685                prefix,
1686                owner_module_id,
1687            })
1688        }
1689    }
1690
1691    /// Whether a consumer connection proved it came from a daemon-spawned module.
1692    ///
1693    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
1694    /// accepted only for module ids the supervisor has spawned.
1695    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
1696        if presented.is_empty() {
1697            return false;
1698        }
1699        let nonces = self
1700            .spawn_nonces
1701            .lock()
1702            .unwrap_or_else(|poisoned| poisoned.into_inner());
1703        let current = nonces
1704            .get(module_id)
1705            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
1706        drop(nonces);
1707        // During a swap two processes of the module are alive, and a consumer
1708        // started by either one presents that process's nonce. Accepting only
1709        // the recorded one would fail the incumbent's consumers for the whole
1710        // overlap once cutover moves the record to the candidate.
1711        current || self.swap_nonce_matches(module_id, presented)
1712    }
1713
1714    /// Whether `presented` is either nonce of an open swap for `module_id`.
1715    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
1716        let swaps = self
1717            .swaps
1718            .lock()
1719            .unwrap_or_else(|poisoned| poisoned.into_inner());
1720        swaps.get(module_id).is_some_and(|swap| {
1721            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
1722                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
1723                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
1724                })
1725        })
1726    }
1727
1728    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
1729    /// Called before the candidate process exists.
1730    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
1731        let incumbent_nonce = self
1732            .spawn_nonces
1733            .lock()
1734            .unwrap_or_else(|poisoned| poisoned.into_inner())
1735            .get(module_id)
1736            .cloned();
1737        self.swaps
1738            .lock()
1739            .unwrap_or_else(|poisoned| poisoned.into_inner())
1740            .insert(
1741                module_id.to_string(),
1742                OpenSwap {
1743                    candidate_nonce,
1744                    incumbent_nonce,
1745                    candidate_admitted: false,
1746                },
1747            );
1748    }
1749
1750    /// Close the swap for `module_id`, releasing whichever nonce is no longer
1751    /// the module's recorded one.
1752    pub(crate) fn close_swap(&self, module_id: &str) {
1753        self.swaps
1754            .lock()
1755            .unwrap_or_else(|poisoned| poisoned.into_inner())
1756            .remove(module_id);
1757    }
1758
1759    /// Install the observer told about swap promotions, replacing any earlier
1760    /// one.
1761    pub(crate) fn set_swap_promotion_observer(
1762        &self,
1763        observer: std::sync::Weak<dyn SwapPromotionObserver>,
1764    ) {
1765        *self
1766            .promotion_observer
1767            .0
1768            .lock()
1769            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
1770    }
1771
1772    /// Tell the installed observer, if it is still alive, that a swap promoted
1773    /// `registration`.
1774    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
1775        let observer = self
1776            .promotion_observer
1777            .0
1778            .lock()
1779            .unwrap_or_else(|poisoned| poisoned.into_inner())
1780            .as_ref()
1781            .and_then(std::sync::Weak::upgrade);
1782        if let Some(observer) = observer {
1783            observer.swap_promoted(registration);
1784        }
1785    }
1786
1787    /// Whether a swap is open for `module_id`.
1788    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
1789        self.swaps
1790            .lock()
1791            .unwrap_or_else(|poisoned| poisoned.into_inner())
1792            .contains_key(module_id)
1793    }
1794
1795    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
1796    /// respawn would, once cutover has made the candidate the module's process.
1797    /// The swap stays open so the incumbent's nonce keeps attesting until the
1798    /// incumbent has drained and exited.
1799    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
1800        let candidate_nonce = self
1801            .swaps
1802            .lock()
1803            .unwrap_or_else(|poisoned| poisoned.into_inner())
1804            .get(module_id)
1805            .map(|swap| swap.candidate_nonce.clone());
1806        let Some(nonce) = candidate_nonce else {
1807            return;
1808        };
1809        self.set_spawn_nonce(module_id, nonce.clone());
1810        if reserved {
1811            self.set_reserved_nonce(module_id, nonce);
1812        }
1813    }
1814
1815    /// The swap gate for a HELLO claiming `module_id`.
1816    ///
1817    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
1818    /// presents the candidate nonce, which the reserved gate (holding the
1819    /// incumbent's nonce) would refuse as `reserved_module` before swap
1820    /// admission was ever reached. And it applies to unreserved ids too: for an
1821    /// unreserved id the only thing that ever stopped a second process claiming
1822    /// a live id was the `duplicate_module_id` refusal, which is exactly the
1823    /// refusal a swap lifts for its candidate.
1824    ///
1825    /// The incumbent's own nonce falls through to the ordinary gates, which
1826    /// treat it as they always have (a live incumbent is refused as a
1827    /// duplicate). Anything else while a swap is open is refused, including an
1828    /// absent nonce.
1829    pub(crate) fn swap_hello_admission(
1830        &self,
1831        module_id: &str,
1832        presented: Option<&str>,
1833    ) -> SwapHelloAdmission {
1834        let swaps = self
1835            .swaps
1836            .lock()
1837            .unwrap_or_else(|poisoned| poisoned.into_inner());
1838        let Some(swap) = swaps.get(module_id) else {
1839            return SwapHelloAdmission::NotSwapping;
1840        };
1841        let Some(presented) = presented else {
1842            return SwapHelloAdmission::Refused;
1843        };
1844        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
1845            return if swap.candidate_admitted {
1846                SwapHelloAdmission::Refused
1847            } else {
1848                SwapHelloAdmission::Candidate
1849            };
1850        }
1851        if swap
1852            .incumbent_nonce
1853            .as_deref()
1854            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
1855        {
1856            return SwapHelloAdmission::NotSwapping;
1857        }
1858        SwapHelloAdmission::Refused
1859    }
1860
1861    /// Record that the swap token has registered a candidate, so it admits no
1862    /// second HELLO.
1863    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
1864        if let Some(swap) = self
1865            .swaps
1866            .lock()
1867            .unwrap_or_else(|poisoned| poisoned.into_inner())
1868            .get_mut(module_id)
1869        {
1870            swap.candidate_admitted = true;
1871        }
1872    }
1873
1874    /// Test/support lookup for the current launch nonce of a supervised spawn.
1875    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1876        self.spawn_nonces
1877            .lock()
1878            .unwrap_or_else(|poisoned| poisoned.into_inner())
1879            .get(module_id)
1880            .cloned()
1881    }
1882
1883    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
1884    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1885        self.reserved_nonces
1886            .lock()
1887            .unwrap_or_else(|poisoned| poisoned.into_inner())
1888            .get(module_id)
1889            .cloned()
1890            .flatten()
1891    }
1892
1893    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
1894        let mut modules = self
1895            .modules
1896            .lock()
1897            .unwrap_or_else(|poisoned| poisoned.into_inner());
1898        modules.insert(module.module_id().to_string(), module)
1899    }
1900
1901    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
1902        let modules = self
1903            .modules
1904            .lock()
1905            .unwrap_or_else(|poisoned| poisoned.into_inner());
1906        modules.get(module_id).cloned()
1907    }
1908
1909    pub(crate) fn record_late_health_answer(
1910        &self,
1911        module_id: &str,
1912        latency_ms: u64,
1913    ) -> Result<bool, SuperviseError> {
1914        let Some(module) = self.get(module_id) else {
1915            return Ok(false);
1916        };
1917        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
1918            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
1919            state.health.last_late_answer_latency_ms = Some(latency_ms);
1920            // A late answer is an answer: the module served the probe, just past
1921            // the deadline. Leaving the miss streak in place while logging
1922            // "proves the module is alive" is how a CPU-starved module that
1923            // answers every probe a few seconds late still marches to the
1924            // threshold and gets killed — the exact kill class `NoAnswer` is
1925            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
1926            // is degradation, and degradation reports; it does not restart.
1927            state.health.consecutive_failures = 0;
1928        })?;
1929        Ok(true)
1930    }
1931
1932    /// Arm the one-shot marker for the module process that this caller
1933    /// deliberately initiated severance against. Generic connection teardown
1934    /// must not call this:
1935    /// a surviving process would otherwise retain an exemption for a later
1936    /// genuine crash.
1937    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
1938        let Some(module) = self.get(module_id) else {
1939            return Ok(false);
1940        };
1941        let status = module.status()?;
1942        let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
1943            return Ok(false);
1944        };
1945        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
1946    }
1947
1948    pub fn list(&self) -> Vec<SupervisedModule> {
1949        let modules = self
1950            .modules
1951            .lock()
1952            .unwrap_or_else(|poisoned| poisoned.into_inner());
1953        let mut modules = modules.values().cloned().collect::<Vec<_>>();
1954        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
1955        modules
1956    }
1957
1958    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
1959        self.spawn_nonces
1960            .lock()
1961            .unwrap_or_else(|poisoned| poisoned.into_inner())
1962            .remove(module_id);
1963        self.close_swap(module_id);
1964        let mut reserved_nonces = self
1965            .reserved_nonces
1966            .lock()
1967            .unwrap_or_else(|poisoned| poisoned.into_inner());
1968        if reserved_nonces.contains_key(module_id) {
1969            // The old nonce must die with the removed process, but the exact-id
1970            // gate remains until an operator explicitly releases it.
1971            reserved_nonces.insert(module_id.to_string(), None);
1972        }
1973        drop(reserved_nonces);
1974        self.reserved_prefix_owners
1975            .lock()
1976            .unwrap_or_else(|poisoned| poisoned.into_inner())
1977            .retain(|_, owner| owner != module_id);
1978        self.modules
1979            .lock()
1980            .unwrap_or_else(|poisoned| poisoned.into_inner())
1981            .remove(module_id)
1982    }
1983
1984    /// Remember a module removed by a non-preview rescan so route.open can
1985    /// distinguish that intentional removal from an unknown id.
1986    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
1987        self.removal_tombstones
1988            .lock()
1989            .unwrap_or_else(|poisoned| poisoned.into_inner())
1990            .insert(module_id.to_string(), unix_ms_now());
1991    }
1992
1993    /// Return how long ago a rescan removed this module in milliseconds.
1994    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
1995        self.removal_tombstones
1996            .lock()
1997            .unwrap_or_else(|poisoned| poisoned.into_inner())
1998            .get(module_id)
1999            .copied()
2000            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2001    }
2002
2003    /// Retire a reserved-id gate only after its module has left supervision.
2004    ///
2005    /// A retained gate has no live nonce (`None`), so releasing any other entry
2006    /// would weaken a currently configured or otherwise active reservation.
2007    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2008        if self.get(module_id).is_some() {
2009            return false;
2010        }
2011        let mut reserved_nonces = self
2012            .reserved_nonces
2013            .lock()
2014            .unwrap_or_else(|poisoned| poisoned.into_inner());
2015        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2016            return false;
2017        }
2018        reserved_nonces.remove(module_id);
2019        true
2020    }
2021
2022    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2023        Arc::clone(&self.operation_lock)
2024    }
2025}
2026
2027/// Process supervisor for subc-owned singleton modules.
2028#[derive(Debug, Clone)]
2029pub struct Supervisor {
2030    registry: Arc<Registry>,
2031    restart_policy: RestartPolicy,
2032    drain_timeout: Duration,
2033    connection_file_path: Option<PathBuf>,
2034    capture_logs_dir: Option<PathBuf>,
2035    forwarding: Option<Arc<ForwardingTable>>,
2036    process_liveness: Arc<SupervisorProcessLiveness>,
2037    supervisor_handle: Option<SupervisorHandle>,
2038    health: HealthConfig,
2039    daemon_start_clock: crate::clock::StartClock,
2040    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2041    spawn_events: SpawnEventFeed,
2042    provenance_probe: ExecutableIdentityProbe,
2043    /// Every process spawned through this supervisor (and its clones) and not
2044    /// yet reaped, so daemon shutdown can end them.
2045    child_roster: ChildRoster,
2046    #[cfg(target_os = "linux")]
2047    cgroup_placement: Option<subc_cgroup::Placement>,
2048}
2049
2050impl Supervisor {
2051    /// The first step of an announced daemon shutdown, before the notice and
2052    /// before any connection is closed.
2053    ///
2054    /// Sets the daemon-shutdown flag first: from here on no module is
2055    /// respawned (crash restart, operator restart, or swap), and every child
2056    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2057    /// the module exits on the EOF this shutdown gives it or is signalled by a
2058    /// service manager that kills the whole cgroup. Then writes the journal's
2059    /// shutdown marker, which records the instant and closes this daemon
2060    /// incarnation's stretch of the journal.
2061    #[cfg(unix)]
2062    pub(crate) fn begin_daemon_shutdown(&self) {
2063        self.child_roster.close();
2064        if let Some(journal) = &self.terminal_journal {
2065            journal.stamp_shutdown();
2066        }
2067    }
2068
2069    /// Announce a cut while established connections can still carry replies.
2070    /// These budgets promise notice and a bounded wait, not child completion;
2071    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2072    #[cfg(unix)]
2073    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2074        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2075        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2076        let Some(forwarding) = &self.forwarding else {
2077            return Ok(());
2078        };
2079        let module_ids = forwarding
2080            .begin_daemon_drain()
2081            .map_err(SuperviseError::Forwarding)?;
2082        let deadline_ms =
2083            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2084        let mut notices = tokio::task::JoinSet::new();
2085        let mut drains = Vec::new();
2086        for module_id in module_ids {
2087            let Some(target) = forwarding
2088                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2089                .map_err(SuperviseError::Forwarding)?
2090            else {
2091                continue;
2092            };
2093            let routes = forwarding
2094                .endpoint_routes(target.endpoint)
2095                .map_err(SuperviseError::Forwarding)?;
2096            // Restart allows deployed consumers to reopen after the new daemon
2097            // appears. The wire reason stays `restart`; what tells a daemon cut
2098            // apart from a module restart afterwards is the terminal record
2099            // itself, whose disposition is `daemon_shutdown` for every exit
2100            // observed once `begin_daemon_shutdown` has run.
2101            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2102                reason: RouteCloseReason::Restart,
2103                deadline_ms,
2104            })
2105            .expect("module draining serializes");
2106            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2107            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2108            for route in routes {
2109                let client = route.goodbye_target;
2110                if let Some((_, channels)) = clients
2111                    .iter_mut()
2112                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2113                {
2114                    channels.push(client.channel);
2115                } else {
2116                    let channel = client.channel;
2117                    clients.push((client, vec![channel]));
2118                }
2119            }
2120            for (client, mut channels) in clients {
2121                channels.sort_unstable();
2122                channels.dedup();
2123                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2124                    module_id: module_id.clone(),
2125                    channels,
2126                    reason: RouteCloseReason::Restart,
2127                })
2128                .expect("route closing serializes");
2129                recipients.push((client.sink, client.negotiated_ver, closing));
2130            }
2131            for (sink, version, body) in recipients {
2132                notices.spawn(async move {
2133                    let frame = Frame::build_with_version(
2134                        version,
2135                        FrameType::Push,
2136                        control_flags(),
2137                        0,
2138                        0,
2139                        0,
2140                        body,
2141                    )
2142                    .expect("bounded lifecycle notice frame builds");
2143                    sink.send_flushed(frame).await
2144                });
2145            }
2146            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2147            drains.push((module_id, target.endpoint, gauges));
2148        }
2149        // A quiet forwarding table is not proof that queued notices reached the
2150        // socket. Wait for writer flush acknowledgements before testing quiescence.
2151        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2152        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2153            if !matches!(result, Ok(Ok(()))) {
2154                warn!(?result, "daemon shutdown notice delivery failed");
2155            }
2156        }
2157        notices.abort_all();
2158        let deadline = Instant::now() + DRAIN_BUDGET;
2159        let mut waits = tokio::task::JoinSet::new();
2160        for (module_id, endpoint, gauges) in drains {
2161            let forwarding = Arc::clone(forwarding);
2162            let mut runtime = self.runtime_config();
2163            runtime.health.cadence = Duration::from_millis(100);
2164            waits.spawn(async move {
2165                wait_for_forwarding_quiescence(
2166                    &forwarding,
2167                    &module_id,
2168                    &runtime,
2169                    endpoint,
2170                    deadline,
2171                    &gauges,
2172                    DrainScope::Active,
2173                )
2174                .await
2175            });
2176        }
2177        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2178            if !matches!(result, Ok(Ok(true))) {
2179                warn!(?result, "daemon shutdown drain did not reach quiescence");
2180            }
2181        }
2182        Ok(())
2183    }
2184
2185    /// The last step of an announced daemon shutdown, after the notice and the
2186    /// drain: send every registered module a module GOODBYE, the same planned
2187    /// stop signal `ck module stop` gives, then close every connection so each
2188    /// subc module sees EOF and starts its own teardown, then end every
2189    /// supervised child that has not exited
2190    /// by its own deadline (its drain budget, capped). Modules lead their own
2191    /// process groups, so a
2192    /// service manager's group kill no longer reaches them; without this a
2193    /// child that does not stop on EOF (every `protocol: "none"` child, which
2194    /// has no connection) would outlive the daemon. Every wait is bounded (see
2195    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2196    #[cfg(unix)]
2197    pub(crate) async fn end_children_for_daemon_shutdown(
2198        &self,
2199        already_escalated: bool,
2200        escalate: impl std::future::Future<Output = ()>,
2201    ) {
2202        tokio::pin!(escalate);
2203        let mut escalated = already_escalated;
2204        if let Some(forwarding) = &self.forwarding {
2205            let reason = CloseReason::new(
2206                "daemon_shutdown",
2207                "the daemon is exiting after its shutdown notice and drain",
2208            );
2209            if escalated {
2210                // The operator asked to stop waiting: queue the GOODBYEs but
2211                // do not wait for them to be written.
2212                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2213            } else {
2214                tokio::select! {
2215                    biased;
2216                    _ = escalate.as_mut() => {
2217                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2218                        escalated = true;
2219                    }
2220                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2221                }
2222            }
2223            let closed = forwarding.close_all_connections(&reason);
2224            debug!(closed, "closed established connections for daemon shutdown");
2225        }
2226        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2227        // already completed and must not be polled again; the child shutdown
2228        // wait is told it is escalated and gets a future that never fires.
2229        let escalated_here = escalated && !already_escalated;
2230        let remaining_escalate = async move {
2231            if escalated_here {
2232                std::future::pending::<()>().await;
2233            } else {
2234                escalate.await;
2235            }
2236        };
2237        crate::child_roster::end_children_for_daemon_shutdown(
2238            &self.child_roster,
2239            escalated,
2240            remaining_escalate,
2241        )
2242        .await;
2243    }
2244
2245    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2246        Self {
2247            registry,
2248            restart_policy,
2249            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2250            connection_file_path: None,
2251            capture_logs_dir: None,
2252            forwarding: None,
2253            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2254            supervisor_handle: None,
2255            health: HealthConfig::default(),
2256            daemon_start_clock: crate::clock::StartClock::capture(),
2257            terminal_journal: None,
2258            spawn_events: SpawnEventFeed::default(),
2259            provenance_probe: ExecutableIdentityProbe::default(),
2260            child_roster: ChildRoster::default(),
2261            #[cfg(target_os = "linux")]
2262            cgroup_placement: None,
2263        }
2264    }
2265
2266    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2267        self.drain_timeout = drain_timeout;
2268        self
2269    }
2270
2271    pub fn with_process_liveness(
2272        mut self,
2273        process_liveness: Arc<SupervisorProcessLiveness>,
2274    ) -> Self {
2275        self.process_liveness = process_liveness;
2276        self
2277    }
2278
2279    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2280        self.connection_file_path = Some(connection_file_path.into());
2281        self
2282    }
2283
2284    /// Enables daemon-owned capture files for supervised stdout and stderr.
2285    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2286        self.capture_logs_dir = Some(logs_dir.into());
2287        self
2288    }
2289
2290    /// Names this daemon lifetime in spawn events, independently of whether a
2291    /// terminal journal is configured.
2292    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2293        // A millisecond start stamp can repeat after clock rollback or a rapid
2294        // restart. Use the connection file's random daemon_id instead: it already
2295        // identifies this daemon lifetime independently of the wall clock.
2296        self.spawn_events.configure_incarnation(daemon_incarnation);
2297        self
2298    }
2299
2300    /// Enables best-effort history shared by every supervised module. Without
2301    /// it, terminal history is kept only in each module's in-memory ring.
2302    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2303        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2304        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2305            path,
2306            daemon_incarnation,
2307        )));
2308        this
2309    }
2310
2311    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2312        self.forwarding = Some(forwarding);
2313        self
2314    }
2315
2316    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2317        self.spawn_events = supervisor_handle.spawn_events.clone();
2318        self.supervisor_handle = Some(supervisor_handle);
2319        self
2320    }
2321
2322    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2323        self.health = health;
2324        self
2325    }
2326
2327    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2328    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2329    /// record is kept.
2330    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2331        self.child_roster.record_to(path.into());
2332        self
2333    }
2334
2335    #[cfg(target_os = "linux")]
2336    pub fn with_cgroup_placement(
2337        mut self,
2338        cgroup_placement: Option<subc_cgroup::Placement>,
2339    ) -> Self {
2340        self.cgroup_placement = cgroup_placement;
2341        self
2342    }
2343
2344    /// Spawn `spec.program` and start monitoring it.
2345    ///
2346    /// The child is expected to parse `--subc <connection-file-path>`, read the
2347    /// TCP+key connection file, authenticate to the already-running listener, and
2348    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2349    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2350        validate_spec(&spec)?;
2351
2352        let runtime = self.runtime_config();
2353        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2354        let child = spawn_child(
2355            &spec,
2356            runtime.connection_file_path.as_deref(),
2357            self.supervisor_handle.as_ref(),
2358            &runtime.stderr_ring,
2359            runtime.capture_logs_dir.as_deref(),
2360            &runtime.child_roster,
2361            #[cfg(target_os = "linux")]
2362            runtime.cgroup_placement.as_ref(),
2363        )?;
2364        set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2365        self.process_liveness
2366            .track(spec.module_id.clone(), Arc::clone(&snapshot));
2367
2368        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2369    }
2370
2371    /// Start supervising a module declared in daemon configuration.
2372    ///
2373    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2374    /// failures in the supervisor handle so operator-facing `supervisor.list`
2375    /// reflects every configured module while daemon startup continues.
2376    pub fn supervise_configured(
2377        &self,
2378        spec: ModuleSpec,
2379        enabled: bool,
2380    ) -> Result<SupervisedModule, SuperviseError> {
2381        validate_spec(&spec)?;
2382
2383        let runtime = self.runtime_config();
2384        if !enabled {
2385            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2386            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2387        }
2388
2389        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2390        match spawn_child(
2391            &spec,
2392            runtime.connection_file_path.as_deref(),
2393            self.supervisor_handle.as_ref(),
2394            &runtime.stderr_ring,
2395            runtime.capture_logs_dir.as_deref(),
2396            &runtime.child_roster,
2397            #[cfg(target_os = "linux")]
2398            runtime.cgroup_placement.as_ref(),
2399        ) {
2400            Ok(child) => {
2401                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2402                self.process_liveness
2403                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2404                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2405            }
2406            Err(err) => {
2407                error!(
2408                    module_id = %spec.module_id,
2409                    program = %spec.program.display(),
2410                    error = %err,
2411                    "configured module failed to spawn; marking failed and continuing"
2412                );
2413                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2414                Ok(self.supervised_module(spec, runtime, snapshot, None))
2415            }
2416        }
2417    }
2418
2419    /// Supervise a configured module with its own health, drain, and crash
2420    /// budget. The restart policy is per-module because the config file is:
2421    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2422    /// module that is expensive to restart should not be forced onto the same
2423    /// budget as one that is cheap.
2424    pub fn supervise_configured_with_health(
2425        &self,
2426        spec: ModuleSpec,
2427        enabled: bool,
2428        health: HealthConfig,
2429        drain_timeout_ms: Option<u64>,
2430        restart_policy: RestartPolicy,
2431    ) -> Result<SupervisedModule, SuperviseError> {
2432        validate_spec(&spec)?;
2433
2434        let mut runtime = self.runtime_config();
2435        runtime.health = health;
2436        runtime.restart_policy = restart_policy;
2437        if let Some(ms) = drain_timeout_ms {
2438            runtime.drain_timeout = Duration::from_millis(ms);
2439            *runtime
2440                .effective_drain_timeout
2441                .lock()
2442                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2443        }
2444        if !enabled {
2445            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2446            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2447        }
2448
2449        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2450        match spawn_child(
2451            &spec,
2452            runtime.connection_file_path.as_deref(),
2453            self.supervisor_handle.as_ref(),
2454            &runtime.stderr_ring,
2455            runtime.capture_logs_dir.as_deref(),
2456            &runtime.child_roster,
2457            #[cfg(target_os = "linux")]
2458            runtime.cgroup_placement.as_ref(),
2459        ) {
2460            Ok(child) => {
2461                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2462                self.process_liveness
2463                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2464                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2465            }
2466            Err(err) => {
2467                if health.critical {
2468                    error!(
2469                        module_id = %spec.module_id,
2470                        program = %spec.program.display(),
2471                        error = %err,
2472                        "critical configured module failed to spawn; marking failed and alerting"
2473                    );
2474                } else {
2475                    error!(
2476                        module_id = %spec.module_id,
2477                        program = %spec.program.display(),
2478                        error = %err,
2479                        "configured module failed to spawn; marking failed and continuing"
2480                    );
2481                }
2482                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2483                Ok(self.supervised_module(spec, runtime, snapshot, None))
2484            }
2485        }
2486    }
2487
2488    fn runtime_config(&self) -> SupervisorRuntimeConfig {
2489        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2490        SupervisorRuntimeConfig {
2491            restart_policy: self.restart_policy,
2492            drain_timeout: self.drain_timeout,
2493            // Shared with this module's roster copy: daemon shutdown waits on
2494            // each child for the module's own drain budget, as resolved now.
2495            child_roster: self
2496                .child_roster
2497                .for_module(Arc::clone(&effective_drain_timeout)),
2498            effective_drain_timeout,
2499            default_drain_timeout: self.drain_timeout,
2500            health: self.health,
2501            connection_file_path: self.connection_file_path.clone(),
2502            capture_logs_dir: self.capture_logs_dir.clone(),
2503            forwarding: self.forwarding.clone(),
2504            supervisor_handle: self.supervisor_handle.clone(),
2505            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
2506            terminal_ring: Arc::new(Mutex::new(
2507                TerminalRing::new(
2508                    TerminalRingConfig::default(),
2509                    self.daemon_start_clock.started_at_ms(),
2510                )
2511                .with_start_clock(self.daemon_start_clock)
2512                .with_journal(self.terminal_journal.clone())
2513                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
2514            )),
2515            spawn_events: self.spawn_events.clone(),
2516            #[cfg(target_os = "linux")]
2517            cgroup_placement: self.cgroup_placement.clone(),
2518            #[cfg(test)]
2519            test_seed_stale_facts_before_enable_spawn: false,
2520        }
2521    }
2522
2523    fn supervised_module(
2524        &self,
2525        spec: ModuleSpec,
2526        runtime: SupervisorRuntimeConfig,
2527        snapshot: SharedSnapshot,
2528        child: Option<SupervisedChild>,
2529    ) -> SupervisedModule {
2530        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
2531            spec: spec.clone(),
2532            health: runtime.health,
2533        }));
2534        let stderr_ring = Arc::clone(&runtime.stderr_ring);
2535        let terminal_ring = Arc::clone(&runtime.terminal_ring);
2536        // The module's OWN policy, which may be its per-module config rather than
2537        // the supervisor-wide one; status must report the budget the supervise
2538        // loop actually enforces.
2539        let restart_policy = runtime.restart_policy;
2540        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
2541        let (tx, rx) = mpsc::channel(4);
2542        let monitor = tokio::spawn(supervise_loop(
2543            spec.clone(),
2544            runtime,
2545            Arc::clone(&self.registry),
2546            Arc::clone(&self.process_liveness),
2547            Arc::clone(&snapshot),
2548            child,
2549            rx,
2550        ));
2551
2552        let module_id = spec.module_id.clone();
2553        let module = SupervisedModule {
2554            inner: Arc::new(SupervisedModuleInner {
2555                module_id: module_id.clone(),
2556                registry: Arc::clone(&self.registry),
2557                snapshot,
2558                configuration,
2559                stderr_ring,
2560                terminal_ring,
2561                commands: tx,
2562                monitor: Mutex::new(Some(monitor)),
2563                restart_policy,
2564                effective_drain_timeout,
2565                provenance_probe: self.provenance_probe.clone(),
2566            }),
2567        };
2568        if let Some(supervisor_handle) = &self.supervisor_handle {
2569            supervisor_handle.apply_identity_configuration(&spec);
2570            supervisor_handle.insert(module.clone());
2571        }
2572        module
2573    }
2574}
2575
2576impl Default for Supervisor {
2577    fn default() -> Self {
2578        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
2579    }
2580}
2581
2582/// Handle to one supervised child process.
2583#[derive(Clone)]
2584pub struct SupervisedModule {
2585    inner: Arc<SupervisedModuleInner>,
2586}
2587
2588struct SupervisedModuleInner {
2589    module_id: String,
2590    registry: Arc<Registry>,
2591    snapshot: SharedSnapshot,
2592    configuration: Arc<Mutex<SupervisedConfiguration>>,
2593    stderr_ring: Arc<Mutex<StderrRing>>,
2594    terminal_ring: Arc<Mutex<TerminalRing>>,
2595    commands: mpsc::Sender<SupervisorCommand>,
2596    monitor: Mutex<Option<JoinHandle<()>>>,
2597    /// Copied from the supervisor's runtime config at spawn so `status()` can
2598    /// report the restart budget without reaching back into the supervisor. The
2599    /// policy is fixed for the process's lifetime, so a copy cannot drift.
2600    restart_policy: RestartPolicy,
2601    effective_drain_timeout: Arc<Mutex<Duration>>,
2602    provenance_probe: ExecutableIdentityProbe,
2603}
2604
2605impl fmt::Debug for SupervisedModule {
2606    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2607        f.debug_struct("SupervisedModule")
2608            .field("module_id", &self.inner.module_id)
2609            .field("status", &self.status())
2610            .finish_non_exhaustive()
2611    }
2612}
2613
2614impl SupervisedModule {
2615    pub fn module_id(&self) -> &str {
2616        &self.inner.module_id
2617    }
2618
2619    /// Test-only: put one probe miss on the streak, the way
2620    /// `handle_health_probe_failure` does, so tests can assert what a later
2621    /// event does to the streak without driving the whole probe loop.
2622    #[cfg(test)]
2623    pub(crate) fn record_health_probe_failure_for_test(
2624        &self,
2625        detail: &str,
2626    ) -> Result<(), SuperviseError> {
2627        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
2628            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
2629            state.health.detail = Some(detail.to_string());
2630        })
2631    }
2632
2633    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
2634        Ok(lock_snapshot(&self.inner.snapshot)?.state)
2635    }
2636
2637    /// The module's retained stderr, newest lines last.
2638    ///
2639    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
2640    /// module, `supervisor.list` renders every module, and putting it in the
2641    /// shared snapshot would make each status read carry a payload almost nobody
2642    /// asked for. Callers that want the text ask for it.
2643    pub fn stderr_tail(
2644        &self,
2645        max_lines: Option<usize>,
2646        max_bytes: Option<usize>,
2647    ) -> StderrTailSnapshot {
2648        self.inner
2649            .stderr_ring
2650            .lock()
2651            .unwrap_or_else(|poisoned| poisoned.into_inner())
2652            .snapshot(max_lines, max_bytes)
2653    }
2654
2655    /// The module's bounded terminal history, oldest retained exit first.
2656    ///
2657    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
2658    /// daemon whose in-memory history was necessarily reset.
2659    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
2660        self.inner
2661            .terminal_ring
2662            .lock()
2663            .unwrap_or_else(|poisoned| poisoned.into_inner())
2664            .snapshot()
2665    }
2666
2667    /// Retained observations from the current ring and all journal generations.
2668    ///
2669    /// Blocking: this reads the journal files. Async callers use
2670    /// [`Self::read_durable_terminal_history`].
2671    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
2672        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
2673    }
2674
2675    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
2676    /// read (up to every retained generation) never occupies a runtime worker.
2677    /// Fails only if the blocking task could not finish (runtime shutdown or a
2678    /// panic in the read).
2679    pub(crate) async fn read_durable_terminal_history(
2680        &self,
2681    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
2682        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
2683        let module_id = self.inner.module_id.clone();
2684        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
2685            .await
2686    }
2687
2688    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
2689        self.status_with_snapshot_lock(&self.inner.snapshot, None)
2690    }
2691
2692    pub(crate) fn record_deliberate_severance(
2693        &self,
2694        identity: ProcessIdentity,
2695    ) -> Result<bool, SuperviseError> {
2696        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2697        if snapshot.pid != Some(identity.pid)
2698            || snapshot.process_start_time != Some(identity.start_time)
2699        {
2700            return Ok(false);
2701        }
2702        snapshot.deliberate_severance = Some(identity);
2703        Ok(true)
2704    }
2705
2706    /// Read status for a channel-0 renderer and report a contended snapshot lock.
2707    ///
2708    /// Internal supervision callers use [`Self::status`] so writer-side machinery
2709    /// does not produce reader-observability logs.
2710    pub(crate) fn status_for_control(
2711        &self,
2712        caller: &'static str,
2713    ) -> Result<ModuleStatus, SuperviseError> {
2714        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
2715    }
2716
2717    fn status_with_snapshot_lock(
2718        &self,
2719        snapshot: &SharedSnapshot,
2720        caller: Option<&'static str>,
2721    ) -> Result<ModuleStatus, SuperviseError> {
2722        let mut guard = match caller {
2723            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
2724            None => lock_snapshot(snapshot)?,
2725        };
2726        // Read the budget through the pruning path so a reader sees the same
2727        // in-window count the restart decision would use, not a stale total.
2728        let restart_count =
2729            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
2730        let snapshot = guard.clone();
2731        drop(guard);
2732        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
2733            SuperviseError::StatePoisoned {
2734                module_id: Some(self.inner.module_id.clone()),
2735            }
2736        })?;
2737        let registration_active = self
2738            .inner
2739            .registry
2740            .get_module(&self.inner.module_id)
2741            .map_err(SuperviseError::Registry)?
2742            .is_some();
2743        let protocol = self.declared_protocol()?;
2744        let running_process =
2745            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
2746        // Registration is the difference between the two protocols and the only
2747        // one: a subc module that has not registered cannot serve a request even
2748        // though its process is up, and a `none` module never registers at all,
2749        // so requiring it there would pin `live` to false for the whole life of
2750        // a perfectly healthy process.
2751        let live = match protocol {
2752            ModuleProtocol::Subc => running_process && registration_active,
2753            ModuleProtocol::None => running_process,
2754        };
2755
2756        Ok(ModuleStatus {
2757            module_id: self.inner.module_id.clone(),
2758            state: snapshot.state,
2759            enabled: snapshot.enabled,
2760            process_alive: snapshot.process_alive,
2761            registration_active,
2762            protocol,
2763            live,
2764            restart_count,
2765            lifetime_restarts: snapshot.lifetime_restarts,
2766            spawn_generation: snapshot.spawn_generation,
2767            max_restarts: self.inner.restart_policy.max_restarts,
2768            restart_window: self.inner.restart_policy.window,
2769            drain_timeout,
2770            restart_backoff: self.inner.restart_policy.backoff,
2771            restart_max_backoff: self.inner.restart_policy.max_backoff,
2772            pid: snapshot.pid,
2773            spawned_at_ms: snapshot.spawned_at_ms,
2774            spawned_from: snapshot.spawned_from,
2775            process_start_time: snapshot.process_start_time,
2776            last_exit: snapshot.last_exit,
2777            health: snapshot.health,
2778        })
2779    }
2780
2781    #[cfg(test)]
2782    pub(crate) fn hold_snapshot_for_test(
2783        &self,
2784        acquired: std::sync::mpsc::Sender<()>,
2785        hold: Duration,
2786    ) -> std::thread::JoinHandle<()> {
2787        let snapshot = Arc::clone(&self.inner.snapshot);
2788        std::thread::spawn(move || {
2789            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
2790            acquired
2791                .send(())
2792                .expect("test receiver waits for snapshot lock");
2793            std::thread::sleep(hold);
2794        })
2795    }
2796
2797    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
2798        let snapshot = match lock_snapshot(&self.inner.snapshot) {
2799            Ok(snapshot) => snapshot.clone(),
2800            Err(_) => {
2801                return subc_control::RunningImageAgreement::Unavailable {
2802                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
2803                };
2804            }
2805        };
2806        self.inner
2807            .provenance_probe
2808            .observe(
2809                snapshot.pid,
2810                snapshot.spawned_from.as_deref(),
2811                snapshot.spawned_file_identity,
2812                snapshot.process_start_time,
2813            )
2814            .await
2815    }
2816
2817    /// Memory and CPU time of the module's current process, read now. Only the
2818    /// process the supervisor spawned is read, not processes it has started.
2819    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
2820        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
2821            Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
2822            Err(_) => {
2823                return subc_control::ChildResourceUsage::Unavailable {
2824                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
2825                }
2826            }
2827        };
2828        crate::child_resources::read(pid, start_time)
2829    }
2830
2831    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
2832        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2833        Ok(match snapshot.state {
2834            ModuleState::Restarting => true,
2835            ModuleState::Failed | ModuleState::Disabled => false,
2836            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
2837        })
2838    }
2839
2840    #[cfg(test)]
2841    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
2842        self.is_warming_with_snapshot_lock(None)
2843    }
2844
2845    pub(crate) fn is_warming_for_control(
2846        &self,
2847        caller: &'static str,
2848    ) -> Result<bool, SuperviseError> {
2849        self.is_warming_with_snapshot_lock(Some(caller))
2850    }
2851
2852    fn is_warming_with_snapshot_lock(
2853        &self,
2854        caller: Option<&'static str>,
2855    ) -> Result<bool, SuperviseError> {
2856        let snapshot = match caller {
2857            Some(caller) => {
2858                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
2859            }
2860            None => lock_snapshot(&self.inner.snapshot)?,
2861        }
2862        .clone();
2863        Ok(matches!(
2864            snapshot.state,
2865            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
2866        ))
2867    }
2868
2869    /// Drain the module and stop monitoring it.
2870    pub async fn drain(&self) -> Result<(), SuperviseError> {
2871        self.stop().await
2872    }
2873
2874    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
2875        match self.state()? {
2876            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2877            ModuleState::Starting
2878            | ModuleState::Running
2879            | ModuleState::Unresponsive
2880            | ModuleState::Restarting
2881            | ModuleState::Draining
2882            | ModuleState::Disabled => {}
2883        }
2884
2885        let (reply_tx, reply_rx) = oneshot::channel();
2886        self.inner
2887            .commands
2888            .send(SupervisorCommand::Retire { reply: reply_tx })
2889            .await
2890            .map_err(|_| SuperviseError::CommandClosed {
2891                module_id: self.inner.module_id.clone(),
2892            })?;
2893        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2894            module_id: self.inner.module_id.clone(),
2895        })?
2896    }
2897
2898    pub async fn stop(&self) -> Result<(), SuperviseError> {
2899        match self.state()? {
2900            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2901            ModuleState::Starting
2902            | ModuleState::Running
2903            | ModuleState::Unresponsive
2904            | ModuleState::Restarting
2905            | ModuleState::Draining
2906            | ModuleState::Disabled => {}
2907        }
2908
2909        let (reply_tx, reply_rx) = oneshot::channel();
2910        self.inner
2911            .commands
2912            .send(SupervisorCommand::Drain { reply: reply_tx })
2913            .await
2914            .map_err(|_| SuperviseError::CommandClosed {
2915                module_id: self.inner.module_id.clone(),
2916            })?;
2917        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2918            module_id: self.inner.module_id.clone(),
2919        })?
2920    }
2921
2922    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
2923        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
2924        let (reply_tx, reply_rx) = oneshot::channel();
2925        self.inner
2926            .commands
2927            .send(SupervisorCommand::Restart {
2928                drain_timeout_ms,
2929                received_at_generation,
2930                queued_at: Instant::now(),
2931                reply: reply_tx,
2932            })
2933            .await
2934            .map_err(|_| SuperviseError::CommandClosed {
2935                module_id: self.inner.module_id.clone(),
2936            })?;
2937        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2938            module_id: self.inner.module_id.clone(),
2939        })?
2940    }
2941
2942    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
2943    /// `supervisor_swap` module. Returns once the swap has cut over (the old
2944    /// process then drains in the background of the supervise loop) or has
2945    /// failed, leaving the old process serving.
2946    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
2947        let (reply_tx, reply_rx) = oneshot::channel();
2948        self.inner
2949            .commands
2950            .send(SupervisorCommand::Swap {
2951                ready_timeout,
2952                reply: reply_tx,
2953            })
2954            .await
2955            .map_err(|_| SuperviseError::CommandClosed {
2956                module_id: self.inner.module_id.clone(),
2957            })?;
2958        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2959            module_id: self.inner.module_id.clone(),
2960        })?
2961    }
2962
2963    pub async fn reload(&self) -> Result<(), SuperviseError> {
2964        let (reply_tx, reply_rx) = oneshot::channel();
2965        self.inner
2966            .commands
2967            .send(SupervisorCommand::Reload { reply: reply_tx })
2968            .await
2969            .map_err(|_| SuperviseError::CommandClosed {
2970                module_id: self.inner.module_id.clone(),
2971            })?;
2972        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2973            module_id: self.inner.module_id.clone(),
2974        })?
2975    }
2976
2977    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
2978        let (reply_tx, reply_rx) = oneshot::channel();
2979        self.inner
2980            .commands
2981            .send(SupervisorCommand::SetEnabled {
2982                enabled,
2983                reply: reply_tx,
2984            })
2985            .await
2986            .map_err(|_| SuperviseError::CommandClosed {
2987                module_id: self.inner.module_id.clone(),
2988            })?;
2989        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2990            module_id: self.inner.module_id.clone(),
2991        })?
2992    }
2993
2994    /// This module's declared protocol, read from the same stored configuration
2995    /// the rescan diff compares and `update_configuration` rewrites, so a status
2996    /// read and the supervise loop can never disagree about which protocol is in
2997    /// force.
2998    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
2999        Ok(self
3000            .inner
3001            .configuration
3002            .lock()
3003            .map_err(|_| SuperviseError::StatePoisoned {
3004                module_id: Some(self.inner.module_id.clone()),
3005            })?
3006            .spec
3007            .protocol)
3008    }
3009
3010    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3011        let configuration =
3012            self.inner
3013                .configuration
3014                .lock()
3015                .map_err(|_| SuperviseError::StatePoisoned {
3016                    module_id: Some(self.inner.module_id.clone()),
3017                })?;
3018        Ok((configuration.spec.clone(), configuration.health))
3019    }
3020
3021    /// Replace this module's launch spec, keeping its health and drain policy,
3022    /// the way a rescan does for a changed config entry. The running process is
3023    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3024    #[cfg(any(test, feature = "test-support"))]
3025    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3026        let (_, health) = self.configuration()?;
3027        let drain_timeout_ms = u64::try_from(
3028            self.inner
3029                .effective_drain_timeout
3030                .lock()
3031                .unwrap_or_else(|poisoned| poisoned.into_inner())
3032                .as_millis(),
3033        )
3034        .ok();
3035        self.update_configuration(spec, health, drain_timeout_ms)
3036            .await
3037    }
3038
3039    pub(crate) async fn update_configuration(
3040        &self,
3041        spec: ModuleSpec,
3042        health: HealthConfig,
3043        drain_timeout_ms: Option<u64>,
3044    ) -> Result<(), SuperviseError> {
3045        if spec.module_id != self.inner.module_id {
3046            return Err(SuperviseError::InvalidSpec {
3047                reason: "a supervised module's module_id cannot be changed".to_string(),
3048            });
3049        }
3050        validate_spec(&spec)?;
3051        let (reply_tx, reply_rx) = oneshot::channel();
3052        self.inner
3053            .commands
3054            .send(SupervisorCommand::UpdateConfiguration {
3055                spec: spec.clone(),
3056                health,
3057                drain_timeout_ms,
3058                reply: reply_tx,
3059            })
3060            .await
3061            .map_err(|_| SuperviseError::CommandClosed {
3062                module_id: self.inner.module_id.clone(),
3063            })?;
3064        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3065            module_id: self.inner.module_id.clone(),
3066        })?;
3067        let mut configuration =
3068            self.inner
3069                .configuration
3070                .lock()
3071                .map_err(|_| SuperviseError::StatePoisoned {
3072                    module_id: Some(self.inner.module_id.clone()),
3073                })?;
3074        configuration.spec = spec;
3075        configuration.health = health;
3076        Ok(())
3077    }
3078}
3079
3080impl Drop for SupervisedModuleInner {
3081    fn drop(&mut self) {
3082        let Ok(mut monitor) = self.monitor.lock() else {
3083            return;
3084        };
3085        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3086            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3087                state.state = ModuleState::Stopped;
3088                clear_current_process_facts(state);
3089            });
3090            monitor.abort();
3091        }
3092        let _ = monitor.take();
3093    }
3094}
3095
3096#[derive(Debug)]
3097enum SupervisorCommand {
3098    Drain {
3099        reply: oneshot::Sender<Result<(), SuperviseError>>,
3100    },
3101    Retire {
3102        reply: oneshot::Sender<Result<(), SuperviseError>>,
3103    },
3104    Restart {
3105        /// Operator override for this one restart's drain budget, in ms. `None`
3106        /// uses the module's configured/default budget; `Some(0)` cuts
3107        /// immediately (wedge bounce: a stuck request never settles, so
3108        /// waiting only delays recovery).
3109        drain_timeout_ms: Option<u64>,
3110        /// The module's `spawn_generation` when the request was received, before
3111        /// it waited in the command queue. A queued restart whose module has
3112        /// since spawned a newer process is already satisfied (see the handler).
3113        received_at_generation: u64,
3114        /// When the request entered the command queue, so the handler can log
3115        /// how long it waited behind the loop's other work.
3116        queued_at: Instant,
3117        reply: oneshot::Sender<Result<(), SuperviseError>>,
3118    },
3119    Reload {
3120        reply: oneshot::Sender<Result<(), SuperviseError>>,
3121    },
3122    SetEnabled {
3123        enabled: bool,
3124        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3125    },
3126    UpdateConfiguration {
3127        spec: ModuleSpec,
3128        health: HealthConfig,
3129        /// Per-module drain override from the new config; `None` re-resolves to
3130        /// the supervisor-wide default.
3131        drain_timeout_ms: Option<u64>,
3132        reply: oneshot::Sender<()>,
3133    },
3134    Swap {
3135        /// How long the candidate may take to register and declare itself
3136        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3137        ready_timeout: Option<Duration>,
3138        /// Answered at cutover or failure; the incumbent's drain follows.
3139        reply: oneshot::Sender<Result<(), SuperviseError>>,
3140    },
3141}
3142
3143#[derive(Debug)]
3144pub enum SuperviseError {
3145    InvalidSpec {
3146        reason: String,
3147    },
3148    Spawn {
3149        program: PathBuf,
3150        source: io::Error,
3151        cgroup_path: Option<PathBuf>,
3152    },
3153    Cgroup {
3154        module_id: String,
3155        source: io::Error,
3156    },
3157    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3158    /// than spawn a reserved module without its identity binding.
3159    LaunchNonce {
3160        reason: String,
3161    },
3162    Wait {
3163        module_id: String,
3164        source: io::Error,
3165    },
3166    Kill {
3167        module_id: String,
3168        source: io::Error,
3169    },
3170    Forwarding(ForwardingError),
3171    Registry(RegistryError),
3172    ReloadUnavailable {
3173        module_id: String,
3174        reason: String,
3175    },
3176    /// An operator restart/reload was requested for a module that is currently
3177    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3178    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3179    /// by a restart, so these commands are rejected instead of re-enabling it.
3180    Disabled {
3181        module_id: String,
3182    },
3183    ReloadFailed {
3184        module_id: String,
3185        reason: String,
3186    },
3187    RegistrationStillActive {
3188        module_id: String,
3189        waited: Duration,
3190    },
3191    StatePoisoned {
3192        module_id: Option<String>,
3193    },
3194    CommandClosed {
3195        module_id: String,
3196    },
3197    /// A restart or reload arrived while a swap's candidate was warming. The
3198    /// swap owns the module until it cuts over or fails; a stop or disable
3199    /// would have aborted it instead.
3200    SwapInProgress {
3201        module_id: String,
3202    },
3203    /// A swap was refused before anything was spawned.
3204    SwapRefused {
3205        module_id: String,
3206        reason: SwapRefusal,
3207    },
3208    /// A swap spawned a candidate and gave up on it. The candidate has been
3209    /// killed and its slot freed; the incumbent was left serving and was never
3210    /// drained, except in the one `CutoverLost` case described on that arm.
3211    SwapFailed {
3212        module_id: String,
3213        arm: SwapFailureArm,
3214        detail: String,
3215        /// How the candidate exited, when it exited on its own before the
3216        /// supervisor gave up on it.
3217        candidate_exit: Option<ExitReport>,
3218    },
3219}
3220
3221/// Why a swap was refused before a candidate was spawned.
3222#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3223pub enum SwapRefusal {
3224    /// The module's config does not declare `overlap: "safe"`.
3225    OverlapExclusive,
3226    /// The module is not registered, so there is no incumbent to keep serving
3227    /// and nothing a swap would improve on; a plain restart is the tool.
3228    NotRegistered,
3229    /// The module does not speak the subc wire, so a candidate could never
3230    /// register or declare itself ready.
3231    ProtocolNone,
3232    /// The supervisor lacks the forwarding table (to cut routes over) or the
3233    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3234    NotConfigured,
3235    /// A swap is already open for this module.
3236    AlreadySwapping,
3237}
3238
3239impl SwapRefusal {
3240    pub fn as_str(self) -> &'static str {
3241        match self {
3242            Self::OverlapExclusive => "overlap_exclusive",
3243            Self::NotRegistered => "not_registered",
3244            Self::ProtocolNone => "protocol_none",
3245            Self::NotConfigured => "not_configured",
3246            Self::AlreadySwapping => "already_swapping",
3247        }
3248    }
3249}
3250
3251/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3252/// serving and undrained; see `CutoverLost`.
3253#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3254pub enum SwapFailureArm {
3255    /// The candidate process could not be started.
3256    SpawnFailed,
3257    /// The candidate did not register within the readiness budget.
3258    NeverRegistered,
3259    /// The candidate registered but did not declare itself ready in time.
3260    NeverReady,
3261    /// The candidate exited before cutover.
3262    CandidateExited,
3263    /// The candidate declared itself ready but failed its health probe.
3264    CandidateUnhealthy,
3265    /// An operator stop, disable or retire arrived while the candidate warmed.
3266    /// The candidate was killed and the operator's command then carried out on
3267    /// the incumbent.
3268    Interrupted,
3269    /// The candidate's connection closed at the moment of cutover. If it
3270    /// closed before forwarding moved, the incumbent is untouched. If it closed
3271    /// between the forwarding and registry halves of cutover, forwarding can no
3272    /// longer route to the incumbent, so the module is restarted plainly.
3273    CutoverLost,
3274}
3275
3276impl SwapFailureArm {
3277    pub fn as_str(self) -> &'static str {
3278        match self {
3279            Self::SpawnFailed => "spawn_failed",
3280            Self::NeverRegistered => "never_registered",
3281            Self::NeverReady => "never_ready",
3282            Self::CandidateExited => "candidate_exited",
3283            Self::CandidateUnhealthy => "candidate_unhealthy",
3284            Self::Interrupted => "interrupted",
3285            Self::CutoverLost => "cutover_lost",
3286        }
3287    }
3288}
3289
3290impl fmt::Display for SuperviseError {
3291    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3292        match self {
3293            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3294            Self::Spawn {
3295                program,
3296                source,
3297                cgroup_path: Some(cgroup_path),
3298            } => write!(
3299                f,
3300                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3301                cgroup_path.display(),
3302                program.display()
3303            ),
3304            Self::Spawn {
3305                program,
3306                source,
3307                cgroup_path: None,
3308            } => write!(
3309                f,
3310                "failed to spawn module '{}': {source}",
3311                program.display()
3312            ),
3313            Self::Cgroup { module_id, source } => {
3314                write!(
3315                    f,
3316                    "failed to prepare cgroup for module '{module_id}': {source}"
3317                )
3318            }
3319            Self::LaunchNonce { reason } => {
3320                write!(
3321                    f,
3322                    "failed to generate reserved-module launch nonce: {reason}"
3323                )
3324            }
3325            Self::Wait { module_id, source } => {
3326                write!(f, "failed to wait for module '{module_id}': {source}")
3327            }
3328            Self::Kill { module_id, source } => {
3329                write!(f, "failed to kill module '{module_id}': {source}")
3330            }
3331            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3332            Self::Registry(err) => write!(f, "registry error: {err}"),
3333            Self::ReloadUnavailable { module_id, reason } => {
3334                write!(f, "reload unavailable for module '{module_id}': {reason}")
3335            }
3336            Self::Disabled { module_id } => {
3337                write!(
3338                    f,
3339                    "module '{module_id}' is disabled; enable it before restart or reload"
3340                )
3341            }
3342            Self::ReloadFailed { module_id, reason } => {
3343                write!(f, "reload failed for module '{module_id}': {reason}")
3344            }
3345            Self::RegistrationStillActive { module_id, waited } => write!(
3346                f,
3347                "module '{module_id}' registration remained active after waiting {waited:?}"
3348            ),
3349            Self::StatePoisoned { module_id } => match module_id {
3350                Some(module_id) => {
3351                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3352                }
3353                None => write!(f, "supervisor state was poisoned"),
3354            },
3355            Self::CommandClosed { module_id } => {
3356                write!(
3357                    f,
3358                    "supervisor command channel for module '{module_id}' is closed"
3359                )
3360            }
3361            Self::SwapInProgress { module_id } => write!(
3362                f,
3363                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3364            ),
3365            Self::SwapRefused { module_id, reason } => match reason {
3366                SwapRefusal::OverlapExclusive => write!(
3367                    f,
3368                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3369                ),
3370                SwapRefusal::NotRegistered => write!(
3371                    f,
3372                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3373                ),
3374                SwapRefusal::ProtocolNone => write!(
3375                    f,
3376                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3377                ),
3378                SwapRefusal::NotConfigured => write!(
3379                    f,
3380                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3381                ),
3382                SwapRefusal::AlreadySwapping => {
3383                    write!(f, "module '{module_id}' is already being swapped")
3384                }
3385            },
3386            Self::SwapFailed {
3387                module_id,
3388                arm,
3389                detail,
3390                ..
3391            } => write!(
3392                f,
3393                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3394                arm.as_str()
3395            ),
3396        }
3397    }
3398}
3399
3400impl Error for SuperviseError {
3401    fn source(&self) -> Option<&(dyn Error + 'static)> {
3402        match self {
3403            Self::Spawn { source, .. }
3404            | Self::Cgroup { source, .. }
3405            | Self::Wait { source, .. }
3406            | Self::Kill { source, .. } => Some(source),
3407            Self::Forwarding(err) => Some(err),
3408            Self::Registry(err) => Some(err),
3409            Self::LaunchNonce { .. }
3410            | Self::InvalidSpec { .. }
3411            | Self::ReloadUnavailable { .. }
3412            | Self::Disabled { .. }
3413            | Self::ReloadFailed { .. }
3414            | Self::RegistrationStillActive { .. }
3415            | Self::StatePoisoned { .. }
3416            | Self::CommandClosed { .. }
3417            | Self::SwapInProgress { .. }
3418            | Self::SwapRefused { .. }
3419            | Self::SwapFailed { .. } => None,
3420        }
3421    }
3422}
3423
3424pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3425    if spec.module_id.trim().is_empty() {
3426        return Err(SuperviseError::InvalidSpec {
3427            reason: "module_id must not be empty".to_string(),
3428        });
3429    }
3430
3431    Ok(())
3432}
3433
3434#[derive(Debug, Default)]
3435struct HealthProbeRuntime {
3436    registered_connection: Option<crate::ConnectionId>,
3437    advertised: bool,
3438    next_probe_at: Option<Instant>,
3439    probe_index: u64,
3440}
3441
3442impl HealthProbeRuntime {
3443    fn refresh_registration(
3444        &mut self,
3445        spec: &ModuleSpec,
3446        runtime: &SupervisorRuntimeConfig,
3447        registry: &Registry,
3448        snapshot: &SharedSnapshot,
3449    ) {
3450        // THE PROBE GATE FOR A MODULE THAT SPEAKS NO SUBC WIRE, placed here
3451        // because this is the only place that ever arms a probe: leaving
3452        // `advertised` false and `next_probe_at` empty makes `due()` false
3453        // forever, so `run_health_probe_cycle` -- and with it every arm of
3454        // `probe_module_health`, including the one that reads an absent
3455        // registration as proof the module is gone and escalates to a restart --
3456        // is unreachable for this module.
3457        //
3458        // That arm is right for a subc module and is exactly wrong here: a
3459        // `protocol: "none"` module never registers by declaration, so the
3460        // absence it would classify is the module working as configured.
3461        if spec.protocol == ModuleProtocol::None {
3462            self.registered_connection = None;
3463            self.advertised = false;
3464            self.next_probe_at = None;
3465            return;
3466        }
3467
3468        let registration = match registry.get_module(&spec.module_id) {
3469            Ok(registration) => registration,
3470            Err(err) => {
3471                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3472                self.advertised = false;
3473                self.next_probe_at = None;
3474                return;
3475            }
3476        };
3477
3478        let Some(registration) = registration else {
3479            self.registered_connection = None;
3480            self.advertised = false;
3481            self.next_probe_at = None;
3482            return;
3483        };
3484
3485        let advertised = registration
3486            .control_ops
3487            .iter()
3488            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
3489        if !advertised {
3490            self.registered_connection = Some(registration.connection_id);
3491            self.advertised = false;
3492            self.next_probe_at = None;
3493            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3494                state.health.status = SupervisorHealthStatus::Unknown;
3495                state.health.consecutive_failures = 0;
3496                state.health.last_probe_ms = None;
3497                state.health.detail = None;
3498                state.health.metrics = None;
3499            });
3500            return;
3501        }
3502
3503        let reregistered = self.registered_connection != Some(registration.connection_id);
3504        self.registered_connection = Some(registration.connection_id);
3505        self.advertised = true;
3506        if reregistered || self.next_probe_at.is_none() {
3507            self.probe_index = 0;
3508            self.next_probe_at = Some(
3509                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3510            );
3511            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3512                state.health.status = SupervisorHealthStatus::Unknown;
3513                state.health.consecutive_failures = 0;
3514                state.health.detail = None;
3515                state.health.metrics = None;
3516            });
3517        }
3518    }
3519
3520    fn wake_after(&self) -> Duration {
3521        if !self.advertised {
3522            return REGISTRY_RELEASE_POLL;
3523        }
3524        self.next_probe_at
3525            .map(|next| next.saturating_duration_since(Instant::now()))
3526            .unwrap_or(REGISTRY_RELEASE_POLL)
3527    }
3528
3529    fn due(&self) -> bool {
3530        self.advertised
3531            && self
3532                .next_probe_at
3533                .is_some_and(|next| Instant::now() >= next)
3534    }
3535
3536    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
3537        self.probe_index = self.probe_index.wrapping_add(1);
3538        self.next_probe_at = Some(
3539            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
3540        );
3541    }
3542}
3543
3544/// What a failed health probe actually OBSERVED, kept apart from how it reads.
3545///
3546/// This was a struct with a single `message: String`, and every one of the
3547/// fifteen construction sites collapsed into it. Each site knows exactly what it
3548/// saw -- the lane is gone, the module did not answer in time, the module
3549/// answered with the wrong thing -- and `handle_health_probe_failure` then
3550/// treated all of them identically: increment a counter, compare to a threshold,
3551/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
3552/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
3553///
3554/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
3555///
3556/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
3557///   answer on it again.
3558/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
3559///   AND with a perfectly healthy one that lost a CPU race -- which is what
3560///   happens under machine load, and is how this supervisor killed a healthy
3561///   module three times in one day.
3562/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
3563///   Restarting on it is defensible, but it is not the silence case and should
3564///   never be counted as one.
3565/// * `Misconfigured` is a daemon-side fault. The module has not been asked
3566///   anything, so it cannot be evidence about the module at all.
3567///
3568/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
3569/// one that fires most often, and while every variant collapsed into one string
3570/// it carried the same weight as the strongest.
3571///
3572/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
3573/// DESIGN and a reader stopping at it gets the build backwards: the restart
3574/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
3575/// probes still increment the failure streak and drive escalation at the
3576/// threshold (see `is_proof_of_death` below for why that is deliberate and
3577/// what gates the change). Absence of evidence restarts modules today.
3578#[derive(Debug)]
3579enum HealthProbeEvidence {
3580    /// The module's control lane is gone. Proof of death.
3581    LaneDead,
3582    /// No reply within the deadline. Proves nothing about the module's state.
3583    NoAnswer,
3584    /// The module replied, but not with a usable health report. Proves it is alive.
3585    BadAnswer,
3586    /// The daemon could not ask. Says nothing about the module.
3587    Misconfigured,
3588}
3589
3590#[derive(Debug)]
3591struct HealthProbeError {
3592    evidence: HealthProbeEvidence,
3593    message: String,
3594}
3595
3596impl HealthProbeError {
3597    fn lane_dead(message: impl Into<String>) -> Self {
3598        Self::with(HealthProbeEvidence::LaneDead, message)
3599    }
3600
3601    fn no_answer(message: impl Into<String>) -> Self {
3602        Self::with(HealthProbeEvidence::NoAnswer, message)
3603    }
3604
3605    fn bad_answer(message: impl Into<String>) -> Self {
3606        Self::with(HealthProbeEvidence::BadAnswer, message)
3607    }
3608
3609    fn misconfigured(message: impl Into<String>) -> Self {
3610        Self::with(HealthProbeEvidence::Misconfigured, message)
3611    }
3612
3613    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
3614        Self {
3615            evidence,
3616            message: message.into(),
3617        }
3618    }
3619
3620    /// Whether this observation is proof the module cannot serve.
3621    ///
3622    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
3623    /// variant that fires under CPU starvation, and treating it as proof is the
3624    /// defect this enum exists to make impossible to reintroduce silently.
3625    ///
3626    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
3627    /// to restart also needs a bound for the case it excludes -- a genuinely
3628    /// wedged module, alive but never answering -- and that bound must come from
3629    /// the distribution of real late-answer latencies, which nothing measures
3630    /// yet. Landing the classification first makes the later change a one-line
3631    /// decision against evidence that already exists, rather than two unproven
3632    /// changes at once.
3633    #[allow(dead_code)]
3634    fn is_proof_of_death(&self) -> bool {
3635        matches!(self.evidence, HealthProbeEvidence::LaneDead)
3636    }
3637
3638    /// Short stable label for logs and the health snapshot.
3639    ///
3640    /// An operator reading `ck health` currently cannot tell "the module is gone"
3641    /// from "the module did not answer in five seconds", because both render as
3642    /// prose in the same field. These labels are what make the two
3643    /// distinguishable at a glance, and they are what a later restart-policy
3644    /// change will be argued from.
3645    fn label(&self) -> &'static str {
3646        match self.evidence {
3647            HealthProbeEvidence::LaneDead => "lane-dead",
3648            HealthProbeEvidence::NoAnswer => "no-answer",
3649            HealthProbeEvidence::BadAnswer => "bad-answer",
3650            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
3651        }
3652    }
3653}
3654
3655impl fmt::Display for HealthProbeError {
3656    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3657        f.write_str(&self.message)
3658    }
3659}
3660
3661async fn run_health_probe_cycle(
3662    spec: &ModuleSpec,
3663    runtime: &SupervisorRuntimeConfig,
3664    registry: &Registry,
3665    process_liveness: &SupervisorProcessLiveness,
3666    snapshot: &SharedSnapshot,
3667    child: &mut Option<SupervisedChild>,
3668) {
3669    let now_ms = unix_ms_now();
3670    match probe_module_health(&spec.module_id, runtime, None).await {
3671        Ok(report) => {
3672            handle_health_report(
3673                spec,
3674                runtime,
3675                registry,
3676                process_liveness,
3677                snapshot,
3678                child,
3679                report,
3680                now_ms,
3681            )
3682            .await;
3683        }
3684        Err(err) => {
3685            handle_health_probe_failure(
3686                spec,
3687                runtime,
3688                registry,
3689                process_liveness,
3690                snapshot,
3691                child,
3692                err,
3693                now_ms,
3694            )
3695            .await;
3696        }
3697    }
3698}
3699
3700async fn probe_module_health(
3701    module_id: &str,
3702    runtime: &SupervisorRuntimeConfig,
3703    drain_deadline: Option<Instant>,
3704) -> Result<HealthReport, HealthProbeError> {
3705    let Some(forwarding) = runtime.forwarding.as_ref() else {
3706        return Err(HealthProbeError::misconfigured(
3707            "supervisor was not configured with a forwarding table",
3708        ));
3709    };
3710    let probe_started_at = Instant::now();
3711    let mut deadline = probe_started_at + runtime.health.deadline;
3712    if let Some(drain_deadline) = drain_deadline {
3713        deadline = deadline.min(drain_deadline);
3714    }
3715    let pending = if drain_deadline.is_some() {
3716        forwarding.begin_drain_health_probe_rpc_for(
3717            module_id,
3718            MODULE_CONTROL_OP_HEALTH_CHECK,
3719            probe_started_at,
3720            deadline,
3721        )
3722    } else {
3723        forwarding.begin_health_probe_rpc_for(
3724            module_id,
3725            MODULE_CONTROL_OP_HEALTH_CHECK,
3726            probe_started_at,
3727            deadline,
3728        )
3729    }
3730    .map_err(|err| {
3731        // The endpoint is not registered, so there is no live control lane to
3732        // ask. That is the module being absent, not slow.
3733        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3734    })?;
3735    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3736}
3737
3738/// [`probe_module_health`] for one endpoint rather than the id's active one.
3739///
3740/// A swap probes two processes that no by-id lookup reaches: its candidate
3741/// before cutover, and its superseded incumbent (for busy gauges) while the
3742/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
3743/// bounds the by-id drain probe.
3744async fn probe_endpoint_health(
3745    endpoint: crate::ModuleEndpointId,
3746    runtime: &SupervisorRuntimeConfig,
3747    deadline_cap: Option<Instant>,
3748) -> Result<HealthReport, HealthProbeError> {
3749    let Some(forwarding) = runtime.forwarding.as_ref() else {
3750        return Err(HealthProbeError::misconfigured(
3751            "supervisor was not configured with a forwarding table",
3752        ));
3753    };
3754    let probe_started_at = Instant::now();
3755    let mut deadline = probe_started_at + runtime.health.deadline;
3756    if let Some(cap) = deadline_cap {
3757        deadline = deadline.min(cap);
3758    }
3759    let pending = forwarding
3760        .begin_endpoint_health_probe_rpc_for(
3761            endpoint,
3762            MODULE_CONTROL_OP_HEALTH_CHECK,
3763            probe_started_at,
3764            deadline,
3765        )
3766        .map_err(|err| {
3767            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3768        })?;
3769    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3770}
3771
3772/// Send a begun health probe and classify its answer.
3773async fn await_health_probe(
3774    forwarding: &ForwardingTable,
3775    pending: PendingModuleControlRpc,
3776    deadline: Instant,
3777    probe_budget: Duration,
3778) -> Result<HealthReport, HealthProbeError> {
3779    let PendingModuleControlRpc {
3780        endpoint,
3781        module_sink,
3782        negotiated_ver,
3783        corr,
3784        receiver,
3785    } = pending;
3786    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
3787        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
3788    })?;
3789    let frame = Frame::build_with_version(
3790        negotiated_ver,
3791        FrameType::Request,
3792        control_flags(),
3793        0,
3794        0,
3795        corr,
3796        body,
3797    )
3798    .map_err(|err| {
3799        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
3800    })?;
3801
3802    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
3803    // blocks waiting for capacity when the module's egress queue is full, and an
3804    // unbounded await here freezes the whole supervision actor (it stops polling
3805    // Child::wait and supervisor commands), making the module unrecoverable
3806    // in-band. On timeout the probe fails like any transport failure.
3807    match timeout_at(deadline, module_sink.send(frame)).await {
3808        Ok(Ok(())) => {}
3809        Ok(Err(err)) => {
3810            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3811            // A closed sink means the module's egress channel is gone -- the
3812            // receiving half is dropped when its connection tears down. Proof.
3813            return Err(HealthProbeError::lane_dead(format!(
3814                "failed to send health.check: {err}"
3815            )));
3816        }
3817        Err(_elapsed) => {
3818            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3819            // A full egress queue means the module is not draining its socket, which
3820            // is consistent with a wedged module AND with one whose reader is merely
3821            // starved. Silence, not proof.
3822            return Err(HealthProbeError::no_answer(
3823                "health.check send timed out before enqueue (module egress full)",
3824            ));
3825        }
3826    }
3827
3828    match timeout_at(deadline, receiver).await {
3829        // Each arm records WHAT WAS OBSERVED. Four of them are the module
3830        // demonstrably answering -- rejected, non-health, malformed, wrong op --
3831        // and those prove it is alive even though the probe failed.
3832        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
3833            response.health_report().ok_or_else(|| {
3834                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
3835            })
3836        }
3837        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
3838            format!("health.check rejected: {}", body.message),
3839        )),
3840        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
3841            Err(HealthProbeError::lane_dead(message))
3842        }
3843        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
3844            Err(HealthProbeError::bad_answer(message))
3845        }
3846        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
3847            Err(HealthProbeError::bad_answer(format!(
3848                "expected module-control op '{expected}', got '{actual}'"
3849            )))
3850        }
3851        // A reply that crosses the deadline before this waiter observes it is
3852        // still proof of life. The forwarding path records its end-to-end latency
3853        // before delivering this classification.
3854        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
3855            "module answered health.check after its daemon deadline",
3856        )),
3857        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
3858            "health.check waiter was canceled before the module responded",
3859        )),
3860        Err(_) => {
3861            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
3862            Err(HealthProbeError::no_answer(format!(
3863                "module did not answer health.check within {probe_budget:?}"
3864            )))
3865        }
3866    }
3867}
3868
3869#[allow(clippy::too_many_arguments)]
3870async fn handle_health_report(
3871    spec: &ModuleSpec,
3872    runtime: &SupervisorRuntimeConfig,
3873    registry: &Registry,
3874    process_liveness: &SupervisorProcessLiveness,
3875    snapshot: &SharedSnapshot,
3876    child: &mut Option<SupervisedChild>,
3877    report: HealthReport,
3878    now_ms: u64,
3879) {
3880    let status = supervisor_health_status(report.status);
3881    let detail = report.detail.clone();
3882    let metrics = truncate_health_metrics(report.metrics);
3883    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3884        state.health.status = status;
3885        state.health.last_probe_ms = Some(now_ms);
3886        state.health.detail = detail.clone();
3887        state.health.metrics = metrics.clone();
3888        state.health.consecutive_failures = 0;
3889    });
3890
3891    let action = match report.status {
3892        HealthStatus::Ok => return,
3893        HealthStatus::Degraded => runtime.health.on_degraded,
3894        HealthStatus::Failing => runtime.health.on_failing,
3895    };
3896    apply_l3_health_action(
3897        spec,
3898        runtime,
3899        registry,
3900        process_liveness,
3901        snapshot,
3902        child,
3903        status,
3904        detail.as_deref(),
3905        action,
3906        now_ms,
3907    )
3908    .await;
3909}
3910
3911#[allow(clippy::too_many_arguments)]
3912async fn handle_health_probe_failure(
3913    spec: &ModuleSpec,
3914    runtime: &SupervisorRuntimeConfig,
3915    registry: &Registry,
3916    process_liveness: &SupervisorProcessLiveness,
3917    snapshot: &SharedSnapshot,
3918    child: &mut Option<SupervisedChild>,
3919    err: HealthProbeError,
3920    now_ms: u64,
3921) {
3922    let threshold = runtime.health.failure_threshold.max(1);
3923    let mut failures = 0;
3924    // Carry the evidence class into the operator-visible detail. Without it,
3925    // "module did not answer within 5s" and "the control lane is gone" are two
3926    // prose strings in the same field, and the reader has to know the codebase to
3927    // tell which one is proof of anything.
3928    let detail = format!("[{}] {err}", err.label());
3929    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3930        state.health.last_probe_ms = Some(now_ms);
3931        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3932        state.health.detail = Some(detail.clone());
3933        state.health.metrics = None;
3934        failures = state.health.consecutive_failures;
3935    });
3936
3937    if failures < threshold {
3938        warn!(
3939            module_id = %spec.module_id,
3940            consecutive_failures = failures,
3941            threshold,
3942            evidence = err.label(),
3943            detail = %detail,
3944            "health.check probe failed"
3945        );
3946        return;
3947    }
3948
3949    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3950        state.state = ModuleState::Unresponsive;
3951        state.health.status = SupervisorHealthStatus::Unresponsive;
3952    });
3953    // The evidence class is logged at the kill site because this is the line an
3954    // operator reads after an unexplained restart. A streak of `no-answer` under
3955    // machine load is the known false-positive shape; a `lane-dead` is not.
3956    if runtime.health.critical {
3957        error!(
3958            module_id = %spec.module_id,
3959            status = "unresponsive",
3960            evidence = err.label(),
3961            detail = %detail,
3962            "critical module health alert"
3963        );
3964    } else {
3965        warn!(
3966            module_id = %spec.module_id,
3967            status = "unresponsive",
3968            evidence = err.label(),
3969            detail = %detail,
3970            "module health threshold breached"
3971        );
3972    }
3973    if let Err(err) = health_restart_child(
3974        spec,
3975        runtime,
3976        registry,
3977        process_liveness,
3978        snapshot,
3979        child,
3980        SupervisorHealthStatus::Unresponsive,
3981        Some(&detail),
3982        now_ms,
3983    )
3984    .await
3985    {
3986        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
3987    }
3988}
3989
3990#[allow(clippy::too_many_arguments)]
3991async fn apply_l3_health_action(
3992    spec: &ModuleSpec,
3993    runtime: &SupervisorRuntimeConfig,
3994    registry: &Registry,
3995    process_liveness: &SupervisorProcessLiveness,
3996    snapshot: &SharedSnapshot,
3997    child: &mut Option<SupervisedChild>,
3998    status: SupervisorHealthStatus,
3999    detail: Option<&str>,
4000    action: HealthAction,
4001    now_ms: u64,
4002) {
4003    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4004    match action {
4005        HealthAction::Report => {
4006            info!(
4007                module_id = %spec.module_id,
4008                status = ?status,
4009                detail,
4010                "module reported non-ok health"
4011            );
4012        }
4013        HealthAction::Alert => {
4014            error!(
4015                module_id = %spec.module_id,
4016                status = ?status,
4017                detail,
4018                "module health alert"
4019            );
4020        }
4021        HealthAction::Restart => {
4022            if let Err(err) = health_restart_child(
4023                spec,
4024                runtime,
4025                registry,
4026                process_liveness,
4027                snapshot,
4028                child,
4029                status,
4030                detail,
4031                now_ms,
4032            )
4033            .await
4034            {
4035                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4036            }
4037        }
4038    }
4039}
4040
4041#[allow(clippy::too_many_arguments)]
4042async fn health_restart_child(
4043    spec: &ModuleSpec,
4044    runtime: &SupervisorRuntimeConfig,
4045    registry: &Registry,
4046    process_liveness: &SupervisorProcessLiveness,
4047    snapshot: &SharedSnapshot,
4048    child: &mut Option<SupervisedChild>,
4049    status: SupervisorHealthStatus,
4050    detail: Option<&str>,
4051    now_ms: u64,
4052) -> Result<(), SuperviseError> {
4053    let (enabled, schedule) = {
4054        let mut state = lock_snapshot(snapshot)?;
4055        let enabled = state.enabled;
4056        let schedule = if enabled {
4057            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4058        } else {
4059            None
4060        };
4061        (enabled, schedule)
4062    };
4063
4064    if !enabled {
4065        return Err(SuperviseError::Disabled {
4066            module_id: spec.module_id.clone(),
4067        });
4068    }
4069
4070    if schedule.is_none() {
4071        record_health_action(snapshot, &spec.module_id, "disabled".to_string(), now_ms);
4072        error!(
4073            module_id = %spec.module_id,
4074            status = ?status,
4075            detail,
4076            max_restarts = runtime.restart_policy.max_restarts,
4077            window_secs = runtime.restart_policy.window.as_secs(),
4078            "health restart budget exhausted; disabling module"
4079        );
4080        let stop_notice = begin_forwarding_drain_if_configured(
4081            spec,
4082            runtime,
4083            registry,
4084            snapshot,
4085            Some(false),
4086            RouteCloseReason::Disable,
4087        )
4088        .await?;
4089        drain_optional_child(
4090            &spec.module_id,
4091            spec.protocol,
4092            stop_notice,
4093            registry,
4094            snapshot,
4095            &runtime.terminal_ring,
4096            &runtime.spawn_events,
4097            child,
4098            runtime.drain_timeout,
4099            ModuleState::Disabled,
4100            Some(false),
4101        )
4102        .await?;
4103        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4104        return Ok(());
4105    }
4106
4107    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4108    let mut restart_count = 0;
4109    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4110        restart_count = state.crash_restarts.len();
4111        state.state = ModuleState::Unresponsive;
4112        state.health.status = status;
4113        state.health.last_action = Some(HealthAction::Restart.to_string());
4114        state.health.last_action_ms = Some(now_ms);
4115    })?;
4116    warn!(
4117        module_id = %spec.module_id,
4118        status = ?status,
4119        detail,
4120        restart_count,
4121        restart_in_window = schedule.restart_in_window,
4122        delay_ms = schedule.delay.as_millis() as u64,
4123        "health-triggered module restart"
4124    );
4125
4126    let stop_notice = begin_forwarding_drain_if_configured(
4127        spec,
4128        runtime,
4129        registry,
4130        snapshot,
4131        Some(true),
4132        RouteCloseReason::Restart,
4133    )
4134    .await?;
4135    drain_optional_child(
4136        &spec.module_id,
4137        spec.protocol,
4138        stop_notice,
4139        registry,
4140        snapshot,
4141        &runtime.terminal_ring,
4142        &runtime.spawn_events,
4143        child,
4144        runtime.drain_timeout,
4145        ModuleState::Restarting,
4146        Some(true),
4147    )
4148    .await?;
4149    sleep(schedule.delay).await;
4150    // The backoff may have outlasted the restart it was counting down to: an
4151    // operator disable or drain in between moves the snapshot out of
4152    // `Restarting`, and that stop must win over this respawn.
4153    if !respawn_still_pending(snapshot) {
4154        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4155        return Ok(());
4156    }
4157    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
4158    match spawn_and_mark_running(spec, runtime, snapshot) {
4159        Ok(next_child) => {
4160            *child = Some(next_child);
4161            Ok(())
4162        }
4163        Err(err) => {
4164            fail_snapshot(snapshot, Some(&spec.module_id), None);
4165            process_liveness.untrack_if_current(&spec.module_id, snapshot);
4166            *child = None;
4167            Err(err)
4168        }
4169    }
4170}
4171
4172fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4173    let _ = update_snapshot(snapshot, Some(module_id), |state| {
4174        state.health.last_action = Some(action);
4175        state.health.last_action_ms = Some(now_ms);
4176    });
4177}
4178
4179fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4180    match status {
4181        HealthStatus::Ok => SupervisorHealthStatus::Ok,
4182        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4183        HealthStatus::Failing => SupervisorHealthStatus::Failing,
4184    }
4185}
4186
4187/// Caps the metrics blob stored in the cached supervisor snapshot, which is
4188/// returned to every `supervisor.list` and `supervisor.health` caller.
4189///
4190/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
4191/// path: that request exists to return a module's complete metrics object, and
4192/// `ck health <module-id>` documents it as the way to see what the cached view
4193/// truncates. The asymmetry is the feature.
4194///
4195/// So a new caller must decide which side it is on rather than assume the cap is
4196/// universal. Reaching for it on a fresh-probe path would silently reintroduce
4197/// the truncation that path exists to avoid.
4198fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4199    let metrics = metrics?;
4200    match serde_json::to_vec(&metrics) {
4201        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4202            "truncated": true,
4203            "original_bytes": encoded.len(),
4204        })),
4205        Ok(_) | Err(_) => Some(metrics),
4206    }
4207}
4208
4209/// Spread health probes so a fleet-wide restart does not converge them.
4210///
4211/// The delay is derived from the module id and probe index rather than a random
4212/// source, so it is deterministic per module: a module keeps its own offset
4213/// across daemon restarts instead of re-rolling into a collision.
4214fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4215    if cadence.is_zero() {
4216        return Duration::ZERO;
4217    }
4218    let cadence_ms = cadence.as_millis() as u64;
4219    // This early return is REDUNDANT, deliberately, and a mutation run will show
4220    // it surviving removal. Recording why here so the next person to notice does
4221    // not have to re-derive it:
4222    //
4223    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
4224    //   a zero cadence and builds the Duration from whole milliseconds, so a
4225    //   sub-millisecond cadence cannot come from config.
4226    // - Even if reached it changes no answer. The `.max(1)` below makes the span
4227    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
4228    //   -- exactly what this returns.
4229    //
4230    // Kept as a guard against a future widening of the config parser (accepting
4231    // microseconds, say), which would make the sub-millisecond case reachable.
4232    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
4233    // divides by zero. Remove this and nothing changes.
4234    if cadence_ms == 0 {
4235        return cadence;
4236    }
4237    // Note that this never returns less than one cadence, including for the FIRST
4238    // probe. So a freshly registered module reports health `unknown` for a full
4239    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
4240    // ready to answer.
4241    //
4242    // That is a property of the supervisor's schedule, not of any module: an
4243    // operator watching a restart sees `unknown` and cannot tell it from a module
4244    // that is slow to warm. Measured on two unrelated modules, both flipping to
4245    // `ok` between 22s and 32s after restart.
4246    //
4247    // Left as-is because spreading the first probe is what keeps a fleet-wide
4248    // restart from firing fourteen simultaneous probes into a cold machine. The
4249    // alternative -- probe at t+0 and jitter only from the second onward -- trades
4250    // that thundering herd for a faster first reading.
4251    let jitter_span = (cadence_ms / 10).max(1);
4252    let hash = module_id.as_bytes().iter().fold(
4253        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
4254        |acc, byte| {
4255            acc.wrapping_mul(1099511628211)
4256                .wrapping_add(u64::from(*byte))
4257        },
4258    );
4259    cadence + Duration::from_millis(hash % jitter_span)
4260}
4261
4262#[cfg(test)]
4263mod tests {
4264    use super::*;
4265
4266    #[test]
4267    fn readding_a_module_clears_its_rescan_removal_tombstone() {
4268        let handle = SupervisorHandle::new();
4269        let module_id = "readded-tombstone";
4270        handle.record_rescan_removal(module_id);
4271        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
4272
4273        handle.apply_identity_configuration(&ModuleSpec {
4274            module_id: module_id.to_string(),
4275            program: PathBuf::from("/test/module"),
4276            args: Vec::new(),
4277            env: Vec::new(),
4278            reserved: false,
4279            reserved_prefixes: Vec::new(),
4280            protocol: ModuleProtocol::Subc,
4281            overlap: Default::default(),
4282        });
4283
4284        assert!(
4285            handle.removal_tombstone_age_ms(module_id).is_none(),
4286            "a re-added module must not retain a stale removal tombstone"
4287        );
4288    }
4289
4290    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
4291        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
4292        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
4293            snapshot.process_alive = true;
4294            snapshot.pid = Some(41);
4295            snapshot.spawned_at_ms = Some(42);
4296            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
4297            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
4298                device: 43,
4299                inode: 44,
4300            });
4301        })
4302        .unwrap();
4303        snapshot
4304    }
4305
4306    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
4307        let snapshot = lock_snapshot(snapshot).unwrap();
4308        assert!(!snapshot.process_alive);
4309        assert_eq!(snapshot.pid, None);
4310        assert_eq!(snapshot.spawned_at_ms, None);
4311        assert_eq!(snapshot.spawned_from, None);
4312        assert_eq!(snapshot.spawned_file_identity, None);
4313    }
4314
4315    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4316    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
4317        let supervisor = Supervisor::default();
4318        let mut runtime = supervisor.runtime_config();
4319        runtime.test_seed_stale_facts_before_enable_spawn = true;
4320        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
4321        let mut child = None;
4322        let spec = ModuleSpec {
4323            module_id: "failed-enable-clears-facts".to_string(),
4324            program: PathBuf::from("/definitely/missing/failed-enable-module"),
4325            args: Vec::new(),
4326            env: Vec::new(),
4327            reserved: false,
4328            reserved_prefixes: Vec::new(),
4329            protocol: ModuleProtocol::Subc,
4330            overlap: Default::default(),
4331        };
4332
4333        let result = set_child_enabled(
4334            &spec,
4335            &runtime,
4336            &supervisor.registry,
4337            &supervisor.process_liveness,
4338            &snapshot,
4339            &mut child,
4340            true,
4341        )
4342        .await;
4343
4344        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
4345        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4346        assert_snapshot_process_facts_cleared(&snapshot);
4347    }
4348
4349    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4350    async fn failed_reload_spawn_clears_current_process_facts() {
4351        let supervisor = Supervisor::default();
4352        let mut runtime = supervisor.runtime_config();
4353        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
4354        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4355        let mut child = None;
4356        let spec = ModuleSpec {
4357            module_id: "failed-reload-clears-facts".to_string(),
4358            program: PathBuf::from("/unused/failed-reload-module"),
4359            args: Vec::new(),
4360            env: Vec::new(),
4361            reserved: false,
4362            reserved_prefixes: Vec::new(),
4363            protocol: ModuleProtocol::Subc,
4364            overlap: Default::default(),
4365        };
4366
4367        let result = handle_reload_spawn_failure(
4368            &spec,
4369            &runtime,
4370            &supervisor.process_liveness,
4371            &snapshot,
4372            &mut child,
4373            "forced reload spawn failure".to_string(),
4374        )
4375        .await;
4376
4377        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
4378        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4379        assert_snapshot_process_facts_cleared(&snapshot);
4380    }
4381
4382    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4383    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
4384        let supervisor = Supervisor::default();
4385        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4386        let module = supervisor.supervised_module(
4387            ModuleSpec {
4388                module_id: "drop-clears-facts".to_string(),
4389                program: PathBuf::from("/unused/drop-module"),
4390                args: Vec::new(),
4391                env: Vec::new(),
4392                reserved: false,
4393                reserved_prefixes: Vec::new(),
4394                protocol: ModuleProtocol::Subc,
4395                overlap: Default::default(),
4396            },
4397            supervisor.runtime_config(),
4398            Arc::clone(&snapshot),
4399            None,
4400        );
4401        assert!(!module
4402            .inner
4403            .monitor
4404            .lock()
4405            .unwrap()
4406            .as_ref()
4407            .unwrap()
4408            .is_finished());
4409
4410        drop(module);
4411
4412        assert_eq!(
4413            lock_snapshot(&snapshot).unwrap().state,
4414            ModuleState::Stopped
4415        );
4416        assert_snapshot_process_facts_cleared(&snapshot);
4417    }
4418
4419    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4420    async fn configuration_update_does_not_replace_captured_running_process_facts() {
4421        let supervisor = Supervisor::default();
4422        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4423        let initial = ModuleSpec {
4424            module_id: "rescan-preserves-spawn-facts".to_string(),
4425            program: PathBuf::from("/spawned/module"),
4426            args: Vec::new(),
4427            env: Vec::new(),
4428            reserved: false,
4429            reserved_prefixes: Vec::new(),
4430            protocol: ModuleProtocol::Subc,
4431            overlap: Default::default(),
4432        };
4433        let module = supervisor.supervised_module(
4434            initial.clone(),
4435            supervisor.runtime_config(),
4436            snapshot,
4437            None,
4438        );
4439        let before = module.status().unwrap();
4440        let mut replacement = initial;
4441        replacement.program = PathBuf::from("/rescanned/replacement-module");
4442
4443        module
4444            .update_configuration(replacement, HealthConfig::default(), None)
4445            .await
4446            .unwrap();
4447
4448        let after = module.status().unwrap();
4449        assert_eq!(after.pid, before.pid);
4450        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
4451        assert_eq!(after.spawned_from, before.spawned_from);
4452        drop(module);
4453    }
4454}
4455
4456fn unix_ms_now() -> u64 {
4457    SystemTime::now()
4458        .duration_since(UNIX_EPOCH)
4459        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
4460        .unwrap_or(0)
4461}
4462
4463async fn supervise_loop(
4464    mut spec: ModuleSpec,
4465    mut runtime: SupervisorRuntimeConfig,
4466    registry: Arc<Registry>,
4467    process_liveness: Arc<SupervisorProcessLiveness>,
4468    snapshot: SharedSnapshot,
4469    mut child: Option<SupervisedChild>,
4470    mut commands: mpsc::Receiver<SupervisorCommand>,
4471) {
4472    let mut health_probe = HealthProbeRuntime::default();
4473    // Deadline of the crash respawn whose backoff is currently elapsing. While
4474    // it is set the loop serves commands instead of sleeping inside the exit
4475    // arm, so a disable or drain lands immediately and cancels the respawn.
4476    let mut pending_respawn: Option<Instant> = None;
4477    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
4478    // before anything else so a stop that interrupted a swap runs at once.
4479    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
4480    loop {
4481        if let Some(command) = requeued.pop_front() {
4482            if !handle_supervisor_command(
4483                command,
4484                &mut spec,
4485                &mut runtime,
4486                &registry,
4487                &process_liveness,
4488                &snapshot,
4489                &mut child,
4490                &mut commands,
4491                &mut requeued,
4492            )
4493            .await
4494            {
4495                return;
4496            }
4497            if child.is_some() || !respawn_still_pending(&snapshot) {
4498                pending_respawn = None;
4499            }
4500            continue;
4501        }
4502        if child.is_some() {
4503            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
4504            let probe_sleep = sleep(health_probe.wake_after());
4505            tokio::pin!(probe_sleep);
4506            let active_child = child.as_mut().expect("child checked above");
4507            tokio::select! {
4508                wait_result = active_child.wait() => {
4509                    // Every arm below that gives up on the CHILD must keep the
4510                    // supervision task itself alive (child = None, loop
4511                    // continues into command-serving mode). Returning here
4512                    // closes the command channel, which makes the module
4513                    // permanently unrestartable in-band: a clean child exit
4514                    // of an enabled module once wedged the fleet this way
4515                    // ('supervisor command channel is closed') and required a
4516                    // full daemon restart to recover.
4517                    let exit_report = match wait_result {
4518                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
4519                        Err(err) => {
4520                            active_child.drain_stderr(&spec.module_id).await;
4521                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4522                            // Every other exit path (on_child_exit's Clean/Crash arms,
4523                            // the reload-registration-failure path) records a terminal
4524                            // before moving on. Without one here, a module whose wait()
4525                            // itself errored (e.g. already reaped) leaves no terminal
4526                            // record at all -- an empty ring reads as "nothing died".
4527                            record_wait_error_terminal(
4528                                &spec.module_id,
4529                                &runtime.terminal_ring,
4530                                &runtime.spawn_events,
4531                            );
4532                            untrack_if_registration_released(
4533                                &process_liveness,
4534                                &registry,
4535                                &spec.module_id,
4536                                &snapshot,
4537                            );
4538                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
4539                            child = None;
4540                            continue;
4541                        }
4542                    };
4543                    active_child.drain_stderr(&spec.module_id).await;
4544
4545                    let next = on_child_exit(
4546                        &spec,
4547                        runtime.restart_policy,
4548                        &registry,
4549                        &snapshot,
4550                        &runtime.terminal_ring,
4551                        &runtime.spawn_events,
4552                        &runtime.child_roster,
4553                        exit_report,
4554                    ).await;
4555                    // The exit is recorded, so a daemon shutdown may stop
4556                    // waiting for this child (see `SupervisedChild::wait`).
4557                    active_child.release_roster();
4558                    match next {
4559                        NextAction::Stop { registration_released } => {
4560                            if registration_released {
4561                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4562                            }
4563                            child = None;
4564                        }
4565                        NextAction::Restart { schedule } => {
4566                            let delay = schedule.map_or(
4567                                runtime.restart_policy.delay_for_restart(0),
4568                                |schedule| schedule.delay,
4569                            );
4570                            if let Some(schedule) = schedule {
4571                                log_crash_respawn(&spec.module_id, schedule);
4572                            }
4573                            // The exited child is fully recorded at this point,
4574                            // so release it and count the backoff down in the
4575                            // command-serving branch below rather than sleeping
4576                            // here: commands cannot be received from inside this
4577                            // select arm, and an operator disable or drain that
4578                            // arrives during the backoff must cancel the pending
4579                            // respawn instead of waiting for it to spawn first.
4580                            child = None;
4581                            pending_respawn = Some(Instant::now() + delay);
4582                        }
4583                    }
4584                }
4585                command = commands.recv() => {
4586                    let Some(command) = command else {
4587                        return;
4588                    };
4589                    if !handle_supervisor_command(
4590                        command,
4591                        &mut spec,
4592                        &mut runtime,
4593                        &registry,
4594                        &process_liveness,
4595                        &snapshot,
4596                        &mut child,
4597                        &mut commands,
4598                        &mut requeued,
4599                    ).await {
4600                        return;
4601                    }
4602                }
4603                _ = &mut probe_sleep => {
4604                    if health_probe.due() {
4605                        run_health_probe_cycle(
4606                            &spec,
4607                            &runtime,
4608                            &registry,
4609                            &process_liveness,
4610                            &snapshot,
4611                            &mut child,
4612                        ).await;
4613                        if child.is_some() {
4614                            health_probe.schedule_next(&spec, runtime.health.cadence);
4615                        }
4616                    }
4617                }
4618            }
4619        } else if let Some(deadline) = pending_respawn {
4620            tokio::select! {
4621                _ = sleep_until(deadline) => {
4622                    pending_respawn = None;
4623                    // A command handled below while the backoff elapsed may
4624                    // have stopped the module; never respawn past an operator's
4625                    // disable or drain.
4626                    if !respawn_still_pending(&snapshot) {
4627                        continue;
4628                    }
4629                    // The daemon began shutting down during the backoff: the
4630                    // spawn would be refused anyway, and refusing it here
4631                    // leaves the module stopped instead of reporting a
4632                    // failed restart.
4633                    if runtime.child_roster.is_closed() {
4634                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4635                            state.state = ModuleState::Stopped;
4636                        });
4637                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
4638                        continue;
4639                    }
4640                    if let Err(err) = wait_for_registration_release(
4641                        &registry,
4642                        &spec.module_id,
4643                        REGISTRY_RELEASE_TIMEOUT,
4644                    ).await {
4645                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
4646                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
4647                        continue;
4648                    }
4649
4650                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
4651                        Ok(next_child) => {
4652                            child = Some(next_child);
4653                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
4654                        }
4655                        Err(err) => {
4656                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4657                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4658                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
4659                        }
4660                    }
4661                }
4662                command = commands.recv() => {
4663                    let Some(command) = command else {
4664                        return;
4665                    };
4666                    if !handle_supervisor_command(
4667                        command,
4668                        &mut spec,
4669                        &mut runtime,
4670                        &registry,
4671                        &process_liveness,
4672                        &snapshot,
4673                        &mut child,
4674                        &mut commands,
4675                        &mut requeued,
4676                    ).await {
4677                        return;
4678                    }
4679                    // Reconcile the pending respawn with what the command did:
4680                    // a restart or reload has already spawned a fresh child,
4681                    // while a disable or drain moved the snapshot out of the
4682                    // state the respawn was counting down from.
4683                    if child.is_some() || !respawn_still_pending(&snapshot) {
4684                        pending_respawn = None;
4685                    }
4686                }
4687            }
4688        } else {
4689            let Some(command) = commands.recv().await else {
4690                return;
4691            };
4692            if !handle_supervisor_command(
4693                command,
4694                &mut spec,
4695                &mut runtime,
4696                &registry,
4697                &process_liveness,
4698                &snapshot,
4699                &mut child,
4700                &mut commands,
4701                &mut requeued,
4702            )
4703            .await
4704            {
4705                return;
4706            }
4707        }
4708    }
4709}
4710
4711fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
4712    info!(
4713        module_id,
4714        restart_in_window = schedule.restart_in_window,
4715        delay_ms = schedule.delay.as_millis() as u64,
4716        "respawning after crash"
4717    );
4718}
4719
4720/// Whether the respawn a backoff was counting down to is still wanted. A
4721/// disable or drain handled while the backoff elapsed moves the snapshot out
4722/// of `Restarting`, and the operator's stop must win over the pending respawn,
4723/// so every sleep-then-spawn path re-validates against the live snapshot
4724/// instead of assuming the state it left behind still holds.
4725fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
4726    matches!(
4727        lock_snapshot(snapshot),
4728        Ok(state) if state.enabled && state.state == ModuleState::Restarting
4729    )
4730}
4731
4732enum NextAction {
4733    Stop {
4734        registration_released: bool,
4735    },
4736    Restart {
4737        schedule: Option<CrashRestartSchedule>,
4738    },
4739}
4740
4741#[allow(clippy::too_many_arguments)]
4742async fn handle_supervisor_command(
4743    command: SupervisorCommand,
4744    spec: &mut ModuleSpec,
4745    runtime: &mut SupervisorRuntimeConfig,
4746    registry: &Registry,
4747    process_liveness: &SupervisorProcessLiveness,
4748    snapshot: &SharedSnapshot,
4749    child: &mut Option<SupervisedChild>,
4750    commands: &mut mpsc::Receiver<SupervisorCommand>,
4751    requeued: &mut VecDeque<SupervisorCommand>,
4752) -> bool {
4753    match command {
4754        SupervisorCommand::Drain { reply } => {
4755            // A plain stop runs no forwarding drain, so nothing reaches the
4756            // module over its connection before the wait: ask by signal.
4757            let result = drain_optional_child(
4758                &spec.module_id,
4759                spec.protocol,
4760                StopNotice::NotSent,
4761                registry,
4762                snapshot,
4763                &runtime.terminal_ring,
4764                &runtime.spawn_events,
4765                child,
4766                runtime.drain_timeout,
4767                ModuleState::Stopped,
4768                None,
4769            )
4770            .await;
4771            let registration_released = result.is_ok();
4772            let _ = reply.send(result);
4773            if registration_released {
4774                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4775            }
4776            false
4777        }
4778        SupervisorCommand::Retire { reply } => {
4779            let result = async {
4780                let stop_notice = begin_forwarding_drain_if_configured(
4781                    spec,
4782                    runtime,
4783                    registry,
4784                    snapshot,
4785                    None,
4786                    RouteCloseReason::Disable,
4787                )
4788                .await?;
4789                drain_optional_child(
4790                    &spec.module_id,
4791                    spec.protocol,
4792                    stop_notice,
4793                    registry,
4794                    snapshot,
4795                    &runtime.terminal_ring,
4796                    &runtime.spawn_events,
4797                    child,
4798                    runtime.drain_timeout,
4799                    ModuleState::Stopped,
4800                    None,
4801                )
4802                .await
4803            }
4804            .await;
4805            let registration_released = result.is_ok();
4806            let _ = reply.send(result);
4807            if registration_released {
4808                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4809            }
4810            false
4811        }
4812        SupervisorCommand::Restart {
4813            drain_timeout_ms,
4814            received_at_generation,
4815            queued_at,
4816            reply,
4817        } => {
4818            // Without this line a restart that waited in the queue (behind a
4819            // health probe cycle or another command) was invisible: the log
4820            // showed only the drain timing out, minutes after the operator's call.
4821            info!(
4822                module_id = %spec.module_id,
4823                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
4824                "restart command dequeued"
4825            );
4826            // ACK AT INITIATION, not completion. The blocking form deadlocked any
4827            // caller whose own request lane rides the module being restarted: the
4828            // caller's in-flight request keeps the drain from quiescing, the drain
4829            // keeps the restart from completing, and the completion keeps the reply
4830            // from releasing the caller — so the drain always timed out and cut the
4831            // initiator with a GOODBYE, even on a healthy module. Replying once the
4832            // restart is validated lets a self-lane caller settle, which is exactly
4833            // what makes the drain succeed. Completion is observable via
4834            // supervisor.list / module status; a post-ack failure lands the module
4835            // in a visible terminal state below rather than in a reply nobody can
4836            // receive.
4837            let validation = match lock_snapshot(snapshot) {
4838                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
4839                    module_id: spec.module_id.clone(),
4840                }),
4841                Ok(_) => Ok(()),
4842                Err(err) => Err(err),
4843            };
4844            let initiated = validation.is_ok();
4845            let _ = reply.send(validation);
4846            // A restart asks for a fresh process. Commands run one at a time,
4847            // so a restart queued behind another restart (two operator calls
4848            // in quick succession) is dequeued the moment the first one has
4849            // spawned its replacement -- before that process has sent HELLO.
4850            // Running it would drain and kill the process the first restart
4851            // just produced, which is the opposite of what both callers asked
4852            // for. If a process spawned after this request was received is
4853            // still supervised, the request is already satisfied. Not when the
4854            // configuration changed since that spawn: then the newer process
4855            // predates the spec this restart may exist to apply.
4856            let satisfied_by_generation = if initiated && child.is_some() {
4857                lock_snapshot(snapshot).ok().and_then(|state| {
4858                    (state.spawn_generation > received_at_generation
4859                        && !state.configuration_updated_since_spawn)
4860                        .then_some(state.spawn_generation)
4861                })
4862            } else {
4863                None
4864            };
4865            if let Some(generation) = satisfied_by_generation {
4866                info!(
4867                    module_id = %spec.module_id,
4868                    received_at_generation,
4869                    "restart already satisfied by generation {generation}; not restarting again"
4870                );
4871            } else if initiated {
4872                // Precedence: this restart's operator override, else the module's
4873                // configured budget (already resolved into the runtime).
4874                let drain_timeout = drain_timeout_ms
4875                    .map(Duration::from_millis)
4876                    .unwrap_or(runtime.drain_timeout);
4877                if let Err(err) = restart_child(
4878                    spec,
4879                    runtime,
4880                    registry,
4881                    process_liveness,
4882                    snapshot,
4883                    child,
4884                    drain_timeout,
4885                )
4886                .await
4887                {
4888                    warn!(
4889                        module_id = %spec.module_id,
4890                        error = %err,
4891                        "operator restart failed after initiation ack; module state carries the outcome"
4892                    );
4893                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4894                        state.state = ModuleState::Failed;
4895                        clear_current_process_facts(state);
4896                    });
4897                }
4898            }
4899            true
4900        }
4901        SupervisorCommand::Reload { reply } => {
4902            let result =
4903                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
4904            let _ = reply.send(result);
4905            true
4906        }
4907        SupervisorCommand::SetEnabled { enabled, reply } => {
4908            let result = set_child_enabled(
4909                spec,
4910                runtime,
4911                registry,
4912                process_liveness,
4913                snapshot,
4914                child,
4915                enabled,
4916            )
4917            .await;
4918            let _ = reply.send(result);
4919            true
4920        }
4921        SupervisorCommand::UpdateConfiguration {
4922            spec: next_spec,
4923            health,
4924            drain_timeout_ms,
4925            reply,
4926        } => {
4927            if let Some(handle) = &runtime.supervisor_handle {
4928                handle.apply_identity_configuration(&next_spec);
4929            }
4930            *spec = next_spec;
4931            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4932                state.configuration_updated_since_spawn = true;
4933            });
4934            runtime.health = health;
4935            runtime.drain_timeout = drain_timeout_ms
4936                .map(Duration::from_millis)
4937                .unwrap_or(runtime.default_drain_timeout);
4938            *runtime
4939                .effective_drain_timeout
4940                .lock()
4941                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
4942            let _ = reply.send(());
4943            true
4944        }
4945        SupervisorCommand::Swap {
4946            ready_timeout,
4947            reply,
4948        } => {
4949            let end = swap::run_swap(
4950                spec,
4951                runtime,
4952                registry,
4953                process_liveness,
4954                snapshot,
4955                child,
4956                commands,
4957                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
4958                reply,
4959            )
4960            .await;
4961            requeued.extend(end.requeue);
4962            true
4963        }
4964    }
4965}
4966
4967async fn restart_child(
4968    spec: &ModuleSpec,
4969    runtime: &SupervisorRuntimeConfig,
4970    registry: &Registry,
4971    process_liveness: &SupervisorProcessLiveness,
4972    snapshot: &SharedSnapshot,
4973    child: &mut Option<SupervisedChild>,
4974    drain_timeout: Duration,
4975) -> Result<(), SuperviseError> {
4976    // Restart cycles a running module; it must not silently start a disabled one.
4977    if !lock_snapshot(snapshot)?.enabled {
4978        return Err(SuperviseError::Disabled {
4979            module_id: spec.module_id.clone(),
4980        });
4981    }
4982    let stop_notice = begin_forwarding_drain_with_timeout(
4983        spec,
4984        runtime,
4985        registry,
4986        snapshot,
4987        None,
4988        RouteCloseReason::Restart,
4989        drain_timeout,
4990    )
4991    .await?;
4992
4993    if child.is_some() {
4994        drain_optional_child(
4995            &spec.module_id,
4996            spec.protocol,
4997            stop_notice,
4998            registry,
4999            snapshot,
5000            &runtime.terminal_ring,
5001            &runtime.spawn_events,
5002            child,
5003            drain_timeout,
5004            ModuleState::Restarting,
5005            Some(true),
5006        )
5007        .await?;
5008    } else {
5009        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5010            state.enabled = true;
5011            state.state = ModuleState::Restarting;
5012            clear_current_process_facts(state);
5013        })?;
5014        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5015    }
5016
5017    reset_restart_count(snapshot, &spec.module_id)?;
5018    sleep(runtime.restart_policy.backoff).await;
5019    // A disable or drain that landed during the backoff cancels this respawn:
5020    // the operator's stop must win over the restart the sleep counted down to.
5021    if !respawn_still_pending(snapshot) {
5022        process_liveness.untrack_if_current(&spec.module_id, snapshot);
5023        return Ok(());
5024    }
5025    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5026    // Mirror health_restart_child's spawn-failure handling: of the four
5027    // spawn-failure sites this was the only one that propagated with the
5028    // snapshot still reading `Restarting` -- neither running nor failed, and
5029    // unrevivable by `set_enabled(true)` (issue #34). `Failed` is the state the
5030    // operator can see and heal.
5031    match spawn_and_mark_running(spec, runtime, snapshot) {
5032        Ok(next_child) => {
5033            *child = Some(next_child);
5034            debug!(module_id = %spec.module_id, "supervised module restarted by operator request");
5035            Ok(())
5036        }
5037        Err(err) => {
5038            fail_snapshot(snapshot, Some(&spec.module_id), None);
5039            process_liveness.untrack_if_current(&spec.module_id, snapshot);
5040            *child = None;
5041            Err(err)
5042        }
5043    }
5044}
5045
5046async fn reload_child(
5047    spec: &ModuleSpec,
5048    runtime: &SupervisorRuntimeConfig,
5049    registry: &Registry,
5050    process_liveness: &SupervisorProcessLiveness,
5051    snapshot: &SharedSnapshot,
5052    child: &mut Option<SupervisedChild>,
5053) -> Result<(), SuperviseError> {
5054    // Reload cycles a running module; it must not silently start a disabled one.
5055    if !lock_snapshot(snapshot)?.enabled {
5056        return Err(SuperviseError::Disabled {
5057            module_id: spec.module_id.clone(),
5058        });
5059    }
5060    let stop_notice = begin_forwarding_drain(
5061        spec,
5062        runtime,
5063        registry,
5064        snapshot,
5065        Some(true),
5066        RouteCloseReason::Reload,
5067    )
5068    .await?;
5069
5070    if child.is_some() {
5071        drain_optional_child(
5072            &spec.module_id,
5073            spec.protocol,
5074            stop_notice,
5075            registry,
5076            snapshot,
5077            &runtime.terminal_ring,
5078            &runtime.spawn_events,
5079            child,
5080            runtime.drain_timeout,
5081            ModuleState::Restarting,
5082            Some(true),
5083        )
5084        .await?;
5085    } else {
5086        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5087            state.enabled = true;
5088            state.state = ModuleState::Restarting;
5089            clear_current_process_facts(state);
5090        })?;
5091        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5092    }
5093
5094    reset_restart_count(snapshot, &spec.module_id)?;
5095    sleep(runtime.restart_policy.backoff).await;
5096    // A disable or drain that landed during the backoff cancels this respawn:
5097    // the operator's stop must win over the restart the sleep counted down to.
5098    if !respawn_still_pending(snapshot) {
5099        process_liveness.untrack_if_current(&spec.module_id, snapshot);
5100        return Ok(());
5101    }
5102    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5103    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5104        Ok(next_child) => next_child,
5105        Err(err) => {
5106            return handle_reload_spawn_failure(
5107                spec,
5108                runtime,
5109                process_liveness,
5110                snapshot,
5111                child,
5112                format!("new child failed to spawn: {err}"),
5113            )
5114            .await;
5115        }
5116    };
5117    *child = Some(next_child);
5118
5119    let wait_outcome = {
5120        let active_child = child.as_mut().expect("new reload child was just stored");
5121        wait_for_registration_after_reload(
5122            registry,
5123            &spec.module_id,
5124            snapshot,
5125            active_child,
5126            REGISTRY_RELEASE_TIMEOUT,
5127        )
5128        .await?
5129    };
5130
5131    match wait_outcome {
5132        RegistrationWaitOutcome::Registered => {
5133            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
5134            Ok(())
5135        }
5136        RegistrationWaitOutcome::Exited(exit_report) => {
5137            if let Some(active_child) = child.as_mut() {
5138                active_child.drain_stderr(&spec.module_id).await;
5139            }
5140            *child = None;
5141            handle_reload_child_registration_failure(
5142                spec,
5143                runtime,
5144                registry,
5145                process_liveness,
5146                snapshot,
5147                child,
5148                ReloadRegistrationFailure {
5149                    exit_report: registration_failure_exit_report(exit_report),
5150                    reason: "new child exited before registering".to_string(),
5151                },
5152            )
5153            .await
5154        }
5155        RegistrationWaitOutcome::TimedOut => {
5156            let mut timed_out_child = child
5157                .take()
5158                .expect("timed-out reload child is still running");
5159            timed_out_child
5160                .start_kill()
5161                .map_err(|source| SuperviseError::Kill {
5162                    module_id: spec.module_id.clone(),
5163                    source,
5164                })?;
5165            let status = timed_out_child
5166                .wait()
5167                .await
5168                .map_err(|source| SuperviseError::Wait {
5169                    module_id: spec.module_id.clone(),
5170                    source,
5171                })?;
5172            timed_out_child.drain_stderr(&spec.module_id).await;
5173            handle_reload_child_registration_failure(
5174                spec,
5175                runtime,
5176                registry,
5177                process_liveness,
5178                snapshot,
5179                child,
5180                ReloadRegistrationFailure {
5181                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
5182                        snapshot,
5183                        &timed_out_child,
5184                        &status,
5185                    )),
5186                    reason: format!(
5187                        "new child did not register within {:?}",
5188                        REGISTRY_RELEASE_TIMEOUT
5189                    ),
5190                },
5191            )
5192            .await
5193        }
5194    }
5195}
5196
5197async fn set_child_enabled(
5198    spec: &ModuleSpec,
5199    runtime: &SupervisorRuntimeConfig,
5200    registry: &Registry,
5201    process_liveness: &SupervisorProcessLiveness,
5202    snapshot: &SharedSnapshot,
5203    child: &mut Option<SupervisedChild>,
5204    enabled: bool,
5205) -> Result<bool, SuperviseError> {
5206    let (current_enabled, current_state) = {
5207        let state = lock_snapshot(snapshot)?;
5208        (state.enabled, state.state)
5209    };
5210    // `start` (enable on an already-enabled module) heals TERMINAL states instead
5211    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
5212    // clean (Stopped) has no live process and no other in-band recovery — the
5213    // operator's start is the explicit recovery act and resets the budget. Without
5214    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
5215    // which the 2026-07-14 aft outage proved is a trap when the failed module is
5216    // the one providing every agent's shell.
5217    let revive_terminal = enabled
5218        && current_enabled
5219        && child.is_none()
5220        && matches!(current_state, ModuleState::Failed | ModuleState::Stopped);
5221    if current_enabled == enabled && !revive_terminal {
5222        return Ok(false);
5223    }
5224
5225    if enabled {
5226        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5227            state.enabled = true;
5228            state.state = ModuleState::Starting;
5229            clear_current_process_facts(state);
5230        })?;
5231        #[cfg(test)]
5232        if runtime.test_seed_stale_facts_before_enable_spawn {
5233            update_snapshot(snapshot, Some(&spec.module_id), |state| {
5234                state.process_alive = true;
5235                state.pid = Some(41);
5236                state.spawned_at_ms = Some(42);
5237                state.spawned_from = Some(PathBuf::from("/spawned/module"));
5238                state.spawned_file_identity = Some(SpawnedFileIdentity {
5239                    device: 43,
5240                    inode: 44,
5241                });
5242            })?;
5243        }
5244        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5245        reset_restart_count(snapshot, &spec.module_id)?;
5246        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5247        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5248            Ok(next_child) => next_child,
5249            Err(err) => {
5250                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5251                    state.state = ModuleState::Failed;
5252                    clear_current_process_facts(state);
5253                }) {
5254                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
5255                }
5256                process_liveness.untrack_if_current(&spec.module_id, snapshot);
5257                return Err(err);
5258            }
5259        };
5260        *child = Some(next_child);
5261        debug!(module_id = %spec.module_id, "supervised module enabled");
5262        Ok(true)
5263    } else {
5264        let stop_notice = begin_forwarding_drain_if_configured(
5265            spec,
5266            runtime,
5267            registry,
5268            snapshot,
5269            Some(false),
5270            RouteCloseReason::Disable,
5271        )
5272        .await?;
5273        drain_optional_child(
5274            &spec.module_id,
5275            spec.protocol,
5276            stop_notice,
5277            registry,
5278            snapshot,
5279            &runtime.terminal_ring,
5280            &runtime.spawn_events,
5281            child,
5282            runtime.drain_timeout,
5283            ModuleState::Disabled,
5284            Some(false),
5285        )
5286        .await?;
5287        debug!(module_id = %spec.module_id, "supervised module disabled");
5288        Ok(true)
5289    }
5290}
5291
5292#[allow(clippy::too_many_arguments)]
5293async fn on_child_exit(
5294    spec: &ModuleSpec,
5295    policy: RestartPolicy,
5296    registry: &Registry,
5297    snapshot: &SharedSnapshot,
5298    terminal_ring: &Arc<Mutex<TerminalRing>>,
5299    spawn_events: &SpawnEventFeed,
5300    roster: &ChildRoster,
5301    exit_report: ExitReport,
5302) -> NextAction {
5303    // Once the daemon has begun shutting down, no exit is a crash to recover
5304    // from: the module is exiting because the daemon is going away (EOF on its
5305    // connection, or a service manager signalling the whole cgroup). Record it
5306    // as such and never schedule a respawn, which would only start a process
5307    // for the shutdown to end again.
5308    if roster.is_closed() {
5309        return on_child_exit_during_daemon_shutdown(
5310            spec,
5311            registry,
5312            snapshot,
5313            terminal_ring,
5314            spawn_events,
5315            exit_report,
5316        )
5317        .await;
5318    }
5319    // Every stop the supervisor itself asks for (operator stop, disable,
5320    // restart, reload, swap, a health restart, a drain that runs out of budget)
5321    // takes the child out of the supervise loop and reaps it in
5322    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
5323    // that reaches this point was not requested by the daemon.
5324    //
5325    // For a subc-wire module a clean exit is still a stop: those modules are
5326    // written to re-raise SIGTERM, so a stray outside signal already reads as a
5327    // crash, and exiting 0 is a deliberate choice the module made. A
5328    // `protocol: "none"` module is a stock program we cannot change, and many
5329    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
5330    // stop would leave the module down for good after any stray signal, so it
5331    // goes through the crash path instead: it spends restart budget, respawns
5332    // with the crash backoff, and ends `failed` when the budget runs out.
5333    let unrequested_clean_exit_of_protocol_none =
5334        exit_report.kind == ExitKind::Clean && spec.protocol == ModuleProtocol::None;
5335    match exit_report.kind {
5336        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
5337            info!(
5338                module_id = %spec.module_id,
5339                exit_code = ?exit_report.code,
5340                exit_signal = ?exit_report.signal,
5341                "supervised module exited cleanly"
5342            );
5343            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5344                state.state = ModuleState::Stopped;
5345                clear_current_process_facts(state);
5346                state.last_exit = Some(exit_report.clone());
5347            }) {
5348                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
5349            }
5350            record_terminal(
5351                &spec.module_id,
5352                terminal_ring,
5353                spawn_events,
5354                &exit_report,
5355                TerminalDisposition::Stopped,
5356            );
5357            let registration_released = match wait_for_registration_release(
5358                registry,
5359                &spec.module_id,
5360                REGISTRY_RELEASE_TIMEOUT,
5361            )
5362            .await
5363            {
5364                Ok(()) => true,
5365                Err(err) => {
5366                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
5367                    false
5368                }
5369            };
5370            NextAction::Stop {
5371                registration_released,
5372            }
5373        }
5374        ExitKind::Clean | ExitKind::Crash => {
5375            if unrequested_clean_exit_of_protocol_none {
5376                warn!(
5377                    module_id = %spec.module_id,
5378                    exit_code = ?exit_report.code,
5379                    exit_signal = ?exit_report.signal,
5380                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
5381                );
5382            } else {
5383                warn!(
5384                    module_id = %spec.module_id,
5385                    exit_code = ?exit_report.code,
5386                    exit_signal = ?exit_report.signal,
5387                    "supervised module exited abnormally (crash)"
5388                );
5389            }
5390            let mut restart_schedule = None;
5391            let mut disposition = TerminalDisposition::Disabled;
5392            // Set only when the budget is what stopped the module, so the
5393            // terminal record says which limit was hit rather than leaving
5394            // `failed` to be read as "crashed once, badly".
5395            let mut disposition_detail = None;
5396            let now = Instant::now();
5397            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5398                clear_current_process_facts(state);
5399                state.last_exit = Some(exit_report.clone());
5400                if state.enabled {
5401                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
5402                        state.state = ModuleState::Restarting;
5403                        restart_schedule = Some(schedule);
5404                        disposition = TerminalDisposition::Restarting;
5405                    } else {
5406                        state.state = ModuleState::Failed;
5407                        disposition = TerminalDisposition::Failed;
5408                        disposition_detail = Some(policy.budget_exhausted_detail());
5409                    }
5410                } else {
5411                    state.state = ModuleState::Disabled;
5412                    disposition = TerminalDisposition::Disabled;
5413                }
5414            }) {
5415                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
5416                return NextAction::Stop {
5417                    registration_released: false,
5418                };
5419            }
5420            if disposition_detail.is_some() {
5421                // The window is in the message, not only in the fields: this line
5422                // is read in a scrollback where a bare `max_restarts=3` reads as a
5423                // lifetime cap and sends the operator looking for three crashes
5424                // that never happened together.
5425                error!(
5426                    module_id = %spec.module_id,
5427                    max_restarts = policy.max_restarts,
5428                    window_secs = policy.window.as_secs(),
5429                    "module stopped: {}",
5430                    policy.budget_exhausted_detail()
5431                );
5432            }
5433            record_terminal_with_detail(
5434                &spec.module_id,
5435                terminal_ring,
5436                spawn_events,
5437                &exit_report,
5438                disposition,
5439                disposition_detail,
5440            );
5441
5442            if let Some(schedule) = restart_schedule {
5443                NextAction::Restart {
5444                    schedule: Some(schedule),
5445                }
5446            } else {
5447                let registration_released = match wait_for_registration_release(
5448                    registry,
5449                    &spec.module_id,
5450                    REGISTRY_RELEASE_TIMEOUT,
5451                )
5452                .await
5453                {
5454                    Ok(()) => true,
5455                    Err(err) => {
5456                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
5457                        false
5458                    }
5459                };
5460                NextAction::Stop {
5461                    registration_released,
5462                }
5463            }
5464        }
5465        ExitKind::DeliberateSeverance => {
5466            warn!(
5467                module_id = %spec.module_id,
5468                exit_code = ?exit_report.code,
5469                exit_signal = ?exit_report.signal,
5470                "supervised module exited after deliberate connection severance"
5471            );
5472            let mut should_restart = false;
5473            let mut disposition = TerminalDisposition::Disabled;
5474            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5475                clear_current_process_facts(state);
5476                state.last_exit = Some(exit_report.clone());
5477                state.lifetime_restarts += 1;
5478                if state.enabled {
5479                    state.state = ModuleState::Restarting;
5480                    should_restart = true;
5481                    disposition = TerminalDisposition::Restarting;
5482                } else {
5483                    state.state = ModuleState::Disabled;
5484                }
5485            }) {
5486                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
5487                return NextAction::Stop {
5488                    registration_released: false,
5489                };
5490            }
5491            record_terminal(
5492                &spec.module_id,
5493                terminal_ring,
5494                spawn_events,
5495                &exit_report,
5496                disposition,
5497            );
5498
5499            if should_restart {
5500                NextAction::Restart { schedule: None }
5501            } else {
5502                let registration_released = match wait_for_registration_release(
5503                    registry,
5504                    &spec.module_id,
5505                    REGISTRY_RELEASE_TIMEOUT,
5506                )
5507                .await
5508                {
5509                    Ok(()) => true,
5510                    Err(err) => {
5511                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
5512                        false
5513                    }
5514                };
5515                NextAction::Stop {
5516                    registration_released,
5517                }
5518            }
5519        }
5520    }
5521}
5522
5523async fn on_child_exit_during_daemon_shutdown(
5524    spec: &ModuleSpec,
5525    registry: &Registry,
5526    snapshot: &SharedSnapshot,
5527    terminal_ring: &Arc<Mutex<TerminalRing>>,
5528    spawn_events: &SpawnEventFeed,
5529    exit_report: ExitReport,
5530) -> NextAction {
5531    info!(
5532        module_id = %spec.module_id,
5533        exit_code = ?exit_report.code,
5534        exit_signal = ?exit_report.signal,
5535        exit_kind = ?exit_report.kind,
5536        "supervised module exited during daemon shutdown; not restarting it"
5537    );
5538    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5539        state.state = ModuleState::Stopped;
5540        clear_current_process_facts(state);
5541        state.last_exit = Some(exit_report.clone());
5542    }) {
5543        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
5544    }
5545    record_terminal(
5546        &spec.module_id,
5547        terminal_ring,
5548        spawn_events,
5549        &exit_report,
5550        TerminalDisposition::DaemonShutdown,
5551    );
5552    let registration_released =
5553        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
5554            .await
5555            .is_ok();
5556    NextAction::Stop {
5557        registration_released,
5558    }
5559}
5560
5561fn record_wait_error_terminal(
5562    module_id: &str,
5563    terminal_ring: &Arc<Mutex<TerminalRing>>,
5564    spawn_events: &SpawnEventFeed,
5565) {
5566    record_terminal(
5567        module_id,
5568        terminal_ring,
5569        spawn_events,
5570        &wait_error_exit_report(),
5571        TerminalDisposition::Failed,
5572    );
5573}
5574
5575fn record_terminal(
5576    module_id: &str,
5577    terminal_ring: &Arc<Mutex<TerminalRing>>,
5578    spawn_events: &SpawnEventFeed,
5579    exit_report: &ExitReport,
5580    disposition: TerminalDisposition,
5581) {
5582    record_terminal_with_detail(
5583        module_id,
5584        terminal_ring,
5585        spawn_events,
5586        exit_report,
5587        disposition,
5588        None,
5589    );
5590}
5591
5592/// The ring lock is held only to capture the read (see
5593/// `TerminalJournal::capture_read`), so this module's exits keep recording
5594/// while the journal files are read. Blocking: it reads files.
5595fn durable_terminal_history_of(
5596    terminal_ring: &Mutex<TerminalRing>,
5597    module_id: &str,
5598) -> subc_control::TerminalHistory {
5599    let read = terminal_ring
5600        .lock()
5601        .unwrap_or_else(|p| p.into_inner())
5602        .capture_durable_history();
5603    read.read(module_id)
5604}
5605
5606fn record_terminal_with_detail(
5607    module_id: &str,
5608    terminal_ring: &Arc<Mutex<TerminalRing>>,
5609    spawn_events: &SpawnEventFeed,
5610    exit_report: &ExitReport,
5611    disposition: TerminalDisposition,
5612    disposition_detail: Option<String>,
5613) {
5614    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
5615    let record = TerminalRecord {
5616        exit_code: exit_report.code,
5617        exit_signal: exit_report.signal,
5618        at_ms: exit_report.at_ms,
5619        disposition,
5620        exit_kind: exit_report.kind.into(),
5621        disposition_detail,
5622    };
5623    terminal_ring
5624        .lock()
5625        .unwrap_or_else(|poisoned| poisoned.into_inner())
5626        .record_exit(module_id, record);
5627}
5628
5629fn untrack_if_registration_released(
5630    process_liveness: &SupervisorProcessLiveness,
5631    registry: &Registry,
5632    module_id: &str,
5633    snapshot: &SharedSnapshot,
5634) {
5635    match registry.get_module(module_id) {
5636        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
5637        Ok(Some(_)) => {}
5638        Err(err) => {
5639            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
5640        }
5641    }
5642}
5643
5644/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
5645/// then apply the module's configured entries minus daemon-private capture keys.
5646///
5647/// Separated from `spawn_child` only so it can be asserted without spawning a
5648/// process — a duplicate of this logic in a test would pass while the real one
5649/// drifted, which is the defect class this function exists to avoid.
5650/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
5651/// nonce. A `protocol: "none"` module gets neither, because it cannot use
5652/// either and the argument would stop a stock binary from starting at all.
5653/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
5654///
5655/// The plain-spawn form, kept for the tests that assert its plan; spawns go
5656/// through [`apply_wire_spawn_args_for_role`].
5657#[cfg(test)]
5658fn apply_wire_spawn_args(
5659    command: &mut Command,
5660    spec: &ModuleSpec,
5661    connection_file_path: Option<&std::path::Path>,
5662    handle: Option<&SupervisorHandle>,
5663) -> Result<Option<NonceHandoff>, SuperviseError> {
5664    apply_wire_spawn_args_for_role(
5665        command,
5666        spec,
5667        connection_file_path,
5668        handle,
5669        SpawnRole::Plain,
5670    )
5671}
5672
5673/// The read end of a spawn's launch-nonce pipe, prepared by
5674/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
5675/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
5676/// handoff and keeps only the environment copy.
5677#[cfg(unix)]
5678type NonceHandoff = subc_os::LaunchNonceHandoff;
5679#[cfg(not(unix))]
5680type NonceHandoff = std::convert::Infallible;
5681
5682/// Prepare wire identity for a plain spawn or a swap candidate.
5683///
5684/// A plain spawn replaces the module's recorded nonce. A swap candidate records
5685/// a separate candidate token so the still-serving incumbent and its consumers
5686/// keep their nonce. Both records are installed before the process exists, so
5687/// the child's initial HELLO registration cannot arrive ahead of its nonce.
5688///
5689/// On Unix the nonce is delivered only through a pipe. It is written into
5690/// a pipe whose read end the child gets as descriptor 3, named by
5691/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
5692/// process of the same user cannot read it with `ps eww`. That handoff is
5693/// returned rather than installed here, because installing it replaces
5694/// whatever the child has at descriptor 3 and so must be the last pre-exec
5695/// step, after the Linux cgroup placement that the caller registers later.
5696/// Windows retains the environment handoff until restricted handle inheritance
5697/// can be implemented outside std's process primitives.
5698fn apply_wire_spawn_args_for_role(
5699    command: &mut Command,
5700    spec: &ModuleSpec,
5701    connection_file_path: Option<&std::path::Path>,
5702    handle: Option<&SupervisorHandle>,
5703    role: SpawnRole,
5704) -> Result<Option<NonceHandoff>, SuperviseError> {
5705    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
5706    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
5707    // included: a daemon started from a module's process tree inherits it,
5708    // and passing it on would point the child at a descriptor it does not
5709    // have.
5710    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
5711    // Remove inherited or configured copies too: withholding must mean absent.
5712    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
5713    if spec.protocol == ModuleProtocol::None {
5714        return Ok(None);
5715    }
5716    if let Some(connection_file_path) = connection_file_path {
5717        command.arg(SUBC_ARG).arg(connection_file_path);
5718    }
5719
5720    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
5721    // route.open attestation. Reserved modules additionally use the same nonce
5722    // for HELLO id-squatting protection. A respawn rotates both records.
5723    let nonce = generate_launch_nonce()?;
5724    if let Some(handle) = handle {
5725        match role {
5726            SpawnRole::Plain => {
5727                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
5728                if spec.reserved {
5729                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
5730                }
5731            }
5732            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
5733        }
5734    }
5735    #[cfg(unix)]
5736    let handoff = {
5737        let handoff =
5738            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
5739                program: spec.program.clone(),
5740                source,
5741                cgroup_path: None,
5742            })?;
5743        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
5744        Some(handoff)
5745    };
5746    #[cfg(not(unix))]
5747    let handoff = None;
5748    // Windows keeps the environment copy: std cannot restrict an inherited pipe
5749    // handle to this child without leaking it to concurrently spawned processes.
5750    #[cfg(not(unix))]
5751    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
5752    Ok(handoff)
5753}
5754
5755fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
5756    command.env_remove(CK_LOG_ENV);
5757    // The spawn role is the supervisor's to set, and only on a swap candidate
5758    // (see `apply_spawn_role`). Removing it here, rather than just not setting
5759    // it, is what makes it absent on a plain spawn: the daemon's own
5760    // environment could carry it, and so could a spec built outside daemon
5761    // config (config refuses it as an `env` key). A module reading it on a
5762    // plain restart would pick the long swap budget and leave callers waiting.
5763    command.env_remove(SUBC_SPAWN_ROLE_ENV);
5764    for (key, value) in &spec.env {
5765        // cortexkit-log currently exposes retention only as a Rust struct, not
5766        // environment names. These values are daemon-private sink metadata and
5767        // must never become a public child-process contract by being inherited.
5768        if matches!(
5769            key.as_str(),
5770            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
5771        ) || key == SUBC_SPAWN_ROLE_ENV
5772        {
5773            continue;
5774        }
5775        command.env(key, value);
5776    }
5777}
5778
5779/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
5780/// of a blue/green swap.
5781#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5782enum SpawnRole {
5783    Plain,
5784    SwapCandidate,
5785}
5786
5787/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
5788/// `apply_child_env` has already removed the variable for every spawn.
5789fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
5790    if role == SpawnRole::SwapCandidate {
5791        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
5792    }
5793}
5794
5795fn spawn_child(
5796    spec: &ModuleSpec,
5797    connection_file_path: Option<&std::path::Path>,
5798    handle: Option<&SupervisorHandle>,
5799    ring: &Arc<Mutex<StderrRing>>,
5800    capture_logs_dir: Option<&std::path::Path>,
5801    roster: &ChildRoster,
5802    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5803) -> Result<SupervisedChild, SuperviseError> {
5804    spawn_child_in_slot(
5805        spec,
5806        connection_file_path,
5807        handle,
5808        ring,
5809        capture_logs_dir,
5810        roster,
5811        #[cfg(target_os = "linux")]
5812        cgroup_placement,
5813        SpawnRole::Plain,
5814        false,
5815    )
5816}
5817
5818/// Spawn one process of `spec` into a slot.
5819///
5820/// `alternate_slot` picks the process's cgroup name (see `swap::cgroup_name`).
5821/// A swap candidate needs a different cgroup from the process it is replacing,
5822/// which is still alive: in the same cgroup the two would be one kill domain,
5823/// and killing a failed candidate could take the incumbent with it.
5824///
5825/// The stderr capture file is `<module_id>.stderr.log` for every process of
5826/// the module, whichever slot it is in, because that is the one file
5827/// `ck module logs` reads. During a swap's overlap both processes append to it;
5828/// the daemon writes whole lines, so the two interleave by line, which is also
5829/// the merged view an operator wants while a swap runs.
5830#[allow(clippy::too_many_arguments)]
5831fn spawn_child_in_slot(
5832    spec: &ModuleSpec,
5833    connection_file_path: Option<&std::path::Path>,
5834    handle: Option<&SupervisorHandle>,
5835    ring: &Arc<Mutex<StderrRing>>,
5836    capture_logs_dir: Option<&std::path::Path>,
5837    roster: &ChildRoster,
5838    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5839    role: SpawnRole,
5840    alternate_slot: bool,
5841) -> Result<SupervisedChild, SuperviseError> {
5842    if roster.is_closed() {
5843        return Err(SuperviseError::Spawn {
5844            program: spec.program.clone(),
5845            source: io::Error::other("the daemon is shutting down; not starting a new process"),
5846            cgroup_path: None,
5847        });
5848    }
5849    #[cfg(target_os = "linux")]
5850    let cgroup_name = swap::cgroup_name(&spec.module_id, alternate_slot);
5851    #[cfg(not(target_os = "linux"))]
5852    let _ = alternate_slot;
5853    let mut command = Command::new(&spec.program);
5854    command.args(&spec.args);
5855    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
5856    // that is the whole of the intent, so remove that one key rather than the
5857    // environment.
5858    //
5859    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
5860    // and took the POSIX environment with it. Modules spawned that way had no
5861    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
5862    // logging:
5863    //
5864    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
5865    //     both unset it fell back to the temp dir alone and `ck` could not find
5866    //     a daemon running on the same machine from inside any module's process
5867    //     tree — reporting a path the file has never lived at, which reads as
5868    //     "the daemon did not write its file".
5869    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
5870    //     the RELATIVE `.local/share`, so a module deriving its own store path
5871    //     resolved it against its own CWD. That is the store-fragmentation
5872    //     defect the daemon already refuses in config (`parse_doc` rejects a
5873    //     relative `storage.data_home`) arriving by derivation instead.
5874    //   * anything a module spawns inherited it: git without ~/.gitconfig,
5875    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
5876    //     quietly rather than erroring.
5877    //
5878    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
5879    // offered one candidate under /tmp while the file sat in /run/user/1000.
5880    //
5881    // A configured module is unaffected either way: `module_spec()` puts the
5882    // resolved CK_LOG into `spec.env`, which is applied below and therefore
5883    // wins over anything ambient.
5884    apply_child_env(&mut command, spec);
5885    apply_spawn_role(&mut command, role);
5886    let nonce_handoff =
5887        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
5888
5889    #[cfg(target_os = "linux")]
5890    let cgroup_path = cgroup_placement
5891        .map(|placement| placement.module_path(&cgroup_name))
5892        .transpose()
5893        .map_err(|source| SuperviseError::Cgroup {
5894            module_id: spec.module_id.clone(),
5895            source,
5896        })?;
5897    #[cfg(not(target_os = "linux"))]
5898    let cgroup_path: Option<PathBuf> = None;
5899    #[cfg(target_os = "linux")]
5900    if let Some(path) = &cgroup_path {
5901        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
5902            if let Some(placement) = cgroup_placement {
5903                remove_module_cgroup(placement, &cgroup_name);
5904            }
5905            return Err(error);
5906        }
5907    }
5908
5909    let output_sink = if let Some(logs_dir) = capture_logs_dir {
5910        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
5911        match ChildOutputSink::open(&path, capture_retention(spec)) {
5912            Ok(sink) => sink,
5913            Err(error) => {
5914                warn!(
5915                    module_id = %spec.module_id,
5916                    path = %path.display(),
5917                    error = %error,
5918                    "could not open child output capture file; forwarding to stderr"
5919                );
5920                ChildOutputSink::Stderr
5921            }
5922        }
5923    } else {
5924        ChildOutputSink::Stderr
5925    };
5926
5927    command.stdout(Stdio::piped());
5928    command.stderr(Stdio::piped());
5929    command.kill_on_drop(true);
5930    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
5931    // before exec). In the daemon's group, a service manager that kills the
5932    // job's process group when the daemon exits (launchd's default) killed
5933    // every module at the same moment its control connection closed, so no
5934    // module ever ran its EOF teardown on a daemon stop. Outside that group a
5935    // module is reached only by the daemon: the EOF it sees when its
5936    // connection closes, and the bounded stop in `child_roster` for anything
5937    // still running after that. On Linux this composes with the cgroup
5938    // placement above: that is a pre_exec write to cgroup.procs, std performs
5939    // setpgid in the child before running pre_exec callbacks, and the two
5940    // change independent process attributes.
5941    //
5942    // stdin is /dev/null because a process outside the terminal's foreground
5943    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
5944    // by hand would otherwise hand down. Under a service manager stdin is
5945    // already /dev/null.
5946    #[cfg(unix)]
5947    command.process_group(0);
5948    command.stdin(Stdio::null());
5949    // The LAST pre-exec step, after the cgroup placement above: installing the
5950    // nonce at descriptor 3 replaces whatever the child had there, which could
5951    // be the descriptor an earlier step writes through.
5952    #[cfg(unix)]
5953    if let Some(handoff) = nonce_handoff {
5954        handoff.install_last(command.as_std_mut());
5955    }
5956    #[cfg(not(unix))]
5957    let _ = nonce_handoff;
5958
5959    // Containment, step 1 of 3 (issue #109): create the child suspended so it
5960    // cannot run a single instruction -- and therefore cannot spawn a
5961    // grandchild -- before it is in the job. See `contain_spawned_child` for the
5962    // other two steps and why the window matters.
5963    #[cfg(windows)]
5964    subc_jobobject::suspend_on_create_async(&mut command);
5965    let mut child = match command.spawn() {
5966        Ok(child) => child,
5967        Err(source) => {
5968            #[cfg(target_os = "linux")]
5969            if let Some(placement) = cgroup_placement {
5970                remove_module_cgroup(placement, &cgroup_name);
5971            }
5972            return Err(SuperviseError::Spawn {
5973                program: spec.program.clone(),
5974                source,
5975                cgroup_path,
5976            });
5977        }
5978    };
5979
5980    // Containment, steps 2 and 3: assign while suspended, then resume.
5981    #[cfg(windows)]
5982    let job = contain_spawned_child(&child, spec)?;
5983    let spawned_at_ms = unix_ms_now();
5984    let spawned_from = spec.program.clone();
5985    let spawned_file_identity = spawned_file_identity(&spawned_from);
5986    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
5987        program: spec.program.clone(),
5988        source: io::Error::other("spawned child exposed no live pid"),
5989        cgroup_path: cgroup_path.clone(),
5990    })?;
5991    let process_start_time = crate::provenance::process_start_time(pid);
5992    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
5993    // The executable identity is the spawned path's, read above, not the
5994    // running image's: right after spawn the child may not have finished its
5995    // exec yet and would still report this daemon's own image.
5996    #[cfg(target_os = "linux")]
5997    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
5998    #[cfg(not(target_os = "linux"))]
5999    let recorded_cgroup_name = None;
6000    let roster_guard = roster.admit(
6001        spec.module_id.clone(),
6002        pid,
6003        spec.protocol,
6004        process_start_time,
6005        crate::child_roster::RecordedIdentity {
6006            start_time: subc_os::start_time(pid),
6007            executable: spawned_file_identity.map(|identity| {
6008                crate::live_children::ExecutableIdentity {
6009                    device: identity.device,
6010                    inode: identity.inode,
6011                }
6012            }),
6013            cgroup_name: recorded_cgroup_name,
6014            #[cfg(target_os = "linux")]
6015            cgroup_placement: cgroup_placement.cloned(),
6016        },
6017    );
6018    // The check at the top of this function can pass just before daemon
6019    // shutdown begins, and the process is only in the roster from here on.
6020    // The shutdown stop returns as soon as it finds the roster empty, so a
6021    // process admitted after that look would outlive the daemon. The roster
6022    // is closed before the stop first reads it and admission happens under
6023    // the roster's lock, so either the stop sees this process or this check
6024    // sees the roster closed: end the process now rather than start a module
6025    // the daemon is about to stop.
6026    if roster.is_closed() {
6027        // This child was never admitted, so there is no module protocol shutdown to wait for.
6028        #[cfg(target_os = "linux")]
6029        kill_module_cgroup(cgroup_placement, &cgroup_name);
6030        if let Err(error) = child.start_kill() {
6031            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
6032        }
6033        drop(roster_guard);
6034        return Err(SuperviseError::Spawn {
6035            program: spec.program.clone(),
6036            source: io::Error::other(
6037                "the daemon began shutting down while this process was starting; ended it",
6038            ),
6039            cgroup_path,
6040        });
6041    }
6042
6043    let stdout_pump = match child.stdout.take() {
6044        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
6045        None => {
6046            warn!(
6047                module_id = %spec.module_id,
6048                "spawned child exposed no stdout pipe; file capture will be incomplete"
6049            );
6050            None
6051        }
6052    };
6053    let stderr_pump = match child.stderr.take() {
6054        Some(stderr) => {
6055            let generation = ring
6056                .lock()
6057                .unwrap_or_else(|poisoned| poisoned.into_inner())
6058                .begin_process();
6059            Some(StderrPump {
6060                task: tokio::spawn(pump_stderr_to(
6061                    stderr,
6062                    Arc::clone(ring),
6063                    generation,
6064                    output_sink,
6065                )),
6066                generation,
6067            })
6068        }
6069        None => {
6070            // Spawning succeeded but the pipe did not materialise. Recording it as
6071            // uncaptured keeps the tail honest: the alternative is an empty tail
6072            // that reads as a module which printed nothing.
6073            ring.lock()
6074                .unwrap_or_else(|poisoned| poisoned.into_inner())
6075                .mark_not_captured("stderr pipe was not available on spawn");
6076            warn!(
6077                module_id = %spec.module_id,
6078                "spawned child exposed no stderr pipe; tail will be unavailable"
6079            );
6080            None
6081        }
6082    };
6083
6084    Ok(SupervisedChild {
6085        child,
6086        #[cfg(target_os = "linux")]
6087        module_id: cgroup_name,
6088        #[cfg(target_os = "linux")]
6089        cgroup_placement: cgroup_placement.cloned(),
6090        #[cfg(windows)]
6091        job,
6092        stdout_pump,
6093        stderr_pump,
6094        stderr_ring: Arc::clone(ring),
6095        spawned_at_ms,
6096        spawned_from,
6097        spawned_file_identity,
6098        process_start_time,
6099        process_identity,
6100        pid,
6101        roster_guard: Some(roster_guard),
6102    })
6103}
6104
6105#[cfg(target_os = "linux")]
6106pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
6107    use subc_cgroup::KillOutcome;
6108    match subc_cgroup::kill_module(placement, module_id) {
6109        KillOutcome::Killed => {}
6110        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
6111            debug!(
6112                module_id,
6113                "cgroup tree kill unavailable; using direct-child kill"
6114            );
6115        }
6116        KillOutcome::IoError { path, error } => {
6117            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
6118        }
6119    }
6120}
6121
6122/// Contain a freshly spawned Windows child and start it.
6123///
6124/// Steps 2 and 3 of the suspended-create contract: the job is created and the
6125/// child assigned **while it is still suspended** (step 1 is
6126/// `suspend_on_create_async` at the spawn site), then the child is resumed.
6127///
6128/// A child that is never resumed hangs forever holding a pid, so a resume
6129/// failure kills the child and fails the spawn rather than returning a
6130/// `SupervisedChild` that can never run.
6131///
6132/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
6133/// it did before this existed, whereas refusing to start one would be a new
6134/// outage. It is logged at warn because it means a helper process could leak.
6135#[cfg(windows)]
6136fn contain_spawned_child(
6137    child: &Child,
6138    spec: &ModuleSpec,
6139) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
6140    let module_id = spec.module_id.as_str();
6141    let Some(pid) = child.id() else {
6142        // The child exited between spawn and here. Its tree, if it made one,
6143        // needs no containment: nothing is left to contain.
6144        warn!(
6145            module_id,
6146            "spawned child had already exited before containment; no job object attached"
6147        );
6148        return Ok(None);
6149    };
6150
6151    let job = match subc_jobobject::JobObject::new() {
6152        Ok(job) => job,
6153        Err(source) => {
6154            warn!(
6155                module_id,
6156                error = %source,
6157                "could not create a job object; this module's helper processes will not be \
6158                 reaped on teardown"
6159            );
6160            // Resume regardless: leaving the child suspended would turn a
6161            // containment gap into a hung module.
6162            resume_suspended_child(pid, spec)?;
6163            return Ok(None);
6164        }
6165    };
6166
6167    if let Err(source) = job.assign(child) {
6168        warn!(
6169            module_id,
6170            error = %source,
6171            "could not assign the child to its job object; this module's helper processes \
6172             will not be reaped on teardown"
6173        );
6174        resume_suspended_child(pid, spec)?;
6175        return Ok(None);
6176    }
6177
6178    resume_suspended_child(pid, spec)?;
6179    Ok(Some(job))
6180}
6181
6182/// Resume a suspended child, killing it if it cannot be started.
6183///
6184/// A suspended process holds a pid and does nothing, so there is no useful
6185/// state to return: the caller gets an error and the spawn fails.
6186#[cfg(windows)]
6187fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
6188    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
6189        // Kill it here rather than leaving a suspended process for the caller
6190        // to notice; `kill_on_drop` would eventually do this, but the module
6191        // would have been reported as running in between.
6192        let _ = std::process::Command::new("taskkill.exe")
6193            .args(["/PID", &pid.to_string(), "/T", "/F"])
6194            .stdin(Stdio::null())
6195            .stdout(Stdio::null())
6196            .stderr(Stdio::null())
6197            .status();
6198        return Err(SuperviseError::Spawn {
6199            program: spec.program.clone(),
6200            source,
6201            cgroup_path: None,
6202        });
6203    }
6204    Ok(())
6205}
6206
6207#[cfg(target_os = "linux")]
6208fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
6209    match placement.remove_module(module_id) {
6210        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
6211        Err(error) => warn!(
6212            module_id,
6213            error = %error,
6214            "could not remove module cgroup after process exit; continuing teardown"
6215        ),
6216    }
6217}
6218
6219#[cfg(target_os = "linux")]
6220fn apply_cgroup_placement(
6221    command: &mut Command,
6222    spec: &ModuleSpec,
6223    path: &std::path::Path,
6224) -> Result<(), SuperviseError> {
6225    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
6226        module_id: spec.module_id.clone(),
6227        source,
6228    })
6229}
6230
6231fn capture_retention(spec: &ModuleSpec) -> Retention {
6232    let defaults = Retention::default();
6233    let value = |name: &str| {
6234        spec.env
6235            .iter()
6236            .rev()
6237            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
6238    };
6239    Retention {
6240        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
6241            .and_then(|value| value.parse().ok())
6242            .unwrap_or(defaults.max_file_mb),
6243        keep: value(CAPTURE_KEEP_ENV)
6244            .and_then(|value| value.parse().ok())
6245            .unwrap_or(defaults.keep),
6246        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
6247            .and_then(|value| value.parse().ok())
6248            .unwrap_or(defaults.max_age_days),
6249    }
6250}
6251
6252/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
6253/// module's registration to the exact process the supervisor spawned.
6254fn generate_launch_nonce() -> Result<String, SuperviseError> {
6255    let mut bytes = [0u8; 32];
6256    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
6257        reason: source.to_string(),
6258    })?;
6259    let mut hex = String::with_capacity(64);
6260    for b in bytes {
6261        use std::fmt::Write;
6262        let _ = write!(hex, "{b:02x}");
6263    }
6264    Ok(hex)
6265}
6266
6267/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
6268/// signal about how many leading bytes matched.
6269fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
6270    if a.len() != b.len() {
6271        return false;
6272    }
6273    let mut diff = 0u8;
6274    for (x, y) in a.iter().zip(b.iter()) {
6275        diff |= x ^ y;
6276    }
6277    diff == 0
6278}
6279
6280fn spawn_and_mark_running(
6281    spec: &ModuleSpec,
6282    runtime: &SupervisorRuntimeConfig,
6283    snapshot: &SharedSnapshot,
6284) -> Result<SupervisedChild, SuperviseError> {
6285    let child = spawn_child(
6286        spec,
6287        runtime.connection_file_path.as_deref(),
6288        runtime.supervisor_handle.as_ref(),
6289        &runtime.stderr_ring,
6290        runtime.capture_logs_dir.as_deref(),
6291        &runtime.child_roster,
6292        #[cfg(target_os = "linux")]
6293        runtime.cgroup_placement.as_ref(),
6294    )?;
6295    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
6296    Ok(child)
6297}
6298
6299enum RegistrationWaitOutcome {
6300    Registered,
6301    Exited(ExitReport),
6302    TimedOut,
6303}
6304
6305struct ReloadRegistrationFailure {
6306    exit_report: ExitReport,
6307    reason: String,
6308}
6309
6310#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6311enum BusyGaugeObservation {
6312    Quiescent,
6313    Busy,
6314    Omitted,
6315}
6316
6317fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
6318    let Some(metrics) = metrics.and_then(Value::as_object) else {
6319        return BusyGaugeObservation::Omitted;
6320    };
6321    let mut sum = 0u128;
6322    for gauge in gauges {
6323        let Some(value) = metrics.get(gauge) else {
6324            return BusyGaugeObservation::Omitted;
6325        };
6326        let Some(value) = value.as_u64() else {
6327            return BusyGaugeObservation::Busy;
6328        };
6329        sum = sum.saturating_add(u128::from(value));
6330    }
6331    if sum == 0 {
6332        BusyGaugeObservation::Quiescent
6333    } else {
6334        BusyGaugeObservation::Busy
6335    }
6336}
6337
6338fn declared_busy_gauges(
6339    registry: &Registry,
6340    module_id: &str,
6341) -> Result<Vec<String>, SuperviseError> {
6342    busy_gauges_of(
6343        registry
6344            .get_module(module_id)
6345            .map_err(SuperviseError::Registry)?,
6346    )
6347}
6348
6349/// [`declared_busy_gauges`] for the registration a connection holds, in any
6350/// slot: after cutover the incumbent is no longer the id's active
6351/// registration, and its own manifest is the one that names its gauges.
6352fn declared_busy_gauges_for_connection(
6353    registry: &Registry,
6354    connection_id: ConnectionId,
6355) -> Result<Vec<String>, SuperviseError> {
6356    busy_gauges_of(
6357        registry
6358            .get_module_by_connection(connection_id)
6359            .map_err(SuperviseError::Registry)?,
6360    )
6361}
6362
6363fn busy_gauges_of(
6364    registration: Option<crate::registry::ModuleRegistration>,
6365) -> Result<Vec<String>, SuperviseError> {
6366    let Some(registration) = registration else {
6367        return Ok(Vec::new());
6368    };
6369    let Some(self_signals) = registration.manifest.self_signals else {
6370        return Ok(Vec::new());
6371    };
6372
6373    let mut gauges = Vec::new();
6374    for declaration in self_signals {
6375        if declaration.kind != SelfSignalKind::Busy {
6376            continue;
6377        }
6378        match declaration.anchored_to {
6379            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
6380                gauges.extend(declared)
6381            }
6382            _ => {
6383                // An invalid Busy anchor is fail-safe: the empty name cannot be
6384                // present in a conforming health report, so this drain stays busy.
6385                gauges.push(String::new());
6386            }
6387        }
6388    }
6389    Ok(gauges)
6390}
6391
6392/// Wait for `endpoint` to have nothing in flight and, when the module declares
6393/// busy gauges, for a health probe to report them quiet. The probe is addressed
6394/// by `scope`: a swap's superseded incumbent must be asked about its own
6395/// gauges, and by module id the probe would reach the promoted candidate.
6396async fn wait_for_forwarding_quiescence(
6397    forwarding: &ForwardingTable,
6398    module_id: &str,
6399    runtime: &SupervisorRuntimeConfig,
6400    endpoint: crate::ModuleEndpointId,
6401    deadline: Instant,
6402    busy_gauges: &[String],
6403    scope: DrainScope,
6404) -> Result<bool, SuperviseError> {
6405    let mut gauges_quiescent = busy_gauges.is_empty();
6406    let mut next_probe_at = Instant::now();
6407    let mut omission_counted = false;
6408
6409    loop {
6410        let now = Instant::now();
6411        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
6412            let report = match scope {
6413                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
6414                DrainScope::Endpoint(endpoint) => {
6415                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
6416                }
6417            };
6418            gauges_quiescent = match report {
6419                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
6420                    BusyGaugeObservation::Quiescent => true,
6421                    BusyGaugeObservation::Busy => false,
6422                    BusyGaugeObservation::Omitted => {
6423                        if !omission_counted {
6424                            forwarding
6425                                .counters()
6426                                .increment_drains_with_undeclared_gauge();
6427                            omission_counted = true;
6428                        }
6429                        false
6430                    }
6431                },
6432                Err(err) => {
6433                    warn!(
6434                        module_id,
6435                        error = %err,
6436                        "drain health.check did not produce declared busy gauges; treating module as busy"
6437                    );
6438                    false
6439                }
6440            };
6441            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
6442        }
6443
6444        let in_flight = forwarding
6445            .endpoint_in_flight_count(endpoint)
6446            .map_err(SuperviseError::Forwarding)?;
6447        if in_flight == 0 && gauges_quiescent {
6448            return Ok(true);
6449        }
6450
6451        let now = Instant::now();
6452        if now >= deadline {
6453            return Ok(false);
6454        }
6455        let mut wait = deadline
6456            .saturating_duration_since(now)
6457            .min(REGISTRY_RELEASE_POLL);
6458        if !busy_gauges.is_empty() {
6459            wait = wait.min(next_probe_at.saturating_duration_since(now));
6460        }
6461        sleep(wait).await;
6462    }
6463}
6464
6465/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
6466///
6467/// `Ok` is always honest and passed straight through -- the wait actually measured
6468/// in-flight state. `Err` means the wait produced no measurement at all (the
6469/// forwarding table's lock was poisoned), so `false` is reported as the one honest
6470/// constant: the drain did not complete. Never recomputed from route state, never a
6471/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
6472fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
6473    match wait_result {
6474        Ok(drained) => *drained,
6475        Err(_) => false,
6476    }
6477}
6478
6479fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
6480    for released in released_routes {
6481        let frame = match Frame::build_with_version(
6482            released.negotiated_ver,
6483            FrameType::Goodbye,
6484            control_flags(),
6485            released.channel,
6486            released.epoch,
6487            0,
6488            Vec::new(),
6489        ) {
6490            Ok(frame) => frame,
6491            Err(err) => {
6492                warn!(
6493                    route_channel = released.channel,
6494                    error = %err,
6495                    "failed to build supervisor drain route GOODBYE frame"
6496                );
6497                continue;
6498            }
6499        };
6500        if !released.close_on_delivery_failure() {
6501            crate::forwarding::send_module_route_goodbye(
6502                &forwarding.counters(),
6503                &released.sink,
6504                frame,
6505                released.module_id.as_deref(),
6506                "supervisor drain",
6507            );
6508            continue;
6509        }
6510        if let Err(err) = released.sink.try_send(frame) {
6511            warn!(
6512                target_connection_id = released.connection_id.get(),
6513                route_channel = released.channel,
6514                error = %err,
6515                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
6516            );
6517            let _ = forwarding.escalate_client_delivery_failure(
6518                released.connection_id,
6519                released.channel,
6520                released.epoch,
6521                CloseReason::new(
6522                    "route_goodbye_delivery_failed",
6523                    format!(
6524                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
6525                        released.channel
6526                    ),
6527                ),
6528                crate::forwarding::UndeliveredFrame {
6529                    module_id: released.module_id.as_deref(),
6530                    sink: &released.sink,
6531                },
6532            );
6533        }
6534    }
6535}
6536
6537fn send_module_draining(
6538    module_id: &str,
6539    reason: RouteCloseReason,
6540    deadline_ms: u64,
6541    target: &ModuleDrainTarget,
6542) {
6543    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
6544        reason,
6545        deadline_ms,
6546    }) {
6547        Ok(body) => body,
6548        Err(err) => {
6549            warn!(
6550                module_id,
6551                error = %err,
6552                "failed to encode module draining command"
6553            );
6554            return;
6555        }
6556    };
6557    let frame = match Frame::build_with_version(
6558        target.negotiated_ver,
6559        FrameType::Push,
6560        control_flags(),
6561        0,
6562        0,
6563        0,
6564        body,
6565    ) {
6566        Ok(frame) => frame,
6567        Err(err) => {
6568            warn!(
6569                module_id,
6570                error = %err,
6571                "failed to build module draining command frame"
6572            );
6573            return;
6574        }
6575    };
6576    if let Err(err) = target.sink.try_send(frame) {
6577        warn!(
6578            module_id,
6579            target_connection_id = target.endpoint.connection_id.get(),
6580            error = %err,
6581            "module draining command was not delivered to peer"
6582        );
6583    }
6584}
6585
6586/// The channel-0 GOODBYE that tells a module its stop is planned.
6587fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
6588    match Frame::build_with_version(
6589        negotiated_ver,
6590        FrameType::Goodbye,
6591        control_flags(),
6592        0,
6593        0,
6594        0,
6595        Vec::new(),
6596    ) {
6597        Ok(frame) => Some(frame),
6598        Err(err) => {
6599            warn!(
6600                module_id,
6601                error = %err,
6602                "failed to build module GOODBYE frame"
6603            );
6604            None
6605        }
6606    }
6607}
6608
6609/// Send every registered module connection its module GOODBYE at daemon
6610/// shutdown, then request that connection's close.
6611///
6612/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
6613/// before EOF, so the GOODBYE must reach the socket before the close. A close
6614/// request does not wait for the connection's queued frames: its writer gets a
6615/// bounded grace after the close, is aborted if it overruns it, and the daemon
6616/// process may exit before that grace ends. So with `wait_for_flush`, each
6617/// connection is closed only after its writer has acknowledged writing the
6618/// GOODBYE, or once a short shared budget runs out, so one module that is not
6619/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
6620/// are only queued, for a shutdown the operator has told to stop waiting.
6621/// A connection that is already gone is skipped.
6622#[cfg(unix)]
6623async fn send_module_goodbyes_for_daemon_shutdown(
6624    forwarding: &Arc<ForwardingTable>,
6625    reason: &CloseReason,
6626    wait_for_flush: bool,
6627) {
6628    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
6629    let targets = match forwarding.module_connections() {
6630        Ok(targets) => targets,
6631        Err(err) => {
6632            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
6633            return;
6634        }
6635    };
6636    let deadline = Instant::now() + GOODBYE_BUDGET;
6637    let mut sends = tokio::task::JoinSet::new();
6638    for target in targets {
6639        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
6640            continue;
6641        };
6642        if !wait_for_flush {
6643            if let Err(err) = target.sink.try_send(frame) {
6644                debug!(
6645                    module_id = %target.module_id,
6646                    error = %err,
6647                    "shutdown module GOODBYE was not queued"
6648                );
6649            }
6650            continue;
6651        }
6652        let forwarding = Arc::clone(forwarding);
6653        let reason = reason.clone();
6654        sends.spawn(async move {
6655            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
6656                Ok(Ok(())) => {}
6657                Ok(Err(err)) => debug!(
6658                    module_id = %target.module_id,
6659                    error = %err,
6660                    "module connection closed before its shutdown GOODBYE was written"
6661                ),
6662                Err(_) => warn!(
6663                    module_id = %target.module_id,
6664                    budget = ?GOODBYE_BUDGET,
6665                    "shutdown module GOODBYE was not written within its budget; closing anyway"
6666                ),
6667            }
6668            forwarding.request_connection_close(target.endpoint.connection_id, reason);
6669        });
6670    }
6671    // Every task ends by the shared deadline, so this wait is bounded too.
6672    while sends.join_next().await.is_some() {}
6673}
6674
6675fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
6676    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
6677        return;
6678    };
6679    if let Err(err) = target.sink.try_send(frame) {
6680        warn!(
6681            module_id,
6682            target_connection_id = target.endpoint.connection_id.get(),
6683            error = %err,
6684            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
6685        );
6686        forwarding.request_connection_close(
6687            target.endpoint.connection_id,
6688            CloseReason::new(
6689                "module_goodbye_delivery_failed",
6690                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
6691            ),
6692        );
6693    }
6694}
6695
6696#[derive(Clone, Copy)]
6697struct ForwardingDrainContext<'a> {
6698    spec: &'a ModuleSpec,
6699    runtime: &'a SupervisorRuntimeConfig,
6700    registry: &'a Registry,
6701    scope: DrainScope,
6702}
6703
6704/// Which process a forwarding drain addresses.
6705#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6706enum DrainScope {
6707    /// Whatever endpoint is active for the module id: every plain stop,
6708    /// restart and reload. Also moves the module's state to `Draining`.
6709    Active,
6710    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
6711    /// module id would resolve to the promoted candidate and leave neither
6712    /// process routable. The module's state is left alone, since the promoted
6713    /// candidate is what it describes and that process is running.
6714    Endpoint(crate::ModuleEndpointId),
6715}
6716
6717/// Whether a child being drained has already been asked to stop by the time
6718/// its drain wait starts.
6719///
6720/// The drain wait is the same budget whatever this says. What it decides is
6721/// whether the supervisor must ask by signal before that wait begins: a child
6722/// that nobody asked will sit out the whole budget and then be SIGKILLed,
6723/// healthy or not.
6724#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6725enum StopNotice {
6726    /// The module was sent `module.draining` and a module GOODBYE over its own
6727    /// registered connection, and stops itself.
6728    SentOverConnection,
6729    /// The forwarding drain found no registered connection for the module: a
6730    /// subc child spawned moments ago that has not sent HELLO yet, or a
6731    /// `protocol: "none"` child, which never registers.
6732    NoConnection,
6733    /// This path sends nothing over the module's connection: the supervisor has
6734    /// no forwarding table, or the caller stops the child without a forwarding
6735    /// drain.
6736    NotSent,
6737}
6738
6739async fn begin_forwarding_drain(
6740    spec: &ModuleSpec,
6741    runtime: &SupervisorRuntimeConfig,
6742    registry: &Registry,
6743    snapshot: &SharedSnapshot,
6744    enabled: Option<bool>,
6745    reason: RouteCloseReason,
6746) -> Result<StopNotice, SuperviseError> {
6747    let Some(forwarding) = runtime.forwarding.as_ref() else {
6748        return Err(SuperviseError::ReloadUnavailable {
6749            module_id: spec.module_id.clone(),
6750            reason: "supervisor was not configured with a forwarding table".to_string(),
6751        });
6752    };
6753
6754    begin_forwarding_drain_with(
6755        forwarding,
6756        ForwardingDrainContext {
6757            spec,
6758            runtime,
6759            registry,
6760            scope: DrainScope::Active,
6761        },
6762        snapshot,
6763        enabled,
6764        reason,
6765        runtime.drain_timeout,
6766    )
6767    .await
6768}
6769
6770async fn begin_forwarding_drain_if_configured(
6771    spec: &ModuleSpec,
6772    runtime: &SupervisorRuntimeConfig,
6773    registry: &Registry,
6774    snapshot: &SharedSnapshot,
6775    enabled: Option<bool>,
6776    reason: RouteCloseReason,
6777) -> Result<StopNotice, SuperviseError> {
6778    begin_forwarding_drain_with_timeout(
6779        spec,
6780        runtime,
6781        registry,
6782        snapshot,
6783        enabled,
6784        reason,
6785        runtime.drain_timeout,
6786    )
6787    .await
6788}
6789
6790/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
6791/// budget, for paths where the operator overrides the module's configured one
6792/// (`supervisor.restart{drain_timeout_ms}`).
6793async fn begin_forwarding_drain_with_timeout(
6794    spec: &ModuleSpec,
6795    runtime: &SupervisorRuntimeConfig,
6796    registry: &Registry,
6797    snapshot: &SharedSnapshot,
6798    enabled: Option<bool>,
6799    reason: RouteCloseReason,
6800    drain_timeout: Duration,
6801) -> Result<StopNotice, SuperviseError> {
6802    let Some(forwarding) = runtime.forwarding.as_ref() else {
6803        return Ok(StopNotice::NotSent);
6804    };
6805
6806    begin_forwarding_drain_with(
6807        forwarding,
6808        ForwardingDrainContext {
6809            spec,
6810            runtime,
6811            registry,
6812            scope: DrainScope::Active,
6813        },
6814        snapshot,
6815        enabled,
6816        reason,
6817        drain_timeout,
6818    )
6819    .await
6820}
6821
6822async fn begin_forwarding_drain_with(
6823    forwarding: &ForwardingTable,
6824    context: ForwardingDrainContext<'_>,
6825    snapshot: &SharedSnapshot,
6826    enabled: Option<bool>,
6827    reason: RouteCloseReason,
6828    drain_timeout: Duration,
6829) -> Result<StopNotice, SuperviseError> {
6830    let ForwardingDrainContext {
6831        spec,
6832        runtime,
6833        registry,
6834        scope,
6835    } = context;
6836    debug_assert_ne!(reason, RouteCloseReason::Crash);
6837    let terminal = matches!(reason, RouteCloseReason::Disable);
6838    let drain_started_at = Instant::now();
6839    let drain_deadline = drain_started_at + drain_timeout;
6840    let deadline_ms =
6841        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
6842    let busy_gauges = match scope {
6843        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
6844        DrainScope::Endpoint(endpoint) => {
6845            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
6846        }
6847    };
6848
6849    // Admission gate first: route.open/commit and route REQUEST admission are closed
6850    // before the first quiescence check, so the outstanding count can only fall.
6851    let gate_started = Instant::now();
6852    let drain_target = match scope {
6853        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
6854        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
6855    }
6856    .map_err(SuperviseError::Forwarding)?;
6857    // The instant admission closed, and how long taking the forwarding write
6858    // lock to close it took. The timeout line reports only the quiescence
6859    // wait, so without this a drain that started late looked like one that
6860    // started on time.
6861    info!(
6862        module_id = %spec.module_id,
6863        ?reason,
6864        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
6865        connected = drain_target.is_some(),
6866        "module drain began; route admission closed"
6867    );
6868    if scope == DrainScope::Active {
6869        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6870            state.state = ModuleState::Draining;
6871            state.draining_to_replace =
6872                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
6873            if let Some(enabled) = enabled {
6874                state.enabled = enabled;
6875            }
6876        })?;
6877    }
6878
6879    let Some(target) = drain_target.as_ref() else {
6880        // Nothing was sent: the module has no registered connection to carry
6881        // `module.draining` or a GOODBYE. The caller must not assume the child
6882        // was asked to stop.
6883        return Ok(StopNotice::NoConnection);
6884    };
6885    {
6886        send_module_draining(&spec.module_id, reason, deadline_ms, target);
6887        let routes = forwarding
6888            .endpoint_routes(target.endpoint)
6889            .map_err(SuperviseError::Forwarding)?;
6890        let routes_notified = routes.len();
6891        crate::control::send_route_control_pushes(
6892            forwarding,
6893            routes.clone(),
6894            ClientControlPush::RouteClosing {
6895                module_id: spec.module_id.clone(),
6896                channels: Vec::new(),
6897                reason,
6898            },
6899        );
6900        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
6901
6902        // `route.closing` was just sent above: from here on every return path,
6903        // including an early one, MUST send `route.closed` before propagating
6904        // anything else. A client holds `closing` as a promise that a verdict is
6905        // coming; leaving early without `closed` strands it waiting forever, since
6906        // `closing` carries no timeout of its own.
6907        let wait_result = wait_for_forwarding_quiescence(
6908            forwarding,
6909            &spec.module_id,
6910            runtime,
6911            target.endpoint,
6912            drain_deadline,
6913            &busy_gauges,
6914            scope,
6915        )
6916        .await;
6917        let drained = drained_after_quiescence_wait(&wait_result);
6918        if let Err(err) = &wait_result {
6919            error!(
6920                module_id = %spec.module_id,
6921                ?reason,
6922                error = %err,
6923                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
6924            );
6925        } else if !drained {
6926            // Name what the drain waited on. Without it the line says only that
6927            // something did not settle, and "one wedged call" and "every
6928            // session's held stream" read the same; the first is a module bug,
6929            // the second is a module that should end its streams on
6930            // module.draining. Read before teardown releases the routes.
6931            let holdouts = forwarding
6932                .endpoint_drain_holdouts(target.endpoint)
6933                .unwrap_or_default();
6934            warn!(
6935                module_id = %spec.module_id,
6936                waited = ?drain_timeout,
6937                ?reason,
6938                held_requests = holdouts.requests,
6939                held_routes = holdouts.routes,
6940                total_routes = holdouts.total_routes,
6941                top_connections = ?holdouts.top_connections,
6942                // `module_channel:corr`, so the module can find each held request
6943                // in its own log; capped, so `held_requests` is the full count.
6944                held = %holdouts
6945                    .held
6946                    .iter()
6947                    .map(|(channel, corr)| format!("{channel}:{corr}"))
6948                    .collect::<Vec<_>>()
6949                    .join(","),
6950                "route drain timed out before request quiescence; forcing teardown"
6951            );
6952        }
6953        crate::control::send_route_control_pushes(
6954            forwarding,
6955            routes,
6956            ClientControlPush::RouteClosed {
6957                module_id: spec.module_id.clone(),
6958                channels: Vec::new(),
6959                reason,
6960                drained,
6961                abandoned: target.abandoned_bindings.len() as u32,
6962                excluded_subscriptions: target.excluded_subscriptions,
6963                terminal: Some(terminal),
6964            },
6965        );
6966        wait_result?;
6967
6968        // `route.closed` has now been sent unconditionally above. From here the
6969        // remaining steps are cleanup (route + module GOODBYE) rather than a
6970        // promise the client is waiting on, but a lock-poisoned
6971        // `release_module_endpoint_routes` would otherwise skip the module
6972        // GOODBYE silently too -- send it before propagating the error.
6973        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
6974            Ok(routes) => routes,
6975            Err(err) => {
6976                warn!(
6977                    module_id = %spec.module_id,
6978                    ?reason,
6979                    error = %err,
6980                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
6981                );
6982                send_module_goodbye(&spec.module_id, forwarding, target);
6983                return Err(SuperviseError::Forwarding(err));
6984            }
6985        };
6986        let route_goodbye_count = released_routes.len();
6987        send_route_goodbyes(forwarding, released_routes);
6988        send_module_goodbye(&spec.module_id, forwarding, target);
6989
6990        // The drain's happy path was previously silent: every emission above is
6991        // best-effort with only its failure arm logged, so "were consumers told"
6992        // was unprovable from the daemon log (surfaced by a 30-minute consumer
6993        // hang where the open question was exactly whether teardown notice went
6994        // out). One summary line makes that class decidable in one grep.
6995        info!(
6996            module_id = %spec.module_id,
6997            ?reason,
6998            routes_notified,
6999            route_goodbyes = route_goodbye_count,
7000            abandoned_reservations = target.abandoned_bindings.len(),
7001            excluded_subscriptions = target.excluded_subscriptions,
7002            drained,
7003            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
7004        );
7005    }
7006
7007    Ok(StopNotice::SentOverConnection)
7008}
7009
7010/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
7011/// the only slot a plain (non-swap) spawn can register into.
7012async fn wait_for_registration_after_reload(
7013    registry: &Registry,
7014    module_id: &str,
7015    snapshot: &SharedSnapshot,
7016    child: &mut SupervisedChild,
7017    wait: Duration,
7018) -> Result<RegistrationWaitOutcome, SuperviseError> {
7019    wait_for_slot_registration(
7020        registry,
7021        crate::registry::RegistrationSlot::Active(module_id),
7022        module_id,
7023        snapshot,
7024        child,
7025        wait,
7026    )
7027    .await
7028}
7029
7030/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
7031///
7032/// Keyed on the slot rather than the bare module id because during a swap the
7033/// id's active slot is already held by the incumbent: an id-keyed wait would
7034/// report the incumbent's registration as the candidate's and a candidate that
7035/// never registers would look registered. A swap candidate waits on
7036/// `crate::registry::RegistrationSlot::Candidate`.
7037async fn wait_for_slot_registration(
7038    registry: &Registry,
7039    slot: crate::registry::RegistrationSlot<'_>,
7040    module_id: &str,
7041    snapshot: &SharedSnapshot,
7042    child: &mut SupervisedChild,
7043    wait: Duration,
7044) -> Result<RegistrationWaitOutcome, SuperviseError> {
7045    let deadline = Instant::now() + wait;
7046    loop {
7047        if registry
7048            .registration(slot)
7049            .map_err(SuperviseError::Registry)?
7050            .is_some()
7051        {
7052            return Ok(RegistrationWaitOutcome::Registered);
7053        }
7054
7055        let now = Instant::now();
7056        if now >= deadline {
7057            return Ok(RegistrationWaitOutcome::TimedOut);
7058        }
7059        let remaining = deadline.saturating_duration_since(now);
7060        let poll = remaining.min(REGISTRY_RELEASE_POLL);
7061
7062        tokio::select! {
7063            wait_result = child.wait() => {
7064                let status = wait_result.map_err(|source| SuperviseError::Wait {
7065                    module_id: module_id.to_string(),
7066                    source,
7067                })?;
7068                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
7069                    snapshot,
7070                    child,
7071                    &status,
7072                )));
7073            }
7074            _ = sleep(poll) => {}
7075        }
7076    }
7077}
7078
7079fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
7080    // A replacement process that exits before HELLO did not provide service, even
7081    // if it used status 0. Count it against the restart cap as a new-binary failure.
7082    if exit_report.kind != ExitKind::DeliberateSeverance {
7083        exit_report.kind = ExitKind::Crash;
7084    }
7085    exit_report
7086}
7087
7088async fn handle_reload_child_registration_failure(
7089    spec: &ModuleSpec,
7090    runtime: &SupervisorRuntimeConfig,
7091    registry: &Registry,
7092    process_liveness: &SupervisorProcessLiveness,
7093    snapshot: &SharedSnapshot,
7094    child: &mut Option<SupervisedChild>,
7095    failure: ReloadRegistrationFailure,
7096) -> Result<(), SuperviseError> {
7097    let ReloadRegistrationFailure {
7098        exit_report,
7099        reason,
7100    } = failure;
7101    match on_child_exit(
7102        spec,
7103        runtime.restart_policy,
7104        registry,
7105        snapshot,
7106        &runtime.terminal_ring,
7107        &runtime.spawn_events,
7108        &runtime.child_roster,
7109        exit_report,
7110    )
7111    .await
7112    {
7113        NextAction::Stop {
7114            registration_released,
7115        } => {
7116            if registration_released {
7117                process_liveness.untrack_if_current(&spec.module_id, snapshot);
7118            }
7119        }
7120        NextAction::Restart { schedule } => {
7121            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
7122                schedule.delay
7123            });
7124            if let Some(schedule) = schedule {
7125                log_crash_respawn(&spec.module_id, schedule);
7126            }
7127            sleep(delay).await;
7128            // A disable or drain that landed during the backoff cancels this
7129            // policy retry: the operator's stop must win over the respawn the
7130            // sleep counted down to.
7131            if respawn_still_pending(snapshot) {
7132                if let Err(err) = wait_for_registration_release(
7133                    registry,
7134                    &spec.module_id,
7135                    REGISTRY_RELEASE_TIMEOUT,
7136                )
7137                .await
7138                {
7139                    fail_snapshot(snapshot, Some(&spec.module_id), None);
7140                    process_liveness.untrack_if_current(&spec.module_id, snapshot);
7141                    return Err(SuperviseError::ReloadFailed {
7142                        module_id: spec.module_id.clone(),
7143                        reason: format!(
7144                            "{reason}; registration did not release before policy retry: {err}"
7145                        ),
7146                    });
7147                }
7148                process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7149                match spawn_and_mark_running(spec, runtime, snapshot) {
7150                    Ok(next_child) => {
7151                        *child = Some(next_child);
7152                    }
7153                    Err(err) => {
7154                        fail_snapshot(snapshot, Some(&spec.module_id), None);
7155                        process_liveness.untrack_if_current(&spec.module_id, snapshot);
7156                        return Err(SuperviseError::ReloadFailed {
7157                            module_id: spec.module_id.clone(),
7158                            reason: format!("{reason}; policy retry spawn failed: {err}"),
7159                        });
7160                    }
7161                }
7162            }
7163        }
7164    }
7165
7166    Err(SuperviseError::ReloadFailed {
7167        module_id: spec.module_id.clone(),
7168        reason,
7169    })
7170}
7171
7172async fn handle_reload_spawn_failure(
7173    spec: &ModuleSpec,
7174    runtime: &SupervisorRuntimeConfig,
7175    process_liveness: &SupervisorProcessLiveness,
7176    snapshot: &SharedSnapshot,
7177    child: &mut Option<SupervisedChild>,
7178    reason: String,
7179) -> Result<(), SuperviseError> {
7180    let mut should_retry = false;
7181    let now = Instant::now();
7182    update_snapshot(snapshot, Some(&spec.module_id), |state| {
7183        clear_current_process_facts(state);
7184        if daemon_will_restart(state, &runtime.restart_policy, now) {
7185            state.record_crash_restart(&runtime.restart_policy, now);
7186            state.state = ModuleState::Restarting;
7187            should_retry = true;
7188        } else if state.enabled {
7189            state.state = ModuleState::Failed;
7190        } else {
7191            state.state = ModuleState::Disabled;
7192        }
7193    })?;
7194
7195    if should_retry {
7196        sleep(runtime.restart_policy.backoff).await;
7197        // A disable or drain that landed during the backoff cancels this
7198        // policy retry: the operator's stop must win over the respawn the
7199        // sleep counted down to.
7200        if respawn_still_pending(snapshot) {
7201            process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7202            match spawn_and_mark_running(spec, runtime, snapshot) {
7203                Ok(next_child) => {
7204                    *child = Some(next_child);
7205                }
7206                Err(err) => {
7207                    fail_snapshot(snapshot, Some(&spec.module_id), None);
7208                    process_liveness.untrack_if_current(&spec.module_id, snapshot);
7209                    return Err(SuperviseError::ReloadFailed {
7210                        module_id: spec.module_id.clone(),
7211                        reason: format!("{reason}; policy retry spawn failed: {err}"),
7212                    });
7213                }
7214            }
7215        }
7216    } else {
7217        process_liveness.untrack_if_current(&spec.module_id, snapshot);
7218    }
7219
7220    Err(SuperviseError::ReloadFailed {
7221        module_id: spec.module_id.clone(),
7222        reason,
7223    })
7224}
7225
7226fn control_flags() -> Flags {
7227    Flags::new(false, Priority::Passive, false)
7228}
7229
7230#[allow(clippy::too_many_arguments)]
7231async fn drain_optional_child(
7232    module_id: &str,
7233    protocol: ModuleProtocol,
7234    stop_notice: StopNotice,
7235    registry: &Registry,
7236    snapshot: &SharedSnapshot,
7237    terminal_ring: &Arc<Mutex<TerminalRing>>,
7238    spawn_events: &SpawnEventFeed,
7239    child: &mut Option<SupervisedChild>,
7240    drain_timeout: Duration,
7241    final_state: ModuleState,
7242    enabled: Option<bool>,
7243) -> Result<(), SuperviseError> {
7244    if let Some(child) = child.take() {
7245        drain_child_to_state(
7246            module_id,
7247            protocol,
7248            stop_notice,
7249            registry,
7250            snapshot,
7251            terminal_ring,
7252            spawn_events,
7253            child,
7254            drain_timeout,
7255            final_state,
7256            enabled,
7257        )
7258        .await
7259    } else {
7260        update_snapshot(snapshot, Some(module_id), |state| {
7261            state.state = final_state;
7262            if let Some(enabled) = enabled {
7263                state.enabled = enabled;
7264            }
7265            clear_current_process_facts(state);
7266        })?;
7267        wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7268    }
7269}
7270
7271#[allow(clippy::too_many_arguments)]
7272async fn drain_child_to_state(
7273    module_id: &str,
7274    protocol: ModuleProtocol,
7275    stop_notice: StopNotice,
7276    registry: &Registry,
7277    snapshot: &SharedSnapshot,
7278    terminal_ring: &Arc<Mutex<TerminalRing>>,
7279    spawn_events: &SpawnEventFeed,
7280    mut child: SupervisedChild,
7281    drain_timeout: Duration,
7282    final_state: ModuleState,
7283    enabled: Option<bool>,
7284) -> Result<(), SuperviseError> {
7285    update_snapshot(snapshot, Some(module_id), |state| {
7286        state.state = ModuleState::Draining;
7287        state.draining_to_replace = final_state == ModuleState::Restarting;
7288        if let Some(enabled) = enabled {
7289            state.enabled = enabled;
7290        }
7291    })?;
7292
7293    // The wait below is the same budget in every case; what differs is
7294    // whether anything has ASKED the child to stop before it starts. Only a
7295    // forwarding drain that reached the module's registered connection has
7296    // (`module.draining`, then a module GOODBYE). Every other child was told
7297    // nothing: a `protocol: "none"` module, which never registers; a subc
7298    // module spawned moments ago that has not sent HELLO yet; or a stop that
7299    // runs no forwarding drain. Without a signal the budget is only a delay
7300    // in front of SIGKILL -- and the not-yet-registered child is the worst
7301    // case, because it registers into a module that is already draining,
7302    // is never told, and is killed while healthy.
7303    if stop_notice != StopNotice::SentOverConnection {
7304        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
7305            info!(
7306                module_id,
7307                pid = child.pid,
7308                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7309                "module has no connection yet; requesting stop by signal"
7310            );
7311        }
7312        request_graceful_stop(module_id, &child);
7313    }
7314
7315    let exit_report = match timeout(drain_timeout, child.wait()).await {
7316        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
7317        Ok(Err(source)) => {
7318            fail_snapshot(snapshot, Some(module_id), None);
7319            return Err(SuperviseError::Wait {
7320                module_id: module_id.to_string(),
7321                source,
7322            });
7323        }
7324        Err(_) => {
7325            // Mirror the sibling arm above: state is already `Draining`, and an
7326            // error propagated from here would strand it there -- a state
7327            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
7328            // `Failed | Stopped`), leaving an operator Restart as the only exit.
7329            // `Failed` before `?` keeps the module operator-visible and
7330            // revivable. Trigger is an ESRCH race (process exits between the
7331            // drain timeout firing and the kill) or a post-kill wait failure
7332            // (issue #34).
7333            //
7334            // Logged because the kill is otherwise visible only as signal 9 in
7335            // the terminal ring, and the budget it follows can be long enough
7336            // that consumers see a stretch of refusals with no stated cause.
7337            warn!(
7338                module_id,
7339                pid = child.pid,
7340                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7341                reason = ?final_state,
7342                ?stop_notice,
7343                "drain budget expired before the module exited; killing it"
7344            );
7345            child.start_kill().map_err(|source| {
7346                fail_snapshot(snapshot, Some(module_id), None);
7347                SuperviseError::Kill {
7348                    module_id: module_id.to_string(),
7349                    source,
7350                }
7351            })?;
7352            let status = child.wait().await.map_err(|source| {
7353                fail_snapshot(snapshot, Some(module_id), None);
7354                SuperviseError::Wait {
7355                    module_id: module_id.to_string(),
7356                    source,
7357                }
7358            })?;
7359            classify_reaped_child_exit(snapshot, &child, &status)
7360        }
7361    };
7362
7363    update_snapshot(snapshot, Some(module_id), |state| {
7364        state.state = final_state;
7365        if let Some(enabled) = enabled {
7366            state.enabled = enabled;
7367        }
7368        clear_current_process_facts(state);
7369        state.last_exit = Some(exit_report.clone());
7370        if exit_report.kind == ExitKind::DeliberateSeverance {
7371            state.lifetime_restarts += 1;
7372        }
7373    })?;
7374    record_terminal(
7375        module_id,
7376        terminal_ring,
7377        spawn_events,
7378        &exit_report,
7379        terminal_disposition(final_state),
7380    );
7381    child.drain_stderr(module_id).await;
7382
7383    wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7384}
7385
7386/// Ask a child that nothing else has asked to stop, by signal.
7387///
7388/// A registered subc module is asked over its own connection: the drain sends
7389/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
7390/// module GOODBYE, and the module stops itself. A module that speaks no subc
7391/// wire receives none of that, and neither does a subc module that has not
7392/// registered yet, so for them the drain budget would be pure delay in front of
7393/// a SIGKILL -- and for a process with a store to flush (JetStream is the
7394/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
7395/// into a recovery on the next start.
7396///
7397/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
7398/// rule rather than an optimisation: that module's graceful stop is already
7399/// running by the time its child is drained, and a signal would race it.
7400///
7401/// Best-effort by construction. A child that has already exited is the ordinary
7402/// case rather than an error (the kill lands on a reaped or exiting pid), so a
7403/// failure is logged at debug and the wait-then-kill below still decides the
7404/// outcome.
7405#[cfg(unix)]
7406fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
7407    let Some(pid) = child
7408        .id()
7409        .and_then(|pid| i32::try_from(pid).ok())
7410        .and_then(rustix::process::Pid::from_raw)
7411    else {
7412        debug!(
7413            module_id,
7414            "no pid to signal for teardown; falling through to the drain wait"
7415        );
7416        return;
7417    };
7418    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
7419        Ok(()) => debug!(
7420            module_id,
7421            "sent SIGTERM to a module nothing else asked to stop"
7422        ),
7423        Err(err) => debug!(
7424            module_id,
7425            error = %err,
7426            "SIGTERM to module failed; the drain wait and kill still apply"
7427        ),
7428    }
7429}
7430
7431/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
7432/// Windows does offer need cooperation this supervisor cannot assume: a console
7433/// control event requires sharing a console with the child, and `WM_CLOSE`
7434/// requires the child to pump a message loop. A supervised server process does
7435/// neither, so there is nothing to send and teardown is the wait followed by the
7436/// kill. Emulating a signal here would mean inventing a stop protocol, which is
7437/// the thing `protocol: "none"` exists to avoid.
7438#[cfg(not(unix))]
7439fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
7440    debug!(
7441        module_id,
7442        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
7443    );
7444}
7445
7446fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
7447    match final_state {
7448        ModuleState::Stopped => TerminalDisposition::Stopped,
7449        ModuleState::Disabled => TerminalDisposition::Disabled,
7450        ModuleState::Restarting => TerminalDisposition::Restarting,
7451        ModuleState::Failed => TerminalDisposition::Failed,
7452        ModuleState::Starting
7453        | ModuleState::Running
7454        | ModuleState::Unresponsive
7455        | ModuleState::Draining => {
7456            unreachable!("terminal exits only finish in terminal or restarting states")
7457        }
7458    }
7459}
7460
7461/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
7462/// plain stop or restart waits for before it spawns a replacement.
7463async fn wait_for_registration_release(
7464    registry: &Registry,
7465    module_id: &str,
7466    wait: Duration,
7467) -> Result<(), SuperviseError> {
7468    wait_for_slot_registration_release(
7469        registry,
7470        crate::registry::RegistrationSlot::Active(module_id),
7471        wait,
7472    )
7473    .await
7474}
7475
7476/// Wait for the registration in `slot` to go away.
7477///
7478/// Keyed on the slot rather than the bare module id because a successful swap
7479/// never empties the id's active slot (the promoted candidate is in it), so an
7480/// id-keyed wait for the incumbent's release would always time out. Draining a
7481/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
7482/// incumbent's connection instead.
7483async fn wait_for_slot_registration_release(
7484    registry: &Registry,
7485    slot: crate::registry::RegistrationSlot<'_>,
7486    wait: Duration,
7487) -> Result<(), SuperviseError> {
7488    let deadline = Instant::now() + wait;
7489    let mut release_events = registration_release_events().subscribe();
7490    let still_active = |registration: &crate::registry::ModuleRegistration| {
7491        SuperviseError::RegistrationStillActive {
7492            module_id: registration.manifest.module_id.clone(),
7493            waited: wait,
7494        }
7495    };
7496    loop {
7497        let _observed_generation = *release_events.borrow_and_update();
7498        let Some(registration) = registry
7499            .registration(slot)
7500            .map_err(SuperviseError::Registry)?
7501        else {
7502            return Ok(());
7503        };
7504
7505        let now = Instant::now();
7506        if now >= deadline {
7507            return Err(still_active(&registration));
7508        }
7509
7510        let remaining = deadline.saturating_duration_since(now);
7511        match timeout(remaining, release_events.changed()).await {
7512            Ok(Ok(())) | Ok(Err(_)) => {}
7513            Err(_) => return Err(still_active(&registration)),
7514        }
7515    }
7516}
7517
7518#[cfg(test)]
7519mod slot_registration_wait_tests {
7520    use super::*;
7521    use crate::registry::{ConnectionId, RegistrationSlot};
7522    use subc_protocol::manifest::ModuleManifest;
7523
7524    const INCUMBENT: u64 = 1;
7525    const CANDIDATE: u64 = 2;
7526
7527    fn swapped_registry() -> Arc<Registry> {
7528        let registry = Arc::new(Registry::default());
7529        let manifest = ModuleManifest::builder("m", "0.1.0").build();
7530        registry
7531            .register_with_control_ops(
7532                manifest.clone(),
7533                1,
7534                ConnectionId::new(INCUMBENT),
7535                Vec::new(),
7536            )
7537            .unwrap();
7538        registry
7539            .register_candidate_with_control_ops(
7540                manifest,
7541                1,
7542                ConnectionId::new(CANDIDATE),
7543                Vec::new(),
7544            )
7545            .unwrap();
7546        registry
7547    }
7548
7549    /// After a promotion the id's active slot is held by the new process, so an
7550    /// id-keyed wait for the incumbent's release can never succeed; the
7551    /// connection-keyed wait completes as soon as the incumbent deregisters.
7552    #[tokio::test]
7553    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
7554        let registry = swapped_registry();
7555        registry.promote_candidate("m").unwrap().unwrap();
7556
7557        assert!(matches!(
7558            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
7559            Err(SuperviseError::RegistrationStillActive { .. })
7560        ));
7561
7562        // Still held while the incumbent's connection has not deregistered.
7563        assert!(matches!(
7564            wait_for_slot_registration_release(
7565                &registry,
7566                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7567                Duration::from_millis(50),
7568            )
7569            .await,
7570            Err(SuperviseError::RegistrationStillActive { .. })
7571        ));
7572
7573        let releaser = Arc::clone(&registry);
7574        let release = tokio::spawn(async move {
7575            sleep(Duration::from_millis(20)).await;
7576            releaser
7577                .deregister_connection(ConnectionId::new(INCUMBENT))
7578                .unwrap();
7579            notify_registration_release();
7580        });
7581        wait_for_slot_registration_release(
7582            &registry,
7583            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7584            Duration::from_secs(5),
7585        )
7586        .await
7587        .expect("the incumbent's own registration is released");
7588        release.await.unwrap();
7589        assert!(registry.get_module("m").unwrap().is_some());
7590    }
7591
7592    /// The candidate slot is waited on separately from the active slot: the
7593    /// incumbent's registration neither holds up nor stands in for it.
7594    #[tokio::test]
7595    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
7596        let registry = swapped_registry();
7597        assert!(matches!(
7598            wait_for_slot_registration_release(
7599                &registry,
7600                RegistrationSlot::Candidate("m"),
7601                Duration::from_millis(50),
7602            )
7603            .await,
7604            Err(SuperviseError::RegistrationStillActive { .. })
7605        ));
7606        registry
7607            .deregister_connection(ConnectionId::new(CANDIDATE))
7608            .unwrap();
7609        wait_for_slot_registration_release(
7610            &registry,
7611            RegistrationSlot::Candidate("m"),
7612            Duration::from_millis(50),
7613        )
7614        .await
7615        .expect("a candidate slot with no candidate is released");
7616        assert!(registry
7617            .registration(RegistrationSlot::Active("m"))
7618            .unwrap()
7619            .is_some());
7620    }
7621}
7622
7623fn classify_exit(status: &ExitStatus) -> ExitReport {
7624    ExitReport {
7625        kind: if status.success() {
7626            ExitKind::Clean
7627        } else {
7628            ExitKind::Crash
7629        },
7630        code: status.code(),
7631        signal: exit_signal(status),
7632        at_ms: unix_ms_now(),
7633    }
7634}
7635
7636/// The terminal record for a module whose `wait()` call itself errored (e.g. the
7637/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
7638/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
7639/// disposition still must be `Failed` so the terminal ring is not silently missing
7640/// an entry, matching what `fail_snapshot` records for this same arm.
7641fn wait_error_exit_report() -> ExitReport {
7642    ExitReport {
7643        kind: ExitKind::Crash,
7644        code: None,
7645        signal: None,
7646        at_ms: unix_ms_now(),
7647    }
7648}
7649
7650#[cfg(unix)]
7651fn exit_signal(status: &ExitStatus) -> Option<i32> {
7652    use std::os::unix::process::ExitStatusExt;
7653
7654    status.signal()
7655}
7656
7657#[cfg(not(unix))]
7658fn exit_signal(_status: &ExitStatus) -> Option<i32> {
7659    None
7660}
7661
7662/// Give an operator-touched module its full crash budget back.
7663///
7664/// Named for the counter it used to zero; it now empties the in-window ring,
7665/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
7666/// ledger of what happened survives every operator action.
7667fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
7668    update_snapshot(snapshot, Some(module_id), |state| {
7669        state.clear_crash_restarts();
7670    })
7671}
7672
7673fn set_running(
7674    snapshot: &SharedSnapshot,
7675    child: &SupervisedChild,
7676    module_id: &str,
7677    spawn_events: &SpawnEventFeed,
7678) -> Result<(), SuperviseError> {
7679    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7680        module_id: Some(module_id.to_string()),
7681    })?;
7682    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
7683    // Every caller of this is a plain spawn, which always uses the primary key;
7684    // a promoted swap candidate sets the flag itself after this returns.
7685    state.in_alternate_slot = false;
7686    state.configuration_updated_since_spawn = false;
7687    state.state = ModuleState::Running;
7688    state.enabled = true;
7689    state.process_alive = true;
7690    state.pid = child.id();
7691    state.spawned_at_ms = Some(child.spawned_at_ms);
7692    state.spawned_from = Some(child.spawned_from.clone());
7693    state.spawned_file_identity = child.spawned_file_identity;
7694    state.process_start_time = child.process_start_time;
7695    Ok(())
7696}
7697
7698fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
7699    state.process_alive = false;
7700    state.pid = None;
7701    state.spawned_at_ms = None;
7702    state.spawned_from = None;
7703    state.spawned_file_identity = None;
7704    state.process_start_time = None;
7705    state.deliberate_severance = None;
7706}
7707
7708#[cfg(test)]
7709fn record_deliberate_severance(
7710    snapshot: &SharedSnapshot,
7711    identity: ProcessIdentity,
7712) -> Result<(), SuperviseError> {
7713    update_snapshot(snapshot, None, |state| {
7714        state.deliberate_severance = Some(identity);
7715    })
7716}
7717
7718fn apply_deliberate_severance_marker(
7719    snapshot: &SharedSnapshot,
7720    exited_identity: Option<ProcessIdentity>,
7721    mut exit_report: ExitReport,
7722) -> ExitReport {
7723    let marker = lock_snapshot(snapshot)
7724        .ok()
7725        .and_then(|mut state| state.deliberate_severance.take());
7726    if marker.is_some() && marker == exited_identity {
7727        exit_report.kind = ExitKind::DeliberateSeverance;
7728    }
7729    exit_report
7730}
7731
7732fn classify_reaped_child_exit(
7733    snapshot: &SharedSnapshot,
7734    child: &SupervisedChild,
7735    status: &ExitStatus,
7736) -> ExitReport {
7737    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
7738}
7739
7740fn fail_snapshot(
7741    snapshot: &SharedSnapshot,
7742    module_id: Option<&str>,
7743    last_exit: Option<ExitReport>,
7744) {
7745    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
7746        state.state = ModuleState::Failed;
7747        clear_current_process_facts(state);
7748        if let Some(last_exit) = last_exit {
7749            state.last_exit = Some(last_exit);
7750        }
7751    }) {
7752        error!(error = %err, "failed to mark supervisor state failed");
7753    }
7754}
7755
7756fn update_snapshot(
7757    snapshot: &SharedSnapshot,
7758    module_id: Option<&str>,
7759    update: impl FnOnce(&mut SupervisorSnapshot),
7760) -> Result<(), SuperviseError> {
7761    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7762        module_id: module_id.map(ToOwned::to_owned),
7763    })?;
7764    update(&mut state);
7765    Ok(())
7766}
7767
7768const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
7769
7770fn lock_snapshot_for_control<'a>(
7771    snapshot: &'a SharedSnapshot,
7772    module_id: &str,
7773    caller: &'static str,
7774) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
7775    let started_at = Instant::now();
7776    let guard = lock_snapshot(snapshot)?;
7777    let waited = started_at.elapsed();
7778    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
7779        warn!(
7780            module_id = %module_id,
7781            waited_ms = waited.as_millis() as u64,
7782            caller = %caller,
7783            "slow snapshot lock"
7784        );
7785    }
7786    Ok(guard)
7787}
7788
7789fn lock_snapshot(
7790    snapshot: &SharedSnapshot,
7791) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
7792    snapshot
7793        .lock()
7794        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
7795}
7796
7797#[cfg(test)]
7798mod terminal_history_tests {
7799    use std::{
7800        path::PathBuf,
7801        sync::Arc,
7802        time::{Duration, Instant},
7803    };
7804
7805    use tokio::time::sleep;
7806
7807    use super::{
7808        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
7809        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
7810        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
7811        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
7812        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
7813        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
7814        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
7815    };
7816    // The supervisor's clock, distinct from the `std::time::Instant` these tests
7817    // use for their own wall-clock deadlines: crash-restart instants must be on
7818    // the same clock the production code stamps them with, which is tokio's (and
7819    // is what `start_paused` tests can move).
7820    use super::Instant as ClockInstant;
7821    use crate::{
7822        registry::Registry,
7823        terminal_ring::{TerminalRing, TerminalRingConfig},
7824    };
7825    use std::sync::Mutex;
7826    use subc_control::TerminalDisposition;
7827
7828    /// See the twin in `control.rs` for why this derives the path from
7829    /// `current_exe()` and why the existence check is here: `--lib` alone does
7830    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
7831    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
7832    fn fake_aft_stub_path() -> PathBuf {
7833        let mut path = std::env::current_exe().expect("current_exe available in tests");
7834        path.pop();
7835        path.pop();
7836        path.push(if cfg!(windows) {
7837            "fake-aft-stub.exe"
7838        } else {
7839            "fake-aft-stub"
7840        });
7841        assert!(
7842            path.exists(),
7843            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
7844             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
7845            path.display()
7846        );
7847        path
7848    }
7849
7850    #[test]
7851    fn reserved_never_spawned_refuses_every_hello() {
7852        // The canary hole: a reserved id whose module has never spawned had NO
7853        // gate entry and admitted anyone -- the reservation protected the nonce
7854        // holder, not the NAME. Now the entry is present with no legitimate
7855        // holder and refuses all comers.
7856        let supervisor = SupervisorHandle::default();
7857        supervisor.apply_identity_configuration(&ModuleSpec {
7858            module_id: "never-spawned".to_string(),
7859            program: PathBuf::from("/usr/bin/false"),
7860            args: Vec::new(),
7861            env: Vec::new(),
7862            reserved: true,
7863            reserved_prefixes: Vec::new(),
7864            protocol: ModuleProtocol::Subc,
7865            overlap: Default::default(),
7866        });
7867        assert!(
7868            supervisor
7869                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
7870                .is_some(),
7871            "forged nonce must refuse on a reserved never-spawned id"
7872        );
7873        assert!(
7874            supervisor
7875                .reserved_hello_rejection("never-spawned", None)
7876                .is_some(),
7877            "absent nonce must refuse on a reserved never-spawned id"
7878        );
7879        // And a real spawn nonce minted later admits exactly that nonce.
7880        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
7881        supervisor.apply_identity_configuration(&ModuleSpec {
7882            module_id: "never-spawned".to_string(),
7883            program: PathBuf::from("/usr/bin/false"),
7884            args: Vec::new(),
7885            env: Vec::new(),
7886            reserved: true,
7887            reserved_prefixes: Vec::new(),
7888            protocol: ModuleProtocol::Subc,
7889            overlap: Default::default(),
7890        });
7891        assert!(supervisor
7892            .reserved_hello_rejection("never-spawned", Some("minted"))
7893            .is_none());
7894        assert!(supervisor
7895            .reserved_hello_rejection("never-spawned", Some("forged"))
7896            .is_some());
7897    }
7898
7899    /// Put `count` crash restarts on a snapshot's ring as if they had all just
7900    /// happened, which is what "spent budget" looks like to every reader.
7901    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
7902        let now = ClockInstant::now();
7903        for _ in 0..count {
7904            state.crash_restarts.push_back(now);
7905        }
7906    }
7907
7908    /// Age the oldest recorded restart out of `window`, standing in for the hours
7909    /// that would otherwise have to pass. Injecting the instant is the point: a
7910    /// test that slept a real window would take ten minutes and still prove less.
7911    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
7912        let aged = state
7913            .crash_restarts
7914            .front()
7915            .expect("a crash restart must be recorded before it can be aged")
7916            .checked_sub(window + Duration::from_secs(1))
7917            .expect("the test clock is far enough from its origin to age an instant");
7918        state.crash_restarts[0] = aged;
7919    }
7920
7921    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
7922        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
7923        seed_crash_restarts(&mut state, count);
7924        state
7925    }
7926
7927    #[test]
7928    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
7929        let policy = RestartPolicy::new(3, Duration::ZERO);
7930        let now = ClockInstant::now();
7931        assert!(daemon_will_restart(
7932            &mut snapshot_with_restarts(true, 2),
7933            &policy,
7934            now
7935        ));
7936        assert!(!daemon_will_restart(
7937            &mut snapshot_with_restarts(true, 3),
7938            &policy,
7939            now
7940        ));
7941        assert!(!daemon_will_restart(
7942            &mut snapshot_with_restarts(false, 0),
7943            &policy,
7944            now
7945        ));
7946    }
7947
7948    #[test]
7949    fn crash_restart_backoff_escalates_with_in_window_count() {
7950        let policy = RestartPolicy::new(4, Duration::from_millis(100))
7951            .with_max_backoff(Duration::from_secs(30));
7952        let now = ClockInstant::now();
7953        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7954        let schedules = (0..4)
7955            .map(|_| {
7956                state
7957                    .next_crash_restart(&policy, now)
7958                    .expect("the test policy allows four crash restarts")
7959            })
7960            .collect::<Vec<_>>();
7961
7962        assert_eq!(
7963            schedules
7964                .iter()
7965                .map(|schedule| schedule.restart_in_window)
7966                .collect::<Vec<_>>(),
7967            vec![0, 1, 2, 3]
7968        );
7969        assert_eq!(
7970            schedules
7971                .iter()
7972                .map(|schedule| schedule.delay)
7973                .collect::<Vec<_>>(),
7974            vec![
7975                Duration::from_millis(100),
7976                Duration::from_secs(1),
7977                Duration::from_secs(10),
7978                Duration::from_secs(30),
7979            ]
7980        );
7981    }
7982
7983    #[test]
7984    fn crash_restart_backoff_resets_after_ring_clear() {
7985        let policy = RestartPolicy::new(3, Duration::from_millis(100));
7986        let now = ClockInstant::now();
7987        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7988        assert_eq!(
7989            state.next_crash_restart(&policy, now).unwrap().delay,
7990            Duration::from_millis(100)
7991        );
7992        assert_eq!(
7993            state.next_crash_restart(&policy, now).unwrap().delay,
7994            Duration::from_secs(1)
7995        );
7996
7997        state.clear_crash_restarts();
7998        let schedule = state
7999            .next_crash_restart(&policy, now)
8000            .expect("a cleared ring must allow another restart");
8001        assert_eq!(schedule.restart_in_window, 0);
8002        assert_eq!(schedule.delay, Duration::from_millis(100));
8003    }
8004
8005    #[test]
8006    fn crash_restart_backoff_ignores_aged_restarts() {
8007        let policy = RestartPolicy::new(3, Duration::from_millis(100));
8008        let now = ClockInstant::now();
8009        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
8010        state
8011            .next_crash_restart(&policy, now)
8012            .expect("the first restart is allowed");
8013        state
8014            .next_crash_restart(&policy, now)
8015            .expect("the second restart is allowed");
8016        state.crash_restarts[0] = now
8017            .checked_sub(policy.window + Duration::from_secs(1))
8018            .expect("the fake clock can age a restart past the window");
8019
8020        let schedule = state
8021            .next_crash_restart(&policy, now)
8022            .expect("an aged restart must release its slot");
8023        assert_eq!(schedule.restart_in_window, 1);
8024        assert_eq!(schedule.delay, Duration::from_secs(1));
8025        assert_eq!(state.crash_restarts.len(), 2);
8026    }
8027
8028    /// The budget is a rate: the same three spent restarts refuse a respawn
8029    /// while they are recent and allow one once they have aged past the window.
8030    /// Nothing about the module changed in between, which is the whole point.
8031    #[test]
8032    fn a_budget_spent_before_the_window_no_longer_refuses() {
8033        let policy = RestartPolicy::new(3, Duration::ZERO);
8034        let mut state = snapshot_with_restarts(true, 3);
8035        let now = ClockInstant::now();
8036        assert!(!daemon_will_restart(&mut state, &policy, now));
8037
8038        assert!(daemon_will_restart(
8039            &mut state,
8040            &policy,
8041            now + policy.window + Duration::from_secs(1)
8042        ));
8043        assert!(
8044            state.crash_restarts.is_empty(),
8045            "reading the budget must drop the instants that left the window"
8046        );
8047    }
8048
8049    fn module_with_recovery_snapshot(
8050        state: ModuleState,
8051        enabled: bool,
8052        restart_count: u32,
8053    ) -> SupervisedModule {
8054        let registry = Arc::new(Registry::default());
8055        let supervisor =
8056            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
8057        let module = supervisor
8058            .spawn(ModuleSpec {
8059                module_id: "recovery-snapshot".to_string(),
8060                program: fake_aft_stub_path(),
8061                args: Vec::new(),
8062                env: Vec::new(),
8063                reserved: false,
8064                reserved_prefixes: Vec::new(),
8065                protocol: ModuleProtocol::Subc,
8066                overlap: Default::default(),
8067            })
8068            .unwrap();
8069        update_snapshot(
8070            &module.inner.snapshot,
8071            Some("recovery-snapshot"),
8072            |snapshot| {
8073                snapshot.state = state;
8074                snapshot.enabled = enabled;
8075                seed_crash_restarts(snapshot, restart_count);
8076            },
8077        )
8078        .unwrap();
8079        module
8080    }
8081
8082    #[cfg(target_os = "linux")]
8083    #[tokio::test]
8084    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
8085        let supervisor = Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
8086            .with_cgroup_placement(None);
8087        let result = supervisor.spawn(ModuleSpec {
8088            module_id: "no-cgroup-placement".to_string(),
8089            program: fake_aft_stub_path(),
8090            args: Vec::new(),
8091            env: Vec::new(),
8092            reserved: false,
8093            reserved_prefixes: Vec::new(),
8094            protocol: ModuleProtocol::Subc,
8095            overlap: Default::default(),
8096        });
8097
8098        assert!(
8099            result.is_ok(),
8100            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
8101        );
8102    }
8103
8104    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8105    async fn undecided_snapshot_uses_shared_restart_predicate() {
8106        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
8107            .will_recover_after_connection_loss()
8108            .unwrap());
8109        assert!(
8110            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
8111                .will_recover_after_connection_loss()
8112                .unwrap()
8113        );
8114    }
8115
8116    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8117    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
8118        assert!(
8119            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
8120                .will_recover_after_connection_loss()
8121                .unwrap()
8122        );
8123    }
8124
8125    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8126    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
8127        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
8128            .will_recover_after_connection_loss()
8129            .unwrap());
8130        assert!(
8131            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
8132                .will_recover_after_connection_loss()
8133                .unwrap()
8134        );
8135    }
8136
8137    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8138    async fn warming_snapshot_is_limited_to_startup_phases() {
8139        for state in [
8140            ModuleState::Starting,
8141            ModuleState::Running,
8142            ModuleState::Restarting,
8143        ] {
8144            assert!(
8145                module_with_recovery_snapshot(state, true, 0)
8146                    .is_warming()
8147                    .unwrap(),
8148                "{state:?} should be warming"
8149            );
8150        }
8151        for state in [
8152            ModuleState::Unresponsive,
8153            ModuleState::Draining,
8154            ModuleState::Stopped,
8155            ModuleState::Failed,
8156            ModuleState::Disabled,
8157        ] {
8158            assert!(
8159                !module_with_recovery_snapshot(state, true, 0)
8160                    .is_warming()
8161                    .unwrap(),
8162                "{state:?} should not be warming"
8163            );
8164        }
8165    }
8166
8167    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8168    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
8169        let registry = Arc::new(Registry::default());
8170        let supervisor =
8171            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
8172        let module = supervisor
8173            .spawn(ModuleSpec {
8174                module_id: "terminal-history".to_string(),
8175                program: fake_aft_stub_path(),
8176                args: Vec::new(),
8177                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8178                reserved: false,
8179                reserved_prefixes: Vec::new(),
8180                protocol: ModuleProtocol::Subc,
8181                overlap: Default::default(),
8182            })
8183            .unwrap();
8184
8185        let deadline = Instant::now() + Duration::from_secs(5);
8186        loop {
8187            let history = module.terminal_history();
8188            if history.entries.len() == 2 {
8189                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
8190                assert_eq!(history.dropped, 0);
8191                assert_eq!(
8192                    history
8193                        .entries
8194                        .iter()
8195                        .map(|entry| entry.exit_code)
8196                        .collect::<Vec<_>>(),
8197                    vec![Some(23), Some(23)]
8198                );
8199                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
8200                return;
8201            }
8202            assert!(
8203                Instant::now() < deadline,
8204                "module did not retain two terminal exits: {history:?}"
8205            );
8206            sleep(Duration::from_millis(10)).await;
8207        }
8208    }
8209
8210    /// A disable issued while a crash respawn is still backing off must preempt
8211    /// that respawn: the operator's stop wins, the disable must not queue behind
8212    /// the backoff, and the module must never come back up afterwards.
8213    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8214    async fn disable_during_crash_backoff_cancels_pending_respawn() {
8215        let backoff = Duration::from_secs(2);
8216        let supervisor = Supervisor::new(
8217            Arc::new(Registry::default()),
8218            RestartPolicy::new(10, backoff),
8219        );
8220        let module = supervisor
8221            .spawn(ModuleSpec {
8222                module_id: "disable-during-backoff".to_string(),
8223                program: fake_aft_stub_path(),
8224                args: Vec::new(),
8225                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8226                reserved: false,
8227                reserved_prefixes: Vec::new(),
8228                protocol: ModuleProtocol::Subc,
8229                overlap: Default::default(),
8230            })
8231            .unwrap();
8232
8233        // Wait for the first crash to put the module into its backoff window.
8234        let deadline = Instant::now() + Duration::from_secs(5);
8235        loop {
8236            if module.status().unwrap().state == ModuleState::Restarting {
8237                break;
8238            }
8239            assert!(
8240                Instant::now() < deadline,
8241                "module never entered the crash backoff"
8242            );
8243            sleep(Duration::from_millis(10)).await;
8244        }
8245
8246        let started = Instant::now();
8247        module.set_enabled(false).await.unwrap();
8248        let waited = started.elapsed();
8249
8250        assert!(
8251            waited < backoff / 2,
8252            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
8253        );
8254        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
8255
8256        // Outlast the backoff: the respawn it was counting down to must never run.
8257        sleep(backoff + Duration::from_millis(500)).await;
8258        let status = module.status().unwrap();
8259        assert_eq!(status.state, ModuleState::Disabled);
8260        assert_eq!(
8261            status.spawn_generation, 1,
8262            "module respawned after the operator disabled it"
8263        );
8264    }
8265
8266    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
8267    /// the shape of nats-server, the program this rule exists for.
8268    #[cfg(unix)]
8269    fn protocol_none_sigterm_exits_clean_spec(
8270        module_id: &str,
8271        dir: &std::path::Path,
8272    ) -> (ModuleSpec, PathBuf, PathBuf) {
8273        let ready = dir.join("ready");
8274        let marker = dir.join("sigterm");
8275        let spec = ModuleSpec {
8276            module_id: module_id.to_string(),
8277            program: fake_aft_stub_path(),
8278            args: Vec::new(),
8279            env: vec![
8280                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
8281                (
8282                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
8283                    marker.display().to_string(),
8284                ),
8285                (
8286                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
8287                    ready.display().to_string(),
8288                ),
8289            ],
8290            reserved: false,
8291            reserved_prefixes: Vec::new(),
8292            protocol: ModuleProtocol::None,
8293            overlap: Default::default(),
8294        };
8295        (spec, ready, marker)
8296    }
8297
8298    /// Wait for a file the child writes, so a signal is never sent before the
8299    /// child's SIGTERM handler is installed (the default disposition would
8300    /// kill it by signal and the exit would not be clean).
8301    #[cfg(unix)]
8302    async fn wait_for_file(path: &std::path::Path) {
8303        let deadline = Instant::now() + Duration::from_secs(10);
8304        while !path.exists() {
8305            assert!(
8306                Instant::now() < deadline,
8307                "{} never appeared",
8308                path.display()
8309            );
8310            sleep(Duration::from_millis(10)).await;
8311        }
8312    }
8313
8314    /// A protocol-none module that exits 0 because something OUTSIDE the
8315    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
8316    /// the crash-path disposition rather than `stopped`.
8317    #[cfg(unix)]
8318    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8319    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
8320        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
8321        let (spec, ready, marker) =
8322            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
8323        let supervisor = Supervisor::new(
8324            Arc::new(Registry::default()),
8325            RestartPolicy::new(3, Duration::ZERO),
8326        );
8327        let module = supervisor.spawn(spec).unwrap();
8328        wait_for_file(&ready).await;
8329        let first_pid = module
8330            .status()
8331            .unwrap()
8332            .pid
8333            .expect("a running module reports its pid");
8334
8335        rustix::process::kill_process(
8336            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
8337            rustix::process::Signal::TERM,
8338        )
8339        .unwrap();
8340
8341        let deadline = Instant::now() + Duration::from_secs(10);
8342        let respawned = loop {
8343            let status = module.status().unwrap();
8344            if status.state == ModuleState::Running
8345                && status.pid.is_some_and(|pid| pid != first_pid)
8346            {
8347                break status;
8348            }
8349            assert!(
8350                Instant::now() < deadline,
8351                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
8352            );
8353            sleep(Duration::from_millis(10)).await;
8354        };
8355        assert_eq!(respawned.spawn_generation, 2);
8356        assert!(
8357            marker.exists(),
8358            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
8359        );
8360
8361        let history = module.terminal_history();
8362        assert_eq!(history.entries.len(), 1, "{history:?}");
8363        let entry = &history.entries[0];
8364        assert_eq!(entry.exit_code, Some(0));
8365        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
8366        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
8367
8368        module.stop().await.unwrap();
8369    }
8370
8371    /// Repeated unrequested clean exits of a protocol-none module spend the
8372    /// restart budget exactly as crashes do, and the module ends `failed` with
8373    /// the budget named.
8374    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8375    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
8376        let supervisor = Supervisor::new(
8377            Arc::new(Registry::default()),
8378            RestartPolicy::new(1, Duration::ZERO),
8379        );
8380        let module = supervisor
8381            .spawn(ModuleSpec {
8382                module_id: "none-clean-exit-budget".to_string(),
8383                program: fake_aft_stub_path(),
8384                args: Vec::new(),
8385                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8386                reserved: false,
8387                reserved_prefixes: Vec::new(),
8388                protocol: ModuleProtocol::None,
8389                overlap: Default::default(),
8390            })
8391            .unwrap();
8392
8393        let deadline = Instant::now() + Duration::from_secs(10);
8394        loop {
8395            let status = module.status().unwrap();
8396            if status.state == ModuleState::Failed {
8397                break;
8398            }
8399            assert!(
8400                Instant::now() < deadline,
8401                "module never exhausted its budget: {status:?} {:?}",
8402                module.terminal_history()
8403            );
8404            sleep(Duration::from_millis(10)).await;
8405        }
8406        let history = module.terminal_history();
8407        assert_eq!(
8408            history
8409                .entries
8410                .iter()
8411                .map(|entry| (entry.exit_code, entry.disposition.clone()))
8412                .collect::<Vec<_>>(),
8413            vec![
8414                (Some(0), TerminalDisposition::Restarting),
8415                (Some(0), TerminalDisposition::Failed),
8416            ]
8417        );
8418        let detail = history.entries[1]
8419            .disposition_detail
8420            .as_deref()
8421            .expect("a budget failure names the budget");
8422        assert!(detail.contains("max_restarts=1"), "{detail}");
8423        assert_eq!(module.status().unwrap().spawn_generation, 2);
8424    }
8425
8426    /// A stop the supervisor itself requests still stops a protocol-none
8427    /// module, even though the child answers the SIGTERM with exit 0.
8428    #[cfg(unix)]
8429    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8430    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
8431        for disable in [false, true] {
8432            let label = if disable {
8433                "none-requested-disable"
8434            } else {
8435                "none-requested-stop"
8436            };
8437            let dir = subc_test_support::TestTempDir::new(label);
8438            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
8439            let supervisor = Supervisor::new(
8440                Arc::new(Registry::default()),
8441                RestartPolicy::new(3, Duration::ZERO),
8442            );
8443            let module = supervisor.spawn(spec).unwrap();
8444            wait_for_file(&ready).await;
8445
8446            if disable {
8447                module.set_enabled(false).await.unwrap();
8448            } else {
8449                module.stop().await.unwrap();
8450            }
8451            assert!(
8452                marker.exists(),
8453                "{label}: the child must have left through its SIGTERM handler with exit 0"
8454            );
8455
8456            // Long enough for a zero-backoff respawn to have happened if the
8457            // exit had been treated as a crash.
8458            sleep(Duration::from_millis(500)).await;
8459            let status = module.status().unwrap();
8460            let expected = if disable {
8461                ModuleState::Disabled
8462            } else {
8463                ModuleState::Stopped
8464            };
8465            assert_eq!(status.state, expected, "{label}");
8466            assert_eq!(
8467                status.spawn_generation, 1,
8468                "{label}: respawned after a requested stop"
8469            );
8470            let history = module.terminal_history();
8471            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
8472            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
8473            assert_ne!(
8474                history.entries[0].disposition,
8475                TerminalDisposition::Restarting,
8476                "{label}"
8477            );
8478        }
8479    }
8480
8481    /// A subc-wire module that exits 0 on its own is still a stop: the
8482    /// protocol-none rule must not reach it.
8483    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8484    async fn subc_wire_clean_exit_is_still_a_stop() {
8485        let supervisor = Supervisor::new(
8486            Arc::new(Registry::default()),
8487            RestartPolicy::new(3, Duration::ZERO),
8488        );
8489        let module = supervisor
8490            .spawn(ModuleSpec {
8491                module_id: "wire-clean-exit".to_string(),
8492                program: fake_aft_stub_path(),
8493                args: Vec::new(),
8494                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8495                reserved: false,
8496                reserved_prefixes: Vec::new(),
8497                protocol: ModuleProtocol::Subc,
8498                overlap: Default::default(),
8499            })
8500            .unwrap();
8501
8502        let deadline = Instant::now() + Duration::from_secs(10);
8503        while module.terminal_history().entries.is_empty() {
8504            assert!(Instant::now() < deadline, "module never exited");
8505            sleep(Duration::from_millis(10)).await;
8506        }
8507        // Long enough for a zero-backoff respawn to have happened.
8508        sleep(Duration::from_millis(500)).await;
8509        let status = module.status().unwrap();
8510        assert_eq!(status.state, ModuleState::Stopped);
8511        assert_eq!(status.spawn_generation, 1);
8512        let history = module.terminal_history();
8513        assert_eq!(history.entries.len(), 1, "{history:?}");
8514        assert_eq!(history.entries[0].exit_code, Some(0));
8515        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
8516    }
8517
8518    /// Each restart-producing arm has its own state transition. Keeping their
8519    /// lifetime count assertions adjacent prevents a later new arm from silently
8520    /// spending budget without recording the historical restart.
8521    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8522    async fn every_restart_increment_path_advances_lifetime_count() {
8523        let supervisor = Supervisor::new(
8524            Arc::new(Registry::default()),
8525            RestartPolicy::new(1, Duration::ZERO),
8526        );
8527        let runtime = supervisor.runtime_config();
8528        let spec = ModuleSpec {
8529            module_id: "lifetime-increment-path".to_string(),
8530            program: PathBuf::from("/unused/lifetime-increment-path"),
8531            args: Vec::new(),
8532            env: Vec::new(),
8533            reserved: false,
8534            reserved_prefixes: Vec::new(),
8535            protocol: ModuleProtocol::Subc,
8536            overlap: Default::default(),
8537        };
8538
8539        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8540        assert!(matches!(
8541            on_child_exit(
8542                &spec,
8543                runtime.restart_policy,
8544                &supervisor.registry,
8545                &crash_snapshot,
8546                &runtime.terminal_ring,
8547                &runtime.spawn_events,
8548                &runtime.child_roster,
8549                ExitReport {
8550                    kind: ExitKind::Crash,
8551                    code: Some(1),
8552                    signal: None,
8553                    at_ms: 1,
8554                },
8555            )
8556            .await,
8557            NextAction::Restart { schedule: _ }
8558        ));
8559        let (crash_restarts, crash_lifetime) = {
8560            let state = lock_snapshot(&crash_snapshot).unwrap();
8561            (state.crash_restarts.len(), state.lifetime_restarts)
8562        };
8563        assert_eq!(crash_restarts, 1);
8564        assert_eq!(crash_lifetime, 1);
8565
8566        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8567        let mut health_child = None;
8568        assert!(matches!(
8569            health_restart_child(
8570                &spec,
8571                &runtime,
8572                &supervisor.registry,
8573                &supervisor.process_liveness,
8574                &health_snapshot,
8575                &mut health_child,
8576                SupervisorHealthStatus::Failing,
8577                None,
8578                2,
8579            )
8580            .await,
8581            Err(SuperviseError::Spawn { .. })
8582        ));
8583        let (health_restarts, health_lifetime) = {
8584            let state = lock_snapshot(&health_snapshot).unwrap();
8585            (state.crash_restarts.len(), state.lifetime_restarts)
8586        };
8587        assert_eq!(health_restarts, 1);
8588        assert_eq!(health_lifetime, 1);
8589
8590        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8591        let mut reload_child = None;
8592        assert!(matches!(
8593            handle_reload_spawn_failure(
8594                &spec,
8595                &runtime,
8596                &supervisor.process_liveness,
8597                &reload_snapshot,
8598                &mut reload_child,
8599                "forced reload spawn failure".to_string(),
8600            )
8601            .await,
8602            Err(SuperviseError::ReloadFailed { .. })
8603        ));
8604        let (reload_restarts, reload_lifetime) = {
8605            let state = lock_snapshot(&reload_snapshot).unwrap();
8606            (state.crash_restarts.len(), state.lifetime_restarts)
8607        };
8608        assert_eq!(reload_restarts, 1);
8609        assert_eq!(reload_lifetime, 1);
8610    }
8611
8612    #[tokio::test]
8613    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
8614        let supervisor = Supervisor::new(
8615            Arc::new(Registry::default()),
8616            RestartPolicy::new(3, Duration::ZERO),
8617        );
8618        let runtime = supervisor.runtime_config();
8619        let spec = ModuleSpec {
8620            module_id: "deliberately-severed".to_string(),
8621            program: PathBuf::from("/unused/deliberately-severed"),
8622            args: Vec::new(),
8623            env: Vec::new(),
8624            reserved: false,
8625            reserved_prefixes: Vec::new(),
8626            protocol: ModuleProtocol::Subc,
8627            overlap: Default::default(),
8628        };
8629        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8630        let process = ProcessIdentity {
8631            pid: 41,
8632            start_time: 101,
8633        };
8634        record_deliberate_severance(&snapshot, process).unwrap();
8635        let exit_report = apply_deliberate_severance_marker(
8636            &snapshot,
8637            Some(process),
8638            ExitReport {
8639                kind: ExitKind::Crash,
8640                code: Some(1),
8641                signal: None,
8642                at_ms: 1,
8643            },
8644        );
8645        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
8646
8647        assert!(matches!(
8648            on_child_exit(
8649                &spec,
8650                runtime.restart_policy,
8651                &supervisor.registry,
8652                &snapshot,
8653                &runtime.terminal_ring,
8654                &runtime.spawn_events,
8655                &runtime.child_roster,
8656                exit_report,
8657            )
8658            .await,
8659            NextAction::Restart { schedule: _ }
8660        ));
8661        let state = lock_snapshot(&snapshot).unwrap();
8662        assert_eq!(state.lifetime_restarts, 1);
8663        assert_eq!(state.crash_restarts.len(), 0);
8664    }
8665
8666    #[tokio::test]
8667    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
8668        let supervisor = Supervisor::new(
8669            Arc::new(Registry::default()),
8670            RestartPolicy::new(3, Duration::ZERO),
8671        );
8672        let runtime = supervisor.runtime_config();
8673        let spec = ModuleSpec {
8674            module_id: "genuine-crash".to_string(),
8675            program: PathBuf::from("/unused/genuine-crash"),
8676            args: Vec::new(),
8677            env: Vec::new(),
8678            reserved: false,
8679            reserved_prefixes: Vec::new(),
8680            protocol: ModuleProtocol::Subc,
8681            overlap: Default::default(),
8682        };
8683        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8684
8685        assert!(matches!(
8686            on_child_exit(
8687                &spec,
8688                runtime.restart_policy,
8689                &supervisor.registry,
8690                &snapshot,
8691                &runtime.terminal_ring,
8692                &runtime.spawn_events,
8693                &runtime.child_roster,
8694                ExitReport {
8695                    kind: ExitKind::Crash,
8696                    code: Some(1),
8697                    signal: None,
8698                    at_ms: 1,
8699                },
8700            )
8701            .await,
8702            NextAction::Restart { schedule: _ }
8703        ));
8704        let state = lock_snapshot(&snapshot).unwrap();
8705        assert_eq!(state.lifetime_restarts, 1);
8706        assert_eq!(state.crash_restarts.len(), 1);
8707    }
8708
8709    fn crash_exit_report(at_ms: u64) -> ExitReport {
8710        ExitReport {
8711            kind: ExitKind::Crash,
8712            code: Some(1),
8713            signal: None,
8714            at_ms,
8715        }
8716    }
8717
8718    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
8719        ModuleSpec {
8720            module_id: module_id.to_string(),
8721            program: PathBuf::from("/unused").join(module_id),
8722            args: Vec::new(),
8723            env: Vec::new(),
8724            reserved: false,
8725            reserved_prefixes: Vec::new(),
8726            protocol: ModuleProtocol::Subc,
8727            overlap: Default::default(),
8728        }
8729    }
8730
8731    /// A real crash loop still stops. Three crashes with nothing aging out spend
8732    /// a budget of two and the third respawn is refused, and both surfaces an
8733    /// operator has -- the log line and the retained terminal record -- name the
8734    /// window rather than only the cap, because `max_restarts=2` alone is what
8735    /// this budget used to mean.
8736    #[tokio::test]
8737    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
8738        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
8739        let supervisor = Supervisor::new(
8740            Arc::new(Registry::default()),
8741            RestartPolicy::new(2, Duration::ZERO),
8742        );
8743        let runtime = supervisor.runtime_config();
8744        let spec = windowed_crash_spec("crash-loop-in-window");
8745        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8746
8747        for attempt in 1..=2 {
8748            assert!(
8749                matches!(
8750                    on_child_exit(
8751                        &spec,
8752                        runtime.restart_policy,
8753                        &supervisor.registry,
8754                        &snapshot,
8755                        &runtime.terminal_ring,
8756                        &runtime.spawn_events,
8757                        &runtime.child_roster,
8758                        crash_exit_report(attempt),
8759                    )
8760                    .await,
8761                    NextAction::Restart { schedule: _ }
8762                ),
8763                "crash {attempt} is inside the budget and must respawn"
8764            );
8765        }
8766
8767        assert!(matches!(
8768            on_child_exit(
8769                &spec,
8770                runtime.restart_policy,
8771                &supervisor.registry,
8772                &snapshot,
8773                &runtime.terminal_ring,
8774                &runtime.spawn_events,
8775                &runtime.child_roster,
8776                crash_exit_report(3),
8777            )
8778            .await,
8779            NextAction::Stop { .. }
8780        ));
8781
8782        {
8783            let state = lock_snapshot(&snapshot).unwrap();
8784            assert_eq!(state.state, ModuleState::Failed);
8785            assert_eq!(state.crash_restarts.len(), 2);
8786            assert_eq!(state.lifetime_restarts, 2);
8787        }
8788
8789        let history = runtime
8790            .terminal_ring
8791            .lock()
8792            .expect("terminal ring is not poisoned")
8793            .snapshot();
8794        let last = history
8795            .entries
8796            .last()
8797            .expect("the refused crash is retained");
8798        assert_eq!(last.disposition, TerminalDisposition::Failed);
8799        assert_eq!(
8800            last.disposition_detail.as_deref(),
8801            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
8802        );
8803
8804        let captured = crate::router::test_log::captured_logs(&logs);
8805        assert!(
8806            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
8807            "the stop must be logged with its window: {captured}"
8808        );
8809    }
8810
8811    /// The rate, stated as a test: three crashes where the first has aged past
8812    /// the window are two crashes as far as the budget is concerned, so the
8813    /// third respawn is allowed and the ring holds only the two recent ones.
8814    ///
8815    /// This is the case a lifetime counter got wrong -- and the case the daemon
8816    /// now hits routinely, since a module exits non-zero every time its
8817    /// connection to the daemon drops.
8818    #[tokio::test]
8819    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
8820        let supervisor = Supervisor::new(
8821            Arc::new(Registry::default()),
8822            RestartPolicy::new(2, Duration::ZERO),
8823        );
8824        let runtime = supervisor.runtime_config();
8825        let spec = windowed_crash_spec("crash-across-windows");
8826        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8827
8828        for attempt in 1..=2 {
8829            assert!(matches!(
8830                on_child_exit(
8831                    &spec,
8832                    runtime.restart_policy,
8833                    &supervisor.registry,
8834                    &snapshot,
8835                    &runtime.terminal_ring,
8836                    &runtime.spawn_events,
8837                    &runtime.child_roster,
8838                    crash_exit_report(attempt),
8839                )
8840                .await,
8841                NextAction::Restart { schedule: _ }
8842            ));
8843        }
8844
8845        // The oldest crash moves out of the window; nothing else about the
8846        // module changes.
8847        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
8848            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
8849        })
8850        .unwrap();
8851
8852        assert!(
8853            matches!(
8854                on_child_exit(
8855                    &spec,
8856                    runtime.restart_policy,
8857                    &supervisor.registry,
8858                    &snapshot,
8859                    &runtime.terminal_ring,
8860                    &runtime.spawn_events,
8861                    &runtime.child_roster,
8862                    crash_exit_report(3),
8863                )
8864                .await,
8865                NextAction::Restart { schedule: _ }
8866            ),
8867            "a crash older than the window must not hold a budget slot"
8868        );
8869
8870        let state = lock_snapshot(&snapshot).unwrap();
8871        assert_eq!(state.state, ModuleState::Restarting);
8872        assert_eq!(
8873            state.crash_restarts.len(),
8874            2,
8875            "the aged instant is dropped and the new one takes its place"
8876        );
8877        assert_eq!(
8878            state.lifetime_restarts, 3,
8879            "the ledger counts every restart, including the ones the window forgot"
8880        );
8881    }
8882
8883    /// An operator restart hands the budget back whole, and the ledger keeps
8884    /// counting. Those are different questions -- "how close is this module to
8885    /// being stopped" and "how many times has it been replaced" -- and the
8886    /// operator action answers only the first.
8887    #[tokio::test]
8888    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
8889        let supervisor = Supervisor::new(
8890            Arc::new(Registry::default()),
8891            RestartPolicy::new(2, Duration::ZERO),
8892        );
8893        let runtime = supervisor.runtime_config();
8894        let spec = windowed_crash_spec("operator-cleared-budget");
8895        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8896
8897        for attempt in 1..=2 {
8898            assert!(matches!(
8899                on_child_exit(
8900                    &spec,
8901                    runtime.restart_policy,
8902                    &supervisor.registry,
8903                    &snapshot,
8904                    &runtime.terminal_ring,
8905                    &runtime.spawn_events,
8906                    &runtime.child_roster,
8907                    crash_exit_report(attempt),
8908                )
8909                .await,
8910                NextAction::Restart { schedule: _ }
8911            ));
8912        }
8913
8914        reset_restart_count(&snapshot, &spec.module_id).unwrap();
8915        {
8916            let state = lock_snapshot(&snapshot).unwrap();
8917            assert!(
8918                state.crash_restarts.is_empty(),
8919                "an operator restart returns the full budget"
8920            );
8921            assert_eq!(
8922                state.lifetime_restarts, 2,
8923                "clearing the budget must not unmake the crashes"
8924            );
8925        }
8926
8927        assert!(
8928            matches!(
8929                on_child_exit(
8930                    &spec,
8931                    runtime.restart_policy,
8932                    &supervisor.registry,
8933                    &snapshot,
8934                    &runtime.terminal_ring,
8935                    &runtime.spawn_events,
8936                    &runtime.child_roster,
8937                    crash_exit_report(3),
8938                )
8939                .await,
8940                NextAction::Restart { schedule: _ }
8941            ),
8942            "the cleared budget must be spendable again"
8943        );
8944        let state = lock_snapshot(&snapshot).unwrap();
8945        assert_eq!(state.crash_restarts.len(), 1);
8946        assert_eq!(state.lifetime_restarts, 3);
8947    }
8948
8949    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8950    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
8951        let severed = ProcessIdentity {
8952            pid: 41,
8953            start_time: 101,
8954        };
8955        let successor = ProcessIdentity {
8956            pid: 41,
8957            start_time: 202,
8958        };
8959        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
8960        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
8961            state.pid = Some(successor.pid);
8962            state.process_start_time = Some(successor.start_time);
8963        })
8964        .unwrap();
8965        assert!(!module.record_deliberate_severance(severed).unwrap());
8966
8967        let exit_report = apply_deliberate_severance_marker(
8968            &module.inner.snapshot,
8969            Some(successor),
8970            ExitReport {
8971                kind: ExitKind::Crash,
8972                code: Some(1),
8973                signal: None,
8974                at_ms: 1,
8975            },
8976        );
8977
8978        assert_eq!(exit_report.kind, ExitKind::Crash);
8979    }
8980
8981    #[tokio::test]
8982    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
8983        let registry = Registry::default();
8984        let supervisor = Supervisor::new(
8985            Arc::new(Registry::default()),
8986            RestartPolicy::new(3, Duration::ZERO),
8987        );
8988        let runtime = supervisor.runtime_config();
8989        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8990        let spec = ModuleSpec {
8991            module_id: "drain-deliberate-severance".to_string(),
8992            program: fake_aft_stub_path(),
8993            args: Vec::new(),
8994            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8995            reserved: false,
8996            reserved_prefixes: Vec::new(),
8997            protocol: ModuleProtocol::Subc,
8998            overlap: Default::default(),
8999        };
9000        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9001        let process = ProcessIdentity {
9002            pid: 41,
9003            start_time: 101,
9004        };
9005        child.process_identity = Some(process);
9006        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
9007            state.pid = Some(process.pid);
9008            state.process_start_time = Some(process.start_time);
9009        })
9010        .unwrap();
9011        record_deliberate_severance(&snapshot, process).unwrap();
9012
9013        drain_child_to_state(
9014            &spec.module_id,
9015            spec.protocol,
9016            // The child exits on its own; no signal may change the exit this
9017            // test classifies.
9018            StopNotice::SentOverConnection,
9019            &registry,
9020            &snapshot,
9021            &runtime.terminal_ring,
9022            &runtime.spawn_events,
9023            child,
9024            Duration::from_secs(1),
9025            ModuleState::Stopped,
9026            Some(false),
9027        )
9028        .await
9029        .unwrap();
9030
9031        let state = lock_snapshot(&snapshot).unwrap();
9032        assert_eq!(
9033            state.last_exit.as_ref().map(|exit| exit.kind),
9034            Some(ExitKind::DeliberateSeverance)
9035        );
9036        assert_eq!(state.lifetime_restarts, 1);
9037        assert_eq!(state.crash_restarts.len(), 0);
9038        drop(state);
9039        let history = runtime.terminal_ring.lock().unwrap().snapshot();
9040        assert_eq!(
9041            history.entries[0].exit_kind,
9042            subc_control::TerminalExitKind::DeliberateSeverance
9043        );
9044    }
9045
9046    #[tokio::test]
9047    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
9048        let registry = Registry::default();
9049        let supervisor = Supervisor::new(
9050            Arc::new(Registry::default()),
9051            RestartPolicy::new(3, Duration::ZERO),
9052        );
9053        let runtime = supervisor.runtime_config();
9054        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9055        let spec = ModuleSpec {
9056            module_id: "ordinary-drain".to_string(),
9057            program: fake_aft_stub_path(),
9058            args: Vec::new(),
9059            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9060            reserved: false,
9061            reserved_prefixes: Vec::new(),
9062            protocol: ModuleProtocol::Subc,
9063            overlap: Default::default(),
9064        };
9065        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9066
9067        drain_child_to_state(
9068            &spec.module_id,
9069            spec.protocol,
9070            // The child exits on its own; no signal may change the exit this
9071            // test classifies.
9072            StopNotice::SentOverConnection,
9073            &registry,
9074            &snapshot,
9075            &runtime.terminal_ring,
9076            &runtime.spawn_events,
9077            child,
9078            Duration::from_secs(1),
9079            ModuleState::Stopped,
9080            Some(false),
9081        )
9082        .await
9083        .unwrap();
9084
9085        let state = lock_snapshot(&snapshot).unwrap();
9086        assert_eq!(
9087            state.last_exit.as_ref().map(|exit| exit.kind),
9088            Some(ExitKind::Crash)
9089        );
9090        assert_eq!(state.lifetime_restarts, 0);
9091        assert_eq!(state.crash_restarts.len(), 0);
9092    }
9093
9094    #[test]
9095    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
9096        // The server's generic fatal-routing branch only knows that the
9097        // connection failed; it does not know that the daemon deliberately
9098        // initiated a process-killing severance. Keep this seam explicit so a
9099        // future connection error path cannot silently reintroduce the stale
9100        // exemption that mislabels a later genuine crash.
9101        assert!(!include_str!("server.rs")
9102            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
9103    }
9104
9105    /// The `route.closed` `drained` value must be the quiescence wait's own
9106    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
9107    /// measurement at all and `false` is the one honest constant. This is the exact
9108    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
9109    /// on every return path, including the one that used to return early via `?`
9110    /// with `route.closing` already sent and no `route.closed` ever following.
9111    #[test]
9112    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
9113        assert!(drained_after_quiescence_wait(&Ok(true)));
9114        assert!(!drained_after_quiescence_wait(&Ok(false)));
9115        assert!(!drained_after_quiescence_wait(&Err(
9116            SuperviseError::StatePoisoned { module_id: None }
9117        )));
9118    }
9119
9120    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
9121    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
9122    /// already reaped out-of-band) still leaves a terminal record rather than none
9123    /// at all. Triggering the real `wait()` I/O error from an integration test would
9124    /// need a genuine already-reaped-child race, which is OS-specific and not
9125    /// something this suite attempts elsewhere; this test instead verifies the
9126    /// record produced for that arm end-to-end through the real `TerminalRing`, and
9127    /// the call site itself is verified by inspection to sit in that exact arm.
9128    #[test]
9129    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
9130        let ring = Arc::new(Mutex::new(TerminalRing::new(
9131            TerminalRingConfig::default(),
9132            0,
9133        )));
9134        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
9135
9136        let snapshot = ring.lock().unwrap().snapshot();
9137        assert_eq!(snapshot.entries.len(), 1);
9138        let entry = &snapshot.entries[0];
9139        assert_eq!(entry.exit_code, None);
9140        assert_eq!(entry.exit_signal, None);
9141        assert_eq!(entry.disposition, TerminalDisposition::Failed);
9142    }
9143
9144    #[test]
9145    fn wait_error_exit_path_preserves_spawn_event_density() {
9146        let feed = super::SpawnEventFeed::default();
9147        feed.configure_incarnation("wait-error-density".to_string());
9148        feed.emit_spawned("wait-error", 41, 1);
9149        let ring = Arc::new(Mutex::new(TerminalRing::new(
9150            TerminalRingConfig::default(),
9151            0,
9152        )));
9153
9154        record_wait_error_terminal("wait-error", &ring, &feed);
9155        feed.emit_spawned("after-wait-error", 42, 2);
9156
9157        let state = feed.0.lock().unwrap();
9158        let sequences = state
9159            .events
9160            .iter()
9161            .map(|event| event.cursor.seq)
9162            .collect::<Vec<_>>();
9163        assert_eq!(sequences, vec![1, 2, 3]);
9164        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
9165        assert_eq!(state.events[1].exit_code, None);
9166        assert_eq!(state.events[1].exit_signal, None);
9167    }
9168
9169    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
9170    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
9171    /// not a clean exit it never actually observed.
9172    #[test]
9173    fn wait_error_exit_report_is_classified_as_a_crash() {
9174        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
9175    }
9176}
9177
9178#[cfg(test)]
9179mod health_evidence_tests {
9180    use super::{HealthProbeError, HealthProbeEvidence};
9181    use std::collections::HashSet;
9182
9183    /// The evidential asymmetry, asserted rather than described.
9184    ///
9185    /// Exactly ONE observation is proof a module cannot serve, and the one that
9186    /// fires under CPU starvation is not it. Before the split, all fifteen
9187    /// construction sites collapsed into a single String, so a timeout carried the
9188    /// same weight as a dead lane -- which is how a healthy module was restarted
9189    /// three times in one day.
9190    #[test]
9191    fn only_a_dead_lane_is_proof_of_death() {
9192        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
9193        // Three non-proof classes, each for a different reason: silence is
9194        // consistent with health, a bad answer proves the module ALIVE, and a
9195        // daemon-side fault never reached the module at all.
9196        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
9197        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
9198        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
9199    }
9200
9201    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
9202    ///
9203    /// A shared label renders two different observations identically in the line an
9204    /// operator reads after an unexplained restart -- the exact confusion this
9205    /// change removes.
9206    #[test]
9207    fn every_evidence_class_has_a_distinct_label() {
9208        let labels = [
9209            HealthProbeError::lane_dead("").label(),
9210            HealthProbeError::no_answer("").label(),
9211            HealthProbeError::bad_answer("").label(),
9212            HealthProbeError::misconfigured("").label(),
9213        ];
9214        let unique: HashSet<_> = labels.iter().collect();
9215        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
9216    }
9217
9218    /// The class is additional information, not a replacement.
9219    ///
9220    /// An operator needs both "this was silence" and the specific text saying how
9221    /// long we waited; a classification that swallowed the message would trade one
9222    /// missing distinction for another.
9223    #[test]
9224    fn classification_preserves_the_original_message() {
9225        let err = HealthProbeError::no_answer("module did not answer within 5s");
9226        assert_eq!(err.to_string(), "module did not answer within 5s");
9227        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9228    }
9229}
9230
9231#[cfg(test)]
9232mod health_tombstone_tests {
9233    use std::{path::PathBuf, sync::Arc, time::Duration};
9234
9235    use subc_protocol::{
9236        manifest::Concurrency,
9237        session::{HealthStatus, ModuleControlResponse},
9238    };
9239    use tokio::sync::mpsc;
9240
9241    use super::{
9242        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
9243        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
9244    };
9245    use crate::{
9246        control::ControlHandler,
9247        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
9248        registry::{ConnectionId, Registry},
9249        router::FrameSink,
9250    };
9251
9252    struct ProbeHarness {
9253        spec: ModuleSpec,
9254        runtime: SupervisorRuntimeConfig,
9255        forwarding: Arc<ForwardingTable>,
9256        module_connection: ConnectionId,
9257        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
9258        handler: ControlHandler,
9259        module: super::SupervisedModule,
9260    }
9261
9262    fn probe_harness() -> ProbeHarness {
9263        let registry = Arc::new(Registry::default());
9264        let forwarding = Arc::new(ForwardingTable::default());
9265        let supervisor_handle = super::SupervisorHandle::new();
9266        let health = HealthConfig {
9267            cadence: Duration::from_secs(30),
9268            deadline: Duration::from_secs(5),
9269            failure_threshold: 3,
9270            on_degraded: HealthAction::Report,
9271            on_failing: HealthAction::Report,
9272            critical: false,
9273        };
9274        let supervisor = Supervisor::new(Arc::clone(&registry), RestartPolicy::default())
9275            .with_forwarding(Arc::clone(&forwarding))
9276            .with_handle(supervisor_handle.clone())
9277            .with_health_config(health);
9278        let spec = ModuleSpec {
9279            module_id: "late-health-module".to_string(),
9280            program: PathBuf::from("disabled-module"),
9281            args: Vec::new(),
9282            env: Vec::new(),
9283            reserved: false,
9284            reserved_prefixes: Vec::new(),
9285            protocol: ModuleProtocol::Subc,
9286            overlap: Default::default(),
9287        };
9288        let module = supervisor
9289            .supervise_configured(spec.clone(), false)
9290            .unwrap();
9291        let runtime = supervisor.runtime_config();
9292        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
9293            .with_supervisor(supervisor_handle);
9294        let module_connection = ConnectionId::new(700);
9295        let (module_tx, module_rx) = mpsc::channel(8);
9296        forwarding
9297            .register_module_connection(
9298                module_connection,
9299                spec.module_id.clone(),
9300                subc_protocol::PROTOCOL_VERSION,
9301                Concurrency::ModuleManaged,
9302                FrameSink::new(module_tx),
9303            )
9304            .unwrap();
9305
9306        ProbeHarness {
9307            spec,
9308            runtime,
9309            forwarding,
9310            module_connection,
9311            module_rx,
9312            handler,
9313            module,
9314        }
9315    }
9316
9317    async fn finish_after(
9318        harness: &mut ProbeHarness,
9319        stall: Duration,
9320    ) -> ModuleControlRpcCompletion {
9321        assert!(stall > harness.runtime.health.deadline);
9322        let deadline = harness.runtime.health.deadline;
9323        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9324        let answer = async {
9325            let frame = harness.module_rx.recv().await.expect("health.check frame");
9326            tokio::time::advance(deadline).await;
9327            tokio::task::yield_now().await;
9328            tokio::time::advance(stall - deadline).await;
9329            harness
9330                .forwarding
9331                .complete_module_control_rpc(
9332                    harness.module_connection,
9333                    frame.header.corr,
9334                    Some("health.check"),
9335                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
9336                        status: HealthStatus::Ok,
9337                        detail: None,
9338                        metrics: None,
9339                    }),
9340                )
9341                .unwrap()
9342        };
9343        let (probe_result, completion) = tokio::join!(probe, answer);
9344        let err = probe_result.expect_err("probe must miss its deadline");
9345        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9346        completion
9347    }
9348
9349    async fn time_out_without_answer(harness: &mut ProbeHarness) {
9350        let deadline = harness.runtime.health.deadline;
9351        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9352        let exhaust_deadline = async {
9353            let _frame = harness.module_rx.recv().await.expect("health.check frame");
9354            tokio::time::advance(deadline).await;
9355            tokio::task::yield_now().await;
9356        };
9357        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
9358        let err = probe_result.expect_err("probe must miss its deadline");
9359        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9360    }
9361
9362    #[tokio::test(start_paused = true)]
9363    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
9364        let mut harness = probe_harness();
9365
9366        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
9367        let first_latency = match &first {
9368            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9369            other => panic!("late answer was not retained: {other:?}"),
9370        };
9371        assert!(harness.handler.observe_module_control_completion(first));
9372
9373        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
9374        let second_latency = match &second {
9375            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9376            other => panic!("late answer was not retained: {other:?}"),
9377        };
9378        assert!(harness.handler.observe_module_control_completion(second));
9379
9380        assert_eq!(first_latency, Duration::from_secs(8));
9381        assert_eq!(
9382            second_latency - first_latency,
9383            Duration::from_secs(3),
9384            "latency must grow linearly with the additional stall"
9385        );
9386        let health = harness.module.status().unwrap().health;
9387        assert_eq!(health.late_answer_count, 2);
9388        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
9389    }
9390
9391    /// A module that answers every probe late must never march to the kill
9392    /// threshold: the late answer proves it is alive, so it must clear the miss
9393    /// streak the timeout recorded. Without the reset, a CPU-starved module
9394    /// that serves every probe seconds past the deadline accumulates
9395    /// `consecutive_failures` to the threshold and is killed — the exact
9396    /// sequence from the 2026-08-14 aft disable, where the daemon logged
9397    /// "proves the module is alive" five times while counting five misses.
9398    #[tokio::test(start_paused = true)]
9399    async fn late_answer_clears_the_consecutive_failure_streak() {
9400        let mut harness = probe_harness();
9401
9402        // Timeout recorded first: the probe path saw no answer in time.
9403        time_out_without_answer(&mut harness).await;
9404        harness
9405            .module
9406            .record_health_probe_failure_for_test("[no-answer] test miss")
9407            .unwrap();
9408        assert_eq!(
9409            harness.module.status().unwrap().health.consecutive_failures,
9410            1,
9411            "precondition: the miss must be on the streak before the late answer"
9412        );
9413
9414        // The stalled reply then lands: proof of life.
9415        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
9416        assert!(matches!(
9417            late,
9418            ModuleControlRpcCompletion::LateHealthAnswer { .. }
9419        ));
9420        assert!(harness.handler.observe_module_control_completion(late));
9421
9422        let health = harness.module.status().unwrap().health;
9423        assert_eq!(
9424            health.consecutive_failures, 0,
9425            "a late answer is an answer: the streak must reset"
9426        );
9427        assert_eq!(health.late_answer_count, 1);
9428    }
9429
9430    #[tokio::test(start_paused = true)]
9431    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
9432        let mut harness = probe_harness();
9433
9434        for _ in 0..20 {
9435            time_out_without_answer(&mut harness).await;
9436            assert_eq!(
9437                harness.forwarding.health_probe_tombstone_count().unwrap(),
9438                1
9439            );
9440        }
9441    }
9442}
9443
9444#[cfg(test)]
9445mod child_env_tests {
9446    use super::{
9447        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
9448        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
9449        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
9450    };
9451    use std::{ffi::OsStr, path::PathBuf};
9452    use tokio::process::Command;
9453
9454    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
9455        ModuleSpec {
9456            module_id: "env-plan".to_string(),
9457            program: PathBuf::from("/nonexistent"),
9458            args: Vec::new(),
9459            env,
9460            reserved: false,
9461            reserved_prefixes: Vec::new(),
9462            protocol: ModuleProtocol::Subc,
9463            overlap: Default::default(),
9464        }
9465    }
9466
9467    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
9468    /// one still gets its own.
9469    ///
9470    /// This is the narrow goal `env_clear()` was reached for, and the reason the
9471    /// fix is `env_remove` rather than deleting the line: an operator's ambient
9472    /// filter silently becoming an unconfigured module's log level is a real
9473    /// defect, just a much smaller one than clearing the environment.
9474    ///
9475    /// Asserted on the command plan rather than a spawned child because proving
9476    /// the ABSENCE of an inherited variable needs the parent's environment
9477    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
9478    /// removal as `(key, None)`, which is exactly the distinction wanted: not
9479    /// "absent because nobody set it" but "explicitly unset for the child".
9480    #[test]
9481    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
9482        let mut command = Command::new("/nonexistent");
9483        apply_child_env(&mut command, &spec(Vec::new()));
9484        let removed = command
9485            .as_std()
9486            .get_envs()
9487            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
9488        assert!(
9489            removed,
9490            "ambient CK_LOG must be explicitly removed for an unconfigured module"
9491        );
9492
9493        let mut configured = Command::new("/nonexistent");
9494        apply_child_env(
9495            &mut configured,
9496            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
9497        );
9498        let effective = configured
9499            .as_std()
9500            .get_envs()
9501            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
9502            .last()
9503            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9504        assert_eq!(
9505            effective,
9506            Some(Some("debug".to_string())),
9507            "a module's configured CK_LOG must survive the ambient removal"
9508        );
9509    }
9510
9511    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
9512    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
9513    /// the same reason as the CK_LOG test above.
9514    ///
9515    /// The argument is the load-bearing half: a stock binary exits on an
9516    /// unknown flag before it listens, so with `--subc` appended the mode
9517    /// could not supervise the one process it exists for. Found by the first
9518    /// conformance run (nats-server: `flag provided but not defined: -subc`).
9519    #[test]
9520    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
9521        let connection_file = std::path::Path::new("/run/subc-connection.json");
9522        let handle = SupervisorHandle::new();
9523
9524        let mut none_spec = spec(Vec::new());
9525        none_spec.protocol = ModuleProtocol::None;
9526        let mut none = Command::new("/nonexistent");
9527        let none_handoff =
9528            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
9529                .expect("protocol-none spawn args apply");
9530        assert!(
9531            none_handoff.is_none(),
9532            "protocol:none spawn must not receive a nonce descriptor"
9533        );
9534        assert!(
9535            !none.as_std().get_envs().any(|(key, value)| key
9536                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
9537                && value.is_some()),
9538            "protocol:none spawn must not name a nonce descriptor"
9539        );
9540        let none_args: Vec<String> = none
9541            .as_std()
9542            .get_args()
9543            .map(|a| a.to_string_lossy().into_owned())
9544            .collect();
9545        assert!(
9546            !none_args.iter().any(|a| a == SUBC_ARG),
9547            "protocol:none argv must not carry --subc; got {none_args:?}"
9548        );
9549        let none_has_nonce = none
9550            .as_std()
9551            .get_envs()
9552            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
9553        assert!(
9554            !none_has_nonce,
9555            "protocol:none spawn must not receive a launch nonce"
9556        );
9557        let none_has_module_id = none
9558            .as_std()
9559            .get_envs()
9560            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
9561        assert!(
9562            none_has_module_id,
9563            "SUBC_MODULE_ID is inert and stays on every path"
9564        );
9565        assert!(
9566            handle.spawn_nonce(&none_spec.module_id).is_none(),
9567            "no nonce record for a process that will never present one"
9568        );
9569
9570        // Control: the subc-wire path is unchanged by the branch above.
9571        let wire_spec = spec(Vec::new());
9572        let mut wire = Command::new("/nonexistent");
9573        let wire_handoff =
9574            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
9575                .expect("subc-wire spawn args apply");
9576        let wire_fd_env = wire
9577            .as_std()
9578            .get_envs()
9579            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
9580            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9581        #[cfg(unix)]
9582        assert_eq!(
9583            wire_fd_env,
9584            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
9585            "a subc-wire spawn names the pipe it will receive at descriptor 3"
9586        );
9587        #[cfg(not(unix))]
9588        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
9589        let wire_args: Vec<String> = wire
9590            .as_std()
9591            .get_args()
9592            .map(|a| a.to_string_lossy().into_owned())
9593            .collect();
9594        assert_eq!(
9595            wire_args,
9596            vec![
9597                SUBC_ARG.to_string(),
9598                connection_file.to_string_lossy().into_owned()
9599            ],
9600            "a subc-wire spawn still carries --subc <path>"
9601        );
9602        assert_eq!(
9603            wire.as_std()
9604                .get_envs()
9605                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
9606            !cfg!(unix),
9607            "only Windows supplies the environment nonce"
9608        );
9609        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
9610    }
9611
9612    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
9613    /// spec tries to set it; only a swap candidate carries it.
9614    ///
9615    /// "Set it only on candidates" is not enough, because spawn applies the
9616    /// spec's env verbatim and the daemon's own environment is inherited: either
9617    /// could hand a plain restart the swap role, and a module reading it would
9618    /// warm on its long swap budget while callers wait. Asserted as an explicit
9619    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
9620    /// test above gives.
9621    #[test]
9622    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
9623        let role = |command: &Command| {
9624            command
9625                .as_std()
9626                .get_envs()
9627                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
9628                .last()
9629                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
9630        };
9631        let forged = spec(vec![(
9632            SUBC_SPAWN_ROLE_ENV.to_string(),
9633            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
9634        )]);
9635
9636        let mut plain = Command::new("/nonexistent");
9637        apply_child_env(&mut plain, &forged);
9638        apply_spawn_role(&mut plain, SpawnRole::Plain);
9639        assert_eq!(
9640            role(&plain),
9641            Some(None),
9642            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
9643        );
9644
9645        let mut candidate = Command::new("/nonexistent");
9646        apply_child_env(&mut candidate, &spec(Vec::new()));
9647        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
9648        assert_eq!(
9649            role(&candidate),
9650            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
9651        );
9652    }
9653
9654    /// Daemon-private capture retention keys never reach the child.
9655    ///
9656    /// cortexkit-log exposes retention as a Rust struct with no environment
9657    /// names, so these entries are supervisor metadata. Passing them through
9658    /// would invent a public child-process contract by accident.
9659    #[test]
9660    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
9661        let mut command = Command::new("/nonexistent");
9662        apply_child_env(
9663            &mut command,
9664            &spec(vec![
9665                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
9666                ("KEPT".to_string(), "yes".to_string()),
9667            ]),
9668        );
9669        let keys: Vec<String> = command
9670            .as_std()
9671            .get_envs()
9672            .filter(|(_, value)| value.is_some())
9673            .map(|(key, _)| key.to_string_lossy().into_owned())
9674            .collect();
9675        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
9676        assert!(
9677            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
9678            "daemon-private capture key leaked to the child: {keys:?}"
9679        );
9680    }
9681}
9682
9683#[cfg(test)]
9684mod jitter_tests {
9685    use super::jittered_health_delay;
9686    use std::{collections::HashSet, time::Duration};
9687
9688    /// Module ids drawn from a real fleet, so the dispersal claim is about names
9689    /// that actually occur rather than invented ones.
9690    ///
9691    /// This is a SAMPLE, not a registry: the property under test is that distinct
9692    /// ids disperse, which holds for any set of distinct strings. Several entries
9693    /// are already historical (modules get renamed), and that costs nothing here --
9694    /// but it means a reader must not mistake this for the live module set, and a
9695    /// rename sweep will match it without there being anything to change.
9696    const FLEET: [&str; 14] = [
9697        "aft",
9698        "alfonso-core",
9699        "magic-context",
9700        "broca",
9701        "thalamus",
9702        "quota",
9703        "engram",
9704        "plexus",
9705        "cerebellum",
9706        "astrocyte",
9707        "synapse",
9708        "subc-mcp",
9709        "cortexkit-credentials",
9710        "subc-federation",
9711    ];
9712
9713    /// Probes must not converge after a fleet-wide restart.
9714    ///
9715    /// This is the property the jitter exists for: every module reconnects at
9716    /// once, and without dispersal all fourteen would then probe on the same
9717    /// tick forever. Nothing failed visibly when this went untested -- a
9718    /// convergent fleet still probes correctly, just in a burst, so the symptom
9719    /// is a periodic load spike that looks like whatever else is running.
9720    #[test]
9721    fn probe_delays_disperse_across_the_fleet() {
9722        let cadence = Duration::from_secs(30);
9723        let delays: HashSet<Duration> = FLEET
9724            .iter()
9725            .map(|id| jittered_health_delay(id, 0, cadence))
9726            .collect();
9727        assert_eq!(
9728            delays.len(),
9729            FLEET.len(),
9730            "every supervised module must land on its own probe offset"
9731        );
9732    }
9733
9734    /// The offset may only ever DELAY a probe, never bring it forward.
9735    ///
9736    /// A delay below the cadence would probe a module more often than
9737    /// configured, which is the opposite of what an operator asked for and
9738    /// would tighten the failure budget without anyone changing it.
9739    #[test]
9740    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
9741        let cadence = Duration::from_secs(30);
9742        let span = cadence / 10;
9743        for id in FLEET {
9744            for probe_index in 0..8 {
9745                let delay = jittered_health_delay(id, probe_index, cadence);
9746                assert!(
9747                    delay >= cadence,
9748                    "{id}#{probe_index}: jitter must not shorten the cadence"
9749                );
9750                assert!(
9751                    delay < cadence + span,
9752                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
9753                );
9754            }
9755        }
9756    }
9757
9758    /// A module keeps its offset across daemon restarts.
9759    ///
9760    /// The delay is derived rather than randomised precisely so a restart does
9761    /// not re-roll every module into a fresh chance of collision. A random
9762    /// source would satisfy the dispersal test above and quietly lose this.
9763    #[test]
9764    fn a_module_offset_is_stable_across_restarts() {
9765        let cadence = Duration::from_secs(30);
9766        for id in FLEET {
9767            assert_eq!(
9768                jittered_health_delay(id, 0, cadence),
9769                jittered_health_delay(id, 0, cadence),
9770                "{id}: the same module and probe index must produce the same offset"
9771            );
9772        }
9773    }
9774
9775    /// A zero cadence disables probing rather than producing a busy loop.
9776    #[test]
9777    fn zero_cadence_yields_zero_delay() {
9778        assert_eq!(
9779            jittered_health_delay("aft", 0, Duration::ZERO),
9780            Duration::ZERO
9781        );
9782    }
9783}
9784
9785#[cfg(all(test, target_os = "linux"))]
9786mod cgroup_placement_tests {
9787    use super::{
9788        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
9789        SupervisedChild,
9790    };
9791    use crate::stderr_tail::{StderrRing, StderrTailConfig};
9792    use std::{
9793        fs, io,
9794        path::{Path, PathBuf},
9795        sync::{Arc, Mutex},
9796    };
9797    use subc_test_support::TestTempDir;
9798    use tokio::process::Command;
9799
9800    #[test]
9801    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
9802        let path = Path::new("/definitely-missing-subc-cgroup");
9803        let mut command = Command::new("true");
9804        let error = apply_cgroup_placement(
9805            &mut command,
9806            &ModuleSpec {
9807                module_id: "broken-cgroup".to_string(),
9808                program: PathBuf::from("true"),
9809                args: Vec::new(),
9810                env: Vec::new(),
9811                reserved: false,
9812                reserved_prefixes: Vec::new(),
9813                protocol: ModuleProtocol::Subc,
9814                overlap: Default::default(),
9815            },
9816            path,
9817        )
9818        .expect_err("a parent cgroup open failure must reject the supervised spawn");
9819        let reason = error.to_string();
9820
9821        assert!(
9822            matches!(error, SuperviseError::Cgroup { .. }),
9823            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
9824        );
9825        assert!(
9826            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
9827            "parent cgroup open failure must name cgroup.procs: {reason}"
9828        );
9829    }
9830
9831    #[tokio::test]
9832    async fn reaping_a_child_removes_its_empty_module_cgroup() {
9833        let root = TestTempDir::new("supervisor-reap-cgroup");
9834        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9835        let placement = subc_cgroup::prepare_at(&root)
9836            .expect("prepare scratch cgroup root")
9837            .expect("scratch root has a cgroup.procs marker");
9838        let module_id = "reaped-module";
9839        let module = placement
9840            .module_path(module_id)
9841            .expect("create scratch module cgroup");
9842        let child = Command::new("true")
9843            .spawn()
9844            .expect("spawn short-lived child");
9845        let pid = child.id().expect("spawned child has pid");
9846        let mut child = SupervisedChild {
9847            child,
9848            module_id: module_id.to_string(),
9849            cgroup_placement: Some(placement),
9850            stdout_pump: None,
9851            stderr_pump: None,
9852            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
9853            spawned_at_ms: 0,
9854            spawned_from: PathBuf::from("true"),
9855            spawned_file_identity: None,
9856            process_start_time: None,
9857            process_identity: None,
9858            pid,
9859            roster_guard: None,
9860        };
9861
9862        child.wait().await.expect("reap short-lived child");
9863
9864        assert!(
9865            !module.exists(),
9866            "reaping the supervised child must remove its empty cgroup"
9867        );
9868    }
9869
9870    #[test]
9871    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
9872        let root = TestTempDir::new("supervisor-non-empty-cgroup");
9873        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9874        let placement = subc_cgroup::prepare_at(&root)
9875            .expect("prepare scratch cgroup root")
9876            .expect("scratch root has a cgroup.procs marker");
9877        let module = placement
9878            .module_path("surviving-module")
9879            .expect("create scratch module cgroup");
9880        fs::write(module.join("surviving-process"), b"still present")
9881            .expect("make scratch cgroup non-empty");
9882        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
9883
9884        remove_module_cgroup(&placement, "surviving-module");
9885
9886        let logs = crate::router::test_log::captured_logs(&logs);
9887        assert!(
9888            module.exists(),
9889            "failed removal must leave the cgroup intact"
9890        );
9891        assert!(
9892            logs.contains("could not remove module cgroup after process exit; continuing teardown")
9893                && logs.contains("surviving-module"),
9894            "best-effort removal must report the failure without returning it: {logs}"
9895        );
9896    }
9897
9898    #[test]
9899    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
9900        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
9901        let reason = SuperviseError::Spawn {
9902            program: PathBuf::from("/bin/true"),
9903            source: io::Error::from_raw_os_error(13),
9904            cgroup_path: Some(cgroup_path.clone()),
9905        }
9906        .to_string();
9907
9908        assert!(
9909            reason.contains(&cgroup_path.display().to_string()),
9910            "a pre_exec spawn failure must name the cgroup path: {reason}"
9911        );
9912    }
9913}
9914
9915#[cfg(test)]
9916mod spawn_subscriber_lag_tests {
9917    use super::*;
9918
9919    /// A subscriber whose connection stops draining is dropped once its frame
9920    /// channel fills. The client must learn that from a terminal Error frame
9921    /// after the frames already queued for it, not from a stream that simply
9922    /// goes quiet.
9923    #[tokio::test]
9924    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
9925        let feed = SpawnEventFeed::default();
9926        feed.configure_incarnation("lag-incarnation".to_string());
9927        // A one-slot connection queue that nobody reads until the emits are
9928        // done: the forwarder parks on it and the subscriber channel fills.
9929        let (tx, mut rx) = mpsc::channel(1);
9930        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
9931            .expect("subscribe");
9932        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
9933        for index in 0..emitted {
9934            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
9935            // Let the forwarder take what it can so the fill point is the
9936            // subscriber channel, not a scheduling accident.
9937            tokio::task::yield_now().await;
9938        }
9939        assert_eq!(
9940            feed.subscriber_count(),
9941            0,
9942            "the lagged subscriber must be removed"
9943        );
9944
9945        let mut data = Vec::new();
9946        let mut last = None;
9947        loop {
9948            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
9949                .await
9950                .expect("the forwarder must finish once the subscriber is dropped");
9951            let Some(outbound) = next else { break };
9952            let frame = outbound.frame;
9953            if frame.header.ty == FrameType::StreamData {
9954                assert!(last.is_none(), "no data may follow the terminal frame");
9955                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
9956                data.push(event.cursor.seq);
9957            } else {
9958                assert!(last.is_none(), "exactly one terminal frame");
9959                last = Some(frame);
9960            }
9961        }
9962        assert!(!data.is_empty(), "queued frames drain before the terminal");
9963        for pair in data.windows(2) {
9964            assert_eq!(
9965                pair[1],
9966                pair[0] + 1,
9967                "queued frames arrive dense and in order"
9968            );
9969        }
9970        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
9971        assert_eq!(terminal.header.ty, FrameType::Error);
9972        assert_eq!(terminal.header.corr, 7);
9973        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
9974        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
9975        let detail = body.detail.expect("lagged error carries detail");
9976        assert_eq!(
9977            detail["first_undelivered_cursor"]["seq"],
9978            data.last().unwrap() + 1,
9979            "the named cursor is the first event the subscriber did not receive"
9980        );
9981        assert_eq!(
9982            detail["first_undelivered_cursor"]["daemon_incarnation"],
9983            "lag-incarnation"
9984        );
9985    }
9986}
9987
9988#[cfg(test)]
9989mod terminal_history_read_concurrency_tests {
9990    use super::*;
9991    use crate::terminal_journal::read_pause;
9992    use std::sync::mpsc as std_mpsc;
9993    use subc_test_support::TestTempDir;
9994
9995    fn journaled_ring(
9996        journal: &Arc<crate::terminal_journal::TerminalJournal>,
9997    ) -> Arc<Mutex<TerminalRing>> {
9998        Arc::new(Mutex::new(
9999            TerminalRing::new(TerminalRingConfig::default(), 1)
10000                .with_journal(Some(Arc::clone(journal))),
10001        ))
10002    }
10003
10004    fn crash(at_ms: u64) -> ExitReport {
10005        ExitReport {
10006            kind: ExitKind::Crash,
10007            code: Some(1),
10008            signal: None,
10009            at_ms,
10010        }
10011    }
10012
10013    /// Record an exit on another thread and report whether it finished within
10014    /// `bound`. The recorder thread is left running if it did not.
10015    fn record_within(
10016        module_id: &'static str,
10017        ring: &Arc<Mutex<TerminalRing>>,
10018        at_ms: u64,
10019        bound: Duration,
10020    ) -> bool {
10021        let ring = Arc::clone(ring);
10022        let (done, done_rx) = std_mpsc::channel();
10023        std::thread::spawn(move || {
10024            record_terminal(
10025                module_id,
10026                &ring,
10027                &SpawnEventFeed::default(),
10028                &crash(at_ms),
10029                TerminalDisposition::Restarting,
10030            );
10031            let _ = done.send(());
10032        });
10033        done_rx.recv_timeout(bound).is_ok()
10034    }
10035
10036    /// A history read in progress must not hold the journal writer (which every
10037    /// module's exit recording needs) or the module's own ring. Exits recorded
10038    /// while the read is paused complete promptly; the paused read answers as of
10039    /// the moment it started, and the next read has each exit exactly once.
10040    #[test]
10041    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
10042        let dir = TestTempDir::new("terminal-history-concurrent-read");
10043        let path = dir.join("terminals.jsonl");
10044        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
10045            path.clone(),
10046            "daemon".into(),
10047        ));
10048        let reader_ring = journaled_ring(&journal);
10049        let other_ring = journaled_ring(&journal);
10050        assert!(record_within(
10051            "reader-module",
10052            &reader_ring,
10053            10,
10054            Duration::from_secs(5)
10055        ));
10056
10057        let (started, release) = read_pause::install(&path);
10058        let reading = {
10059            let ring = Arc::clone(&reader_ring);
10060            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
10061        };
10062        started
10063            .recv_timeout(Duration::from_secs(5))
10064            .expect("the history read reached its pause");
10065
10066        let bound = Duration::from_secs(1);
10067        assert!(
10068            record_within("other-module", &other_ring, 20, bound),
10069            "another module's exit waited on a history read (journal writer held)"
10070        );
10071        assert!(
10072            record_within("reader-module", &reader_ring, 30, bound),
10073            "the read module's own exit waited on its history read (ring held)"
10074        );
10075
10076        drop(release);
10077        let paused = reading.join().unwrap();
10078        assert_eq!(
10079            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10080            vec![10],
10081            "an exit recorded after the read began lands in neither half of it"
10082        );
10083        assert_eq!(paused.journal_skipped_lines, 0);
10084        assert_eq!(paused.journal_read_errors, 0);
10085
10086        let after = durable_terminal_history_of(&reader_ring, "reader-module");
10087        assert_eq!(
10088            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10089            vec![10, 30],
10090            "the next read merges ring and journal with no duplicate"
10091        );
10092        assert_eq!(after.journal_skipped_lines, 0);
10093    }
10094}
10095
10096/// What a restart does with the exited process's stderr reader. These drive
10097/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
10098/// holds, so a reader that has not been scheduled by the bound is a controlled
10099/// input rather than something only a loaded machine produces.
10100#[cfg(test)]
10101mod stderr_settle_tests {
10102    use std::{
10103        future::Future,
10104        io,
10105        pin::Pin,
10106        sync::{Arc, Mutex},
10107        task::{Context, Poll},
10108        time::Duration,
10109    };
10110
10111    use tokio::{
10112        io::{AsyncRead, ReadBuf},
10113        sync::oneshot,
10114        time::Instant,
10115    };
10116
10117    use super::{settle_stderr_pump, StderrPump};
10118    use crate::stderr_tail::{
10119        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
10120    };
10121
10122    const BOUND: Duration = Duration::from_millis(250);
10123
10124    /// Yields `before`, then stays pending until the gate is released, then
10125    /// yields `after` and reaches EOF. The bytes after the gate were written
10126    /// by a process that has already exited; only the reader is behind.
10127    struct HeldReader {
10128        before: Option<Vec<u8>>,
10129        gate: Option<oneshot::Receiver<()>>,
10130        after: io::Cursor<Vec<u8>>,
10131    }
10132
10133    impl AsyncRead for HeldReader {
10134        fn poll_read(
10135            mut self: Pin<&mut Self>,
10136            cx: &mut Context<'_>,
10137            buf: &mut ReadBuf<'_>,
10138        ) -> Poll<io::Result<()>> {
10139            if let Some(bytes) = self.before.take() {
10140                buf.put_slice(&bytes);
10141                return Poll::Ready(Ok(()));
10142            }
10143            if let Some(gate) = self.gate.as_mut() {
10144                match Pin::new(gate).poll(cx) {
10145                    Poll::Pending => return Poll::Pending,
10146                    Poll::Ready(_) => self.gate = None,
10147                }
10148            }
10149            Pin::new(&mut self.after).poll_read(cx, buf)
10150        }
10151    }
10152
10153    struct DiscardSink;
10154
10155    impl OutputSink for DiscardSink {
10156        fn write_line(&mut self, _line: &[u8]) {}
10157    }
10158
10159    fn line(text: &str) -> TailEntry {
10160        TailEntry::Line {
10161            text: text.to_string(),
10162            truncated: false,
10163            at_ms: None,
10164        }
10165    }
10166
10167    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
10168        ring.lock().unwrap()
10169    }
10170
10171    /// Start a reader for a new process generation that delivers `before`
10172    /// immediately and `after` only once the returned sender fires (or is
10173    /// dropped).
10174    fn held_pump(
10175        ring: &Arc<Mutex<StderrRing>>,
10176        before: &str,
10177        after: &str,
10178    ) -> (StderrPump, oneshot::Sender<()>) {
10179        let generation = lock(ring).begin_process();
10180        let (release, gate) = oneshot::channel();
10181        let reader = HeldReader {
10182            before: Some(before.as_bytes().to_vec()),
10183            gate: Some(gate),
10184            after: io::Cursor::new(after.as_bytes().to_vec()),
10185        };
10186        let task = tokio::spawn(pump_stderr_to(
10187            reader,
10188            Arc::clone(ring),
10189            generation,
10190            DiscardSink,
10191        ));
10192        (StderrPump { task, generation }, release)
10193    }
10194
10195    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
10196        for _ in 0..1000 {
10197            if done(&lock(ring)) {
10198                return;
10199            }
10200            tokio::time::sleep(Duration::from_millis(1)).await;
10201        }
10202        panic!(
10203            "ring never reached the expected state: {:?}",
10204            lock(ring).snapshot(None, None)
10205        );
10206    }
10207
10208    #[tokio::test(start_paused = true)]
10209    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
10210        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10211        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
10212
10213        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
10214        let before_release = lock(&ring).snapshot(None, None);
10215        assert!(
10216            matches!(before_release.capture, CaptureState::Incomplete { .. }),
10217            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
10218        );
10219
10220        // The restart: the next process starts and writes before the old
10221        // reader catches up.
10222        let next = lock(&ring).begin_process();
10223        lock(&ring).push_line_from(next, "next process booting");
10224        release.send(()).unwrap();
10225        wait_until(&ring, |ring| {
10226            ring.snapshot(None, None).capture == CaptureState::Captured
10227        })
10228        .await;
10229
10230        assert_eq!(
10231            untimed(lock(&ring).snapshot(None, None).entries),
10232            vec![
10233                line("booting"),
10234                line("config error: missing storage"),
10235                TailEntry::ProcessStart,
10236                line("next process booting"),
10237            ],
10238            "the crash's last line must survive a slow reader and stay in the crashed process's section"
10239        );
10240    }
10241
10242    #[tokio::test(start_paused = true)]
10243    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
10244    ) {
10245        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10246        // `_held` is never fired: a descendant keeps the pipe open for the
10247        // whole test.
10248        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
10249
10250        let started = Instant::now();
10251        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
10252        assert_eq!(
10253            started.elapsed(),
10254            BOUND,
10255            "the restart must wait exactly the bound for a pipe that stays open, no longer"
10256        );
10257
10258        let next = lock(&ring).begin_process();
10259        lock(&ring).push_line_from(next, "next process booting");
10260        tokio::time::sleep(Duration::from_secs(60)).await;
10261
10262        let snapshot = lock(&ring).snapshot(None, None);
10263        match &snapshot.capture {
10264            CaptureState::Incomplete { reason } => assert!(
10265                reason.contains("had not reached EOF") && reason.contains("250ms"),
10266                "the reason must say what is missing and after how long: {reason}"
10267            ),
10268            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
10269        }
10270        assert_eq!(
10271            untimed(snapshot.entries),
10272            vec![
10273                line("parent exiting"),
10274                TailEntry::ProcessStart,
10275                line("next process booting"),
10276            ]
10277        );
10278    }
10279
10280    #[tokio::test(start_paused = true)]
10281    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
10282        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10283        let (pump, release) = held_pump(&ring, "one\n", "two\n");
10284        release.send(()).unwrap();
10285
10286        settle_stderr_pump("clean", &ring, pump, BOUND).await;
10287
10288        let snapshot = lock(&ring).snapshot(None, None);
10289        assert_eq!(snapshot.capture, CaptureState::Captured);
10290        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
10291    }
10292}
10293
10294/// Containment of a module's process tree (issue #109).
10295///
10296/// The behaviour these defend against is a module helper surviving its module:
10297/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
10298/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
10299/// compounds it.
10300///
10301/// They run against the SUPERVISOR rather than the job-object crate because the
10302/// claim is about teardown: a crate-level test proves a job can reap a tree, not
10303/// that the daemon's drain path reaches it.
10304///
10305/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
10306/// lane there is a separate containment path with its own tests.
10307#[cfg(all(test, windows))]
10308mod job_containment_tests {
10309    use super::*;
10310    use std::{
10311        path::{Path, PathBuf},
10312        sync::{Arc, Mutex},
10313        time::{Duration, Instant},
10314    };
10315    use subc_test_support::TestTempDir;
10316
10317    /// The stub, expected beside this test executable.
10318    ///
10319    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
10320    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
10321    /// failure then reads as a broken test rather than an unbuilt dependency.
10322    fn stub_path() -> PathBuf {
10323        let mut path = std::env::current_exe().expect("current_exe available in tests");
10324        path.pop();
10325        path.pop();
10326        path.push("fake-aft-stub.exe");
10327        assert!(
10328            path.exists(),
10329            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
10330             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
10331            path.display()
10332        );
10333        path
10334    }
10335
10336    /// Poll for the grandchild pid the stub records, and parse it.
10337    fn read_grandchild_pid(path: &Path) -> u32 {
10338        let deadline = Instant::now() + Duration::from_secs(10);
10339        loop {
10340            if let Ok(contents) = std::fs::read_to_string(path) {
10341                if let Ok(pid) = contents.trim().parse() {
10342                    return pid;
10343                }
10344            }
10345            assert!(
10346                Instant::now() < deadline,
10347                "the stub never recorded a grandchild pid at {}",
10348                path.display()
10349            );
10350            std::thread::sleep(Duration::from_millis(10));
10351        }
10352    }
10353
10354    /// Everything one fixture run needs, so the two tests below differ in exactly
10355    /// one place: whether the child is contained.
10356    struct Fixture {
10357        _dir: TestTempDir,
10358        module_id: String,
10359        grandchild: u32,
10360        child: Option<SupervisedChild>,
10361        registry: Arc<Registry>,
10362        snapshot: Arc<Mutex<SupervisorSnapshot>>,
10363        terminal_ring: Arc<Mutex<TerminalRing>>,
10364        spawn_events: SpawnEventFeed,
10365    }
10366
10367    fn fixture(label: &str, module_id: &str) -> Fixture {
10368        let dir = TestTempDir::new(label);
10369        let pid_file = dir.join("grandchild.pid");
10370        let supervisor = Supervisor::new(
10371            Arc::new(Registry::default()),
10372            RestartPolicy::new(3, Duration::ZERO),
10373        );
10374        let runtime = supervisor.runtime_config();
10375        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10376        let spec = ModuleSpec {
10377            module_id: module_id.to_string(),
10378            program: stub_path(),
10379            // Zero args deliberately: a `--subc` argument would make the stub dial
10380            // a daemon that is not there, and the failure would land in the same
10381            // stderr ring this fixture exists to keep quiet.
10382            args: Vec::new(),
10383            env: vec![
10384                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10385                (
10386                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
10387                    pid_file.display().to_string(),
10388                ),
10389            ],
10390            reserved: false,
10391            reserved_prefixes: Vec::new(),
10392            protocol: ModuleProtocol::Subc,
10393            overlap: Default::default(),
10394        };
10395        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
10396            .expect("spawn the supervised fixture");
10397        let grandchild = read_grandchild_pid(&pid_file);
10398        Fixture {
10399            _dir: dir,
10400            module_id: module_id.to_string(),
10401            grandchild,
10402            child: Some(child),
10403            registry: Arc::new(Registry::default()),
10404            snapshot,
10405            terminal_ring: Arc::clone(&runtime.terminal_ring),
10406            spawn_events: SpawnEventFeed::default(),
10407        }
10408    }
10409
10410    impl Fixture {
10411        /// Drain through the supervisor's own teardown path.
10412        async fn drain(&mut self) {
10413            let child = self
10414                .child
10415                .take()
10416                .expect("the fixture child is still present");
10417            drain_child_to_state(
10418                &self.module_id,
10419                ModuleProtocol::Subc,
10420                // No forwarding table in this fixture, so nothing reaches the
10421                // child over a connection.
10422                StopNotice::NotSent,
10423                &self.registry,
10424                &self.snapshot,
10425                &self.terminal_ring,
10426                &self.spawn_events,
10427                child,
10428                Duration::from_millis(500),
10429                ModuleState::Stopped,
10430                Some(false),
10431            )
10432            .await
10433            .expect("drain the supervised fixture");
10434        }
10435    }
10436
10437    /// Teardown reaps the grandchild, not merely the direct child.
10438    ///
10439    /// This is the assertion the change exists for. Before containment the
10440    /// grandchild survived: it is a separate process, and `start_kill` is
10441    /// `TerminateProcess` scoped to one pid.
10442    #[tokio::test]
10443    async fn teardown_reaps_the_grandchild() {
10444        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
10445        let grandchild = fixture.grandchild;
10446
10447        assert!(
10448            subc_jobobject::process_exists(grandchild),
10449            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
10450        );
10451
10452        fixture.drain().await;
10453
10454        assert!(
10455            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10456            "grandchild {grandchild} outlived module teardown: the tree was not contained"
10457        );
10458    }
10459
10460    /// The mutation control: with containment withheld, the grandchild survives
10461    /// the same kill.
10462    ///
10463    /// This is the defect reproduction from #109 — a direct-child kill reaches
10464    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
10465    /// supervisor because `spawn_and_mark_running` now always contains on
10466    /// Windows, which is the point: there is no longer a path that spawns
10467    /// uncontained, so the control has to construct one.
10468    ///
10469    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
10470    /// grandchild ever dies here, that test is passing for a reason unrelated to
10471    /// the job object and the containment claim is unproven.
10472    #[test]
10473    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
10474        let dir = TestTempDir::new("teardown-uncontained");
10475        let pid_file = dir.join("grandchild.pid");
10476        let mut child = std::process::Command::new(stub_path())
10477            .env("FAKE_AFT_NEVER_CONNECT", "1")
10478            .env(
10479                "FAKE_AFT_GRANDCHILD_PID_FILE",
10480                pid_file.display().to_string(),
10481            )
10482            .stdin(std::process::Stdio::null())
10483            .stdout(std::process::Stdio::null())
10484            .stderr(std::process::Stdio::null())
10485            .spawn()
10486            .expect("spawn the uncontained fixture");
10487        let grandchild = read_grandchild_pid(&pid_file);
10488
10489        // Exactly what the pre-fix teardown did: kill the direct child.
10490        child.kill().expect("kill the direct child");
10491        let _ = child.wait();
10492
10493        assert!(
10494            subc_jobobject::process_exists(grandchild),
10495            "grandchild {grandchild} died with the direct child, so this control no longer \
10496             distinguishes contained from uncontained teardown and the regression test is \
10497             passing vacuously"
10498        );
10499
10500        // The orphan this control demonstrates is the leak the fix prevents, so
10501        // the control must not leave one behind.
10502        kill_tree(grandchild);
10503    }
10504
10505    /// Crash durability: closing the containment handle reaps the tree with no
10506    /// teardown code running at all.
10507    ///
10508    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
10509    /// call anything — and it is why containment is a kernel property of the
10510    /// handle rather than a step in the drain. Discovered by getting the
10511    /// mutation control wrong: clearing `job` to "disable" containment instead
10512    /// killed the tree, which is the guarantee, not a mistake.
10513    #[tokio::test]
10514    async fn dropping_containment_reaps_the_grandchild() {
10515        let mut fixture = fixture("drop-containment", "tree-drop");
10516        let grandchild = fixture.grandchild;
10517
10518        assert!(subc_jobobject::process_exists(grandchild));
10519
10520        // No `drain` call, no kill: dropping the handle is the entire mechanism.
10521        fixture.child.as_mut().expect("child present").job = None;
10522
10523        assert!(
10524            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10525            "grandchild {grandchild} survived the containment handle closing, so a daemon \
10526             crash would leave the tree behind"
10527        );
10528    }
10529
10530    /// Kill a pid and its tree, then confirm it is gone.
10531    fn kill_tree(pid: u32) {
10532        let _ = std::process::Command::new("taskkill.exe")
10533            .args(["/PID", &pid.to_string(), "/T", "/F"])
10534            .stdin(std::process::Stdio::null())
10535            .stdout(std::process::Stdio::null())
10536            .stderr(std::process::Stdio::null())
10537            .status();
10538        assert!(
10539            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
10540            "could not clean up grandchild {pid}"
10541        );
10542    }
10543}
10544
10545/// The daemon's real spawn path hands a subc-wire child its launch nonce on
10546/// descriptor 3, without an environment copy. The shell records the nonce
10547/// and its environment after exec so these tests observe the real handover.
10548#[cfg(all(test, unix))]
10549mod launch_nonce_descriptor_tests {
10550    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
10551    use crate::stderr_tail::{StderrRing, StderrTailConfig};
10552    use std::{
10553        path::PathBuf,
10554        sync::{Arc, Mutex},
10555        time::{Duration, Instant},
10556    };
10557    use subc_test_support::TestTempDir;
10558
10559    async fn probe(role: super::SpawnRole) {
10560        let scratch = TestTempDir::new("launch-nonce-descriptor");
10561        let fd_copy = scratch.join("from-descriptor");
10562        let env_copy = scratch.join("environment");
10563        let script = format!(
10564            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
10565            fd = fd_copy.display(), env = env_copy.display(),
10566        );
10567        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
10568        let spec = ModuleSpec {
10569            module_id: "nonce-descriptor-probe".to_string(),
10570            program: PathBuf::from("/bin/sh"),
10571            args: vec!["-c".to_string(), script],
10572            env: vec![
10573                xdg("XDG_DATA_HOME"),
10574                xdg("XDG_RUNTIME_DIR"),
10575                xdg("XDG_CONFIG_HOME"),
10576                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
10577            ],
10578            reserved: true,
10579            reserved_prefixes: Vec::new(),
10580            protocol: ModuleProtocol::Subc,
10581            overlap: Default::default(),
10582        };
10583        let handle = SupervisorHandle::new();
10584        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10585        let roster = ChildRoster::default();
10586        let child = super::spawn_child_in_slot(
10587            &spec,
10588            None,
10589            Some(&handle),
10590            &ring,
10591            None,
10592            &roster,
10593            #[cfg(target_os = "linux")]
10594            None,
10595            role,
10596            matches!(role, super::SpawnRole::SwapCandidate),
10597        )
10598        .expect("spawn probe");
10599        let deadline = Instant::now() + Duration::from_secs(10);
10600        while !(fd_copy.exists() && env_copy.exists()) {
10601            assert!(Instant::now() < deadline, "probe never wrote its copies");
10602            tokio::time::sleep(Duration::from_millis(20)).await;
10603        }
10604        let nonce = std::fs::read_to_string(fd_copy).unwrap();
10605        assert!(!nonce.is_empty());
10606        let environment = std::fs::read_to_string(env_copy).unwrap();
10607        assert!(environment
10608            .lines()
10609            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
10610        let copy = environment
10611            .lines()
10612            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
10613        assert_eq!(
10614            copy, None,
10615            "Unix children must never receive the environment nonce"
10616        );
10617        if matches!(role, super::SpawnRole::Plain) {
10618            assert_eq!(
10619                handle.spawn_nonce(&spec.module_id).as_deref(),
10620                Some(nonce.as_str())
10621            );
10622        }
10623        drop(child);
10624    }
10625
10626    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10627    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
10628        probe(super::SpawnRole::Plain).await;
10629    }
10630
10631    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10632    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
10633        probe(super::SpawnRole::SwapCandidate).await;
10634    }
10635}
10636
10637#[cfg(all(test, target_os = "linux"))]
10638mod cgroup_containment_tests {
10639    use super::*;
10640    use subc_test_support::TestTempDir;
10641
10642    fn running(pid: u32) -> bool {
10643        // An orphan can remain a zombie until the container init reaps it.
10644        std::fs::read_to_string(format!("/proc/{pid}/stat"))
10645            .ok()
10646            .and_then(|stat| {
10647                stat.rsplit_once(") ")
10648                    .map(|(_, rest)| rest.starts_with('Z'))
10649            })
10650            .is_some_and(|zombie| !zombie)
10651    }
10652
10653    #[tokio::test]
10654    async fn linux_teardown_reaps_the_grandchild() {
10655        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
10656    }
10657
10658    #[tokio::test]
10659    async fn linux_shutdown_straggler_reaps_the_grandchild() {
10660        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
10661    }
10662
10663    async fn teardown_tree(test_name: &str, shutdown: bool) {
10664        let dir = TestTempDir::new(test_name);
10665        let root = PathBuf::from(format!(
10666            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
10667            std::process::id(),
10668            unix_ms_now()
10669        ));
10670        if let Err(error) = std::fs::create_dir(&root) {
10671            assert!(
10672                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
10673                "required cgroup test cannot execute: {error}"
10674            );
10675            eprintln!(
10676                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
10677                root.display()
10678            );
10679            return;
10680        }
10681        let placement = subc_cgroup::prepare_at(&root)
10682            .expect("prepare isolated kernel cgroup")
10683            .expect("isolated cgroup is delegated");
10684        let module_id = "tree-teardown";
10685        let module = placement
10686            .module_path(module_id)
10687            .expect("create isolated module cgroup");
10688        if !module.join("cgroup.kill").exists() {
10689            std::fs::remove_dir(&module).unwrap();
10690            std::fs::remove_dir(root.join("subc-modules")).unwrap();
10691            std::fs::remove_dir(&root).unwrap();
10692            assert!(
10693                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
10694                "required cgroup.kill interface unavailable"
10695            );
10696            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
10697            return;
10698        }
10699        let supervisor = Supervisor::new(
10700            Arc::new(Registry::default()),
10701            RestartPolicy::new(3, Duration::ZERO),
10702        )
10703        .with_cgroup_placement(Some(placement));
10704        let mut runtime = supervisor.runtime_config();
10705        runtime.child_roster = runtime
10706            .child_roster
10707            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
10708        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10709        let pid_file = dir.join("grandchild.pid");
10710        let spec = ModuleSpec {
10711            module_id: module_id.to_string(),
10712            program: PathBuf::from("/bin/sh"),
10713            args: vec![
10714                "-c".into(),
10715                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
10716                "fixture".into(),
10717                pid_file.display().to_string(),
10718            ],
10719            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
10720                .into_iter()
10721                .map(|key| (key.to_string(), dir.display().to_string()))
10722                .collect(),
10723            reserved: false,
10724            reserved_prefixes: Vec::new(),
10725            protocol: ModuleProtocol::None,
10726            overlap: Default::default(),
10727        };
10728        let child =
10729            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
10730        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
10731        let grandchild: u32 = loop {
10732            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
10733                if let Ok(pid) = contents.trim().parse() {
10734                    break pid;
10735                }
10736            }
10737            assert!(
10738                tokio::time::Instant::now() < deadline,
10739                "grandchild pid was not recorded"
10740            );
10741            tokio::time::sleep(Duration::from_millis(10)).await;
10742        };
10743        assert!(
10744            running(grandchild),
10745            "grandchild must be alive before teardown"
10746        );
10747        if shutdown {
10748            let mut child = child;
10749            crate::child_roster::end_children_for_daemon_shutdown(
10750                &runtime.child_roster,
10751                false,
10752                std::future::pending(),
10753            )
10754            .await;
10755            child.wait().await.expect("reap shutdown straggler");
10756        } else {
10757            drain_child_to_state(
10758                module_id,
10759                ModuleProtocol::None,
10760                StopNotice::NotSent,
10761                &Registry::default(),
10762                &snapshot,
10763                &runtime.terminal_ring,
10764                &SpawnEventFeed::default(),
10765                child,
10766                Duration::from_millis(100),
10767                ModuleState::Stopped,
10768                Some(false),
10769            )
10770            .await
10771            .expect("real supervisor teardown");
10772        }
10773        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
10774        while running(grandchild) && tokio::time::Instant::now() < deadline {
10775            tokio::time::sleep(Duration::from_millis(10)).await;
10776        }
10777        let survived = running(grandchild);
10778        // Kill a surviving grandchild so a failed test does not leave it behind.
10779        if survived {
10780            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
10781            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
10782            tokio::time::sleep(Duration::from_millis(100)).await;
10783        }
10784        if module.exists() {
10785            std::fs::remove_dir(&module).expect("remove empty module cgroup");
10786        }
10787        std::fs::remove_dir(root.join("subc-modules")).unwrap();
10788        std::fs::remove_dir(&root).unwrap();
10789        assert!(
10790            !survived,
10791            "grandchild {grandchild} outlived module teardown"
10792        );
10793        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
10794    }
10795}