Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113struct SupervisedChild {
114    child: Child,
115    /// The name of this process's cgroup: the module id, or for a swap
116    /// candidate the alternate name (see `swap::cgroup_name`).
117    #[cfg(target_os = "linux")]
118    module_id: String,
119    #[cfg(target_os = "linux")]
120    cgroup_placement: Option<subc_cgroup::Placement>,
121    /// The job that contains this child and every process it spawns (issue #109).
122    ///
123    /// Dropping this handle is what reaps a surviving tree when no supervisor
124    /// code runs — a daemon crash — because the job carries
125    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
126    ///
127    /// That limit is not crash-only, and the difference is worth knowing: a
128    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
129    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
130    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
131    /// module at once. Before this change they survived that, saw EOF on the
132    /// control socket, and ran their own teardown; Unix keeps that path
133    /// deliberately, so a module can seal a WAL or close a capture rather than
134    /// be killed mid-write. So this trades graceful teardown on every Windows
135    /// daemon stop for containment on a crash, which is the right way round
136    /// today: orphaned GPU workers are a reported, recurring problem, and the
137    /// modules that write most heavily do not run on Windows.
138    ///
139    /// The fix is a real Windows stop path — the daemon draining before it
140    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
141    /// reaches only what the drain left behind, which is what it should reach.
142    #[cfg(windows)]
143    job: Option<subc_jobobject::JobObject>,
144    stdout_pump: Option<JoinHandle<()>>,
145    stderr_pump: Option<StderrPump>,
146    stderr_ring: Arc<Mutex<StderrRing>>,
147    spawned_at_ms: u64,
148    spawned_from: PathBuf,
149    spawned_file_identity: Option<SpawnedFileIdentity>,
150    process_start_time: Option<u64>,
151    process_identity: Option<ProcessIdentity>,
152    pid: u32,
153    /// This process's entry in the daemon's child roster, released when the
154    /// process is reaped or this handle is dropped.
155    roster_guard: Option<crate::child_roster::RosterGuard>,
156}
157
158impl SupervisedChild {
159    fn id(&self) -> Option<u32> {
160        Some(self.pid)
161    }
162
163    fn process_identity(&self) -> Option<ProcessIdentity> {
164        self.process_identity
165    }
166
167    async fn wait(&mut self) -> io::Result<ExitStatus> {
168        // The roster entry is NOT released here. A daemon shutdown waits for the
169        // roster to empty and then exits the process, so releasing at the reap
170        // let it exit before the exit handler wrote this child's terminal record
171        // (the stderr drain and snapshot update sit in between), and the
172        // shutdown's own `daemon_shutdown` record was intermittently lost. The
173        // caller releases it after recording the exit (`release_roster`), and
174        // dropping the handle releases it too.
175        let result = self.child.wait().await;
176        #[cfg(target_os = "linux")]
177        if result.is_ok() {
178            if let Some(placement) = self.cgroup_placement.take() {
179                remove_module_cgroup(&placement, &self.module_id);
180            }
181        }
182        result
183    }
184
185    /// Releases this child's daemon-shutdown roster entry once its exit has
186    /// been recorded. The pid is already reaped and free for reuse, so the
187    /// entry must not outlive the record any longer than that.
188    fn release_roster(&mut self) {
189        self.roster_guard = None;
190    }
191
192    /// Kill the child, and on Windows the whole tree it spawned (issue #109).
193    ///
194    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
195    /// helper process leaked the helper — the Synapse embedding module's CUDA
196    /// worker holds the GPU allocation, so the leak cost VRAM until the next
197    /// restart of something else. Terminating the job reaches grandchildren that
198    /// a tree walk cannot, including one whose parent has already exited and
199    /// been reparented away.
200    ///
201    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
202    /// direct-child kill still decides the outcome, so containment can never
203    /// change whether a module is reported as stopped.
204    fn start_kill(&mut self) -> io::Result<()> {
205        #[cfg(windows)]
206        if let Some(job) = &self.job {
207            if let Err(error) = job.terminate() {
208                debug!(
209                    error = %error,
210                    "job termination failed; the direct-child kill still owns the outcome"
211                );
212            }
213        }
214        self.child.start_kill()
215    }
216
217    async fn drain_stderr(&mut self, module_id: &str) {
218        if let Some(mut pump) = self.stdout_pump.take() {
219            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
220                Ok(Ok(())) => {}
221                Ok(Err(error)) => {
222                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
223                }
224                Err(_) => {
225                    pump.abort();
226                    warn!(
227                        module_id,
228                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
229                        "stdout pump did not drain before restart; stopped it before the next process"
230                    );
231                }
232            }
233        }
234
235        let Some(pump) = self.stderr_pump.take() else {
236            return;
237        };
238        settle_stderr_pump(
239            module_id,
240            &self.stderr_ring,
241            pump,
242            STDERR_PUMP_DRAIN_TIMEOUT,
243        )
244        .await;
245    }
246}
247
248/// The reader task for one process's stderr, with the ring generation its
249/// lines are attributed to.
250struct StderrPump {
251    task: JoinHandle<()>,
252    generation: u64,
253}
254
255/// Retire an exited process's stderr reader and wait up to `bound` for it to
256/// reach EOF. A reader still running at the bound is detached, not stopped: it
257/// keeps filling the exited process's section of the ring until its pipe
258/// closes, and the tail reads `Incomplete` until then. See
259/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
260async fn settle_stderr_pump(
261    module_id: &str,
262    ring: &Arc<Mutex<StderrRing>>,
263    pump: StderrPump,
264    bound: Duration,
265) {
266    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
267    let StderrPump {
268        mut task,
269        generation,
270    } = pump;
271    lock().retire_pump(generation);
272    match timeout(bound, &mut task).await {
273        Ok(Ok(())) => {}
274        Ok(Err(err)) => {
275            let mut ring = lock();
276            ring.mark_incomplete(format!("stderr pump ended unexpectedly: {err}"));
277            ring.finish_pump(generation);
278            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
279        }
280        Err(_) => {
281            // Dropping the handle detaches the task; it ends at EOF on its pipe.
282            drop(task);
283            lock().mark_pump_late(
284                generation,
285                format!(
286                    "stderr of the exited process had not reached EOF {bound:?} after it was \
287                     retired (a descendant may still hold the pipe open); lines it still \
288                     writes are kept in that process's section"
289                ),
290            );
291            warn!(
292                module_id,
293                waited = ?bound,
294                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
295            );
296        }
297    }
298}
299
300fn registration_release_events() -> &'static watch::Sender<u64> {
301    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
302    EVENTS.get_or_init(|| {
303        let (sender, _receiver) = watch::channel(0);
304        sender
305    })
306}
307
308pub(crate) fn notify_registration_release() {
309    let events = registration_release_events();
310    let next_generation = (*events.borrow()).wrapping_add(1);
311    events.send_replace(next_generation);
312}
313
314/// How to launch one singleton module process.
315#[derive(Debug, Clone, PartialEq, Eq)]
316pub struct ModuleSpec {
317    pub module_id: String,
318    pub program: PathBuf,
319    pub args: Vec<String>,
320    pub env: Vec<(String, String)>,
321    /// When true this is a reserved module: each spawn gets a fresh one-time launch
322    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
323    /// process can register this module_id (a security-boundary module like the
324    /// credential vault must not be impersonable while it is down/restarting).
325    pub reserved: bool,
326    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
327    /// Prefixes come from daemon config and must end in `:` before they reach the
328    /// supervisor; the owner module's current spawn nonce authorizes claims under
329    /// each prefix.
330    pub reserved_prefixes: Vec<String>,
331    /// The wire protocol this module speaks, as DECLARED in daemon config.
332    ///
333    /// [`ModuleProtocol::None`] changes five things and nothing else: health
334    /// probing is suppressed, teardown sends SIGTERM before waiting,
335    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
336    /// and NO launch nonce, and a clean exit the daemon did not request is
337    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
338    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
339    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
340    /// because a process ignores an environment variable it does not read.
341    ///
342    /// The argument is the part that cannot be "harmless to a process that
343    /// ignores it": a stock binary exits on an unknown flag before it listens
344    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
345    /// first conformance run against this mode found it. The nonce is withheld
346    /// because a process that will never present it gains nothing from holding
347    /// it, and a secret in the environment of a process that does not need it is
348    /// a leak surface for no benefit.
349    pub protocol: ModuleProtocol,
350    /// Whether two processes of this module may run at once, which is what a
351    /// blue/green swap does for the length of its overlap. Declared in daemon
352    /// config because the daemon must be able to answer it while the module is
353    /// down, and so a module cannot talk itself into it after registering.
354    pub overlap: ModuleOverlap,
355}
356
357/// Whether a module tolerates a second process of itself running alongside.
358///
359/// Most modules are single-writer on their store (a WAL, a capture log, a
360/// resident index behind a writer barrier), and two processes on one store
361/// corrupt it. So a swap, which overlaps the old and new process by design,
362/// is refused unless the module's config opts in with `overlap: "safe"`.
363#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
364pub enum ModuleOverlap {
365    /// Never run two processes of this module at once. The default.
366    #[default]
367    Exclusive,
368    /// The module has said a second process of itself is harmless for the
369    /// length of a swap.
370    ///
371    /// Declare it only if a second instance can run for a few seconds without
372    /// touching ANY single-writer store: every database, WAL, index, projector
373    /// and scheduled job the module owns. A lease on part of that state is not
374    /// enough. broca's session lease guards WAL appends while its run index, its
375    /// store projector and its archive fold timer (which unlinks live WAL files)
376    /// stay single-writer, so broca is exclusive despite holding a lease. The
377    /// refusal only fires after this has been decided, so the decision is the
378    /// check.
379    Safe,
380}
381
382impl ModuleOverlap {
383    pub fn as_str(self) -> &'static str {
384        match self {
385            Self::Exclusive => "exclusive",
386            Self::Safe => "safe",
387        }
388    }
389}
390
391/// Environment variable telling a spawned module which case it was started
392/// for, before it sends HELLO. Only a swap candidate carries it, as
393/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
394///
395/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
396/// longer because nobody waits on it, while a plain restart must flip ready
397/// quickly because callers see `module_warming` until it does. Absence means
398/// plain restart, the safe reading. The daemon trusts nothing about it; the
399/// candidate is proven by its launch nonce at HELLO.
400pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
401/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
402pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
403/// How long a swap waits for its candidate to register and declare itself
404/// ready when the operator does not say. A module warming as a swap candidate
405/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
406/// daemon allows that plus time to start the process and send HELLO.
407pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
408
409/// Bounded restart policy for crash exits.
410///
411/// `max_restarts` is the number of replacement processes allowed after the
412/// initial spawn WITHIN `window`. After that many crash restarts inside one
413/// window the module enters [`ModuleState::Failed`] and the supervisor stops
414/// the crash loop.
415///
416/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
417/// and that only survived because crashes were rare: a module that crashed
418/// three times across a week was disabled forever by crashes that had nothing
419/// to do with each other. That stopped being survivable once modules began
420/// exiting non-zero whenever the daemon's connection to them drops, because
421/// then every daemon-side connection drop spends a unit of the same budget and
422/// one flappy hour permanently stops a healthy module. Restarts older than
423/// `window` release their slot, so a module that crashed twice yesterday has a
424/// full budget today, while a genuine crash loop -- which is fast by
425/// definition -- still reaches the cap and stops.
426#[derive(Debug, Clone, Copy, PartialEq, Eq)]
427pub struct RestartPolicy {
428    pub max_restarts: u32,
429    /// Base delay before a crash replacement. The actual delay escalates with
430    /// the number of recent crash replacements and is capped by `max_backoff`.
431    pub backoff: Duration,
432    /// Maximum delay before a crash replacement.
433    pub max_backoff: Duration,
434    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
435    /// budget effectively infinite (nothing is ever in-window), which is why
436    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
437    pub window: Duration,
438}
439
440impl RestartPolicy {
441    /// A policy with the default crash window. Callers that care about the
442    /// window say so with [`Self::with_window`]; the ones that do not are
443    /// asking for the standard rate limit, not for no limit.
444    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
445        Self {
446            max_restarts,
447            backoff,
448            max_backoff: DEFAULT_MAX_BACKOFF,
449            window: DEFAULT_RESTART_WINDOW,
450        }
451    }
452
453    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
454        self.max_backoff = max_backoff;
455        self
456    }
457
458    pub fn with_window(mut self, window: Duration) -> Self {
459        self.window = window;
460        self
461    }
462
463    /// Calculate the capped exponential delay for the next crash replacement.
464    /// `restart_in_window` is zero for the first replacement after an operator
465    /// action (restart, reload, re-enable) cleared the crash ring, or after all
466    /// older crash replacements have aged out of the window.
467    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
468        if self.backoff.is_zero() || self.max_backoff.is_zero() {
469            return Duration::ZERO;
470        }
471
472        let mut delay = self.backoff;
473        for _ in 0..restart_in_window {
474            if delay >= self.max_backoff {
475                return self.max_backoff;
476            }
477            delay = delay
478                .checked_mul(10)
479                .unwrap_or(self.max_backoff)
480                .min(self.max_backoff);
481        }
482        delay.min(self.max_backoff)
483    }
484
485    /// The one sentence that explains a budget-exhausted stop, used for both the
486    /// log line and the terminal record so the two cannot drift. It names the
487    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
488    /// exactly what this budget is not.
489    fn budget_exhausted_detail(&self) -> String {
490        format!(
491            "crash budget exhausted: max_restarts={} within window_secs={}",
492            self.max_restarts,
493            self.window.as_secs()
494        )
495    }
496}
497
498impl Default for RestartPolicy {
499    fn default() -> Self {
500        Self {
501            max_restarts: DEFAULT_MAX_RESTARTS,
502            backoff: DEFAULT_BACKOFF,
503            max_backoff: DEFAULT_MAX_BACKOFF,
504            window: DEFAULT_RESTART_WINDOW,
505        }
506    }
507}
508
509#[derive(Debug, Clone, Copy, PartialEq, Eq)]
510struct CrashRestartSchedule {
511    restart_in_window: u32,
512    delay: Duration,
513}
514
515/// Whether the daemon itself will bring this module back after the exit being
516/// handled: it is enabled AND its in-window crash restarts are below the cap.
517///
518/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
519/// the window are dropped here rather than by a timer, so the count is right
520/// the moment somebody asks and no bookkeeping runs for idle modules.
521fn daemon_will_restart(
522    state: &mut SupervisorSnapshot,
523    policy: &RestartPolicy,
524    now: Instant,
525) -> bool {
526    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
527}
528
529const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
530const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
531const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
532const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
533
534#[derive(Debug, Clone, Copy, PartialEq, Eq)]
535pub enum HealthAction {
536    Report,
537    Restart,
538    Alert,
539}
540
541impl fmt::Display for HealthAction {
542    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
543        f.write_str(match self {
544            Self::Report => "report",
545            Self::Restart => "restart",
546            Self::Alert => "alert",
547        })
548    }
549}
550
551#[derive(Debug, Clone, Copy, PartialEq, Eq)]
552pub struct HealthConfig {
553    pub cadence: Duration,
554    pub deadline: Duration,
555    pub failure_threshold: u32,
556    pub on_degraded: HealthAction,
557    pub on_failing: HealthAction,
558    pub critical: bool,
559}
560
561impl Default for HealthConfig {
562    fn default() -> Self {
563        Self {
564            cadence: DEFAULT_HEALTH_CADENCE,
565            deadline: DEFAULT_HEALTH_DEADLINE,
566            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
567            on_degraded: HealthAction::Report,
568            on_failing: HealthAction::Report,
569            critical: false,
570        }
571    }
572}
573
574/// The supervisor's view of one module's health, relayed to clients over
575/// channel-0 and rendered by `ck health`.
576///
577/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
578/// stated here rather than only at the wire type a consumer reads. A reader can
579/// look up what `None` means; only a writer can silently change it, and the
580/// writer has no reason to go looking at a downstream contract before editing.
581///
582/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
583/// back to `None` on re-registration precisely so a respawned module does not
584/// carry its predecessor's timestamp — so an old value and an absent one call for
585/// opposite readings, and anything that defaulted this to a number would make a
586/// never-probed module indistinguishable from one probed at the epoch.
587///
588/// `detail` and `metrics` are `None` when the module published none on this
589/// probe, which does not mean it reported nothing wrong — it is also the shape
590/// when the probe never reached it. `last_probe_ms` is what separates those.
591#[derive(Debug, Clone, PartialEq)]
592pub struct ModuleHealthStatus {
593    pub status: SupervisorHealthStatus,
594    pub last_probe_ms: Option<u64>,
595    pub detail: Option<String>,
596    pub metrics: Option<Value>,
597    pub consecutive_failures: u32,
598    /// Number of replies received after a recurring health probe's deadline.
599    /// Unlike a timeout, every increment proves the module was alive.
600    pub late_answer_count: u64,
601    /// End-to-end latency of the newest late reply, measured from probe start.
602    pub last_late_answer_latency_ms: Option<u64>,
603    pub last_action: Option<String>,
604    /// Set together with `last_action`; the pair moves as one, and both being
605    /// absent means no escalation has ever been taken rather than that the last
606    /// one succeeded.
607    pub last_action_ms: Option<u64>,
608}
609
610impl Default for ModuleHealthStatus {
611    fn default() -> Self {
612        Self {
613            status: SupervisorHealthStatus::Unknown,
614            last_probe_ms: None,
615            detail: None,
616            metrics: None,
617            consecutive_failures: 0,
618            late_answer_count: 0,
619            last_late_answer_latency_ms: None,
620            last_action: None,
621            last_action_ms: None,
622        }
623    }
624}
625
626/// Typed lifecycle state for a supervised module.
627#[derive(Debug, Clone, Copy, PartialEq, Eq)]
628pub enum ModuleState {
629    Starting,
630    Running,
631    Unresponsive,
632    Restarting,
633    Draining,
634    Stopped,
635    Failed,
636    Disabled,
637}
638
639impl fmt::Display for ModuleState {
640    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
641        f.write_str(match self {
642            Self::Starting => "starting",
643            Self::Running => "running",
644            Self::Unresponsive => "unresponsive",
645            Self::Restarting => "restarting",
646            Self::Draining => "draining",
647            Self::Stopped => "stopped",
648            Self::Failed => "failed",
649            Self::Disabled => "disabled",
650        })
651    }
652}
653
654/// Supervisor classification of a child-process exit.
655#[derive(Debug, Clone, Copy, PartialEq, Eq)]
656pub enum ExitKind {
657    Clean,
658    Crash,
659    DeliberateSeverance,
660}
661
662impl From<ExitKind> for TerminalExitKind {
663    fn from(kind: ExitKind) -> Self {
664        match kind {
665            ExitKind::Clean => Self::Clean,
666            ExitKind::Crash => Self::Crash,
667            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
668        }
669    }
670}
671
672/// Exact process identity retained when a supervised module registers its
673/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
674#[derive(Debug, Clone, Copy, PartialEq, Eq)]
675pub(crate) struct ProcessIdentity {
676    pub(crate) pid: u32,
677    pub(crate) start_time: u64,
678}
679
680/// Last observed child exit, if any.
681#[derive(Debug, Clone, PartialEq, Eq)]
682pub struct ExitReport {
683    pub kind: ExitKind,
684    pub code: Option<i32>,
685    pub signal: Option<i32>,
686    pub at_ms: u64,
687}
688
689/// Point-in-time module status answerable by subc without forwarding to the
690/// module process.
691#[derive(Debug, Clone, PartialEq)]
692pub struct ModuleStatus {
693    pub module_id: String,
694    pub state: ModuleState,
695    pub enabled: bool,
696    pub process_alive: bool,
697    pub registration_active: bool,
698    /// The module's declared wire protocol, carried beside `live` because it is
699    /// what makes `live` readable: the two fields answer one question together.
700    pub protocol: ModuleProtocol,
701    /// Whether the module is serving, under the strongest definition the daemon
702    /// can assert for its protocol.
703    ///
704    /// A subc module must also be REGISTERED: its process being alive says
705    /// nothing about whether it can take a request. A `protocol: "none"` module
706    /// never registers, so that term is dropped and this falls back to "enabled,
707    /// running, and the process the daemon launched is alive" -- which is all
708    /// the daemon observes about a process that speaks no subc wire. It stays a
709    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
710    /// rather than printing it bare.
711    pub live: bool,
712    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
713    /// restarts have already released their slot, so this count can go down
714    /// without anybody touching the module.
715    pub restart_count: u32,
716    /// Replacement processes spawned over this module's entire supervisor lifetime;
717    /// unlike `restart_count`, this value is never reset by an operator action
718    /// and never falls out of a window.
719    pub lifetime_restarts: u32,
720    pub spawn_generation: u64,
721    /// The budget `restart_count` is spent against. Carried alongside the count
722    /// because the count alone does not say how close the module is to being
723    /// disabled, and reporting one without the other is what makes an
724    /// about-to-be-retired module look ordinary.
725    pub max_restarts: u32,
726    /// The span `restart_count` is counted over. Carried with the pair above for
727    /// the same reason they are carried together: "2 of 3" means one thing for a
728    /// ten-minute window and something else entirely for a lifetime.
729    pub restart_window: Duration,
730    /// Effective drain and restart timing policy used by this running module.
731    /// These values are carried together with the restart budget so status
732    /// readers can compare configured intent with what the supervisor applied.
733    pub drain_timeout: Duration,
734    pub restart_backoff: Duration,
735    pub restart_max_backoff: Duration,
736    pub pid: Option<u32>,
737    pub spawned_at_ms: Option<u64>,
738    pub spawned_from: Option<PathBuf>,
739    pub process_start_time: Option<u64>,
740    pub last_exit: Option<ExitReport>,
741    pub health: ModuleHealthStatus,
742}
743
744#[derive(Debug, Clone, PartialEq)]
745struct SupervisorSnapshot {
746    state: ModuleState,
747    enabled: bool,
748    process_alive: bool,
749    /// When each crash restart was spent, oldest first. This IS the crash
750    /// budget: its in-window length is the count an operator sees and the count
751    /// the restart decision is made against, so there is no second counter that
752    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
753    /// operator actions that used to zero the old lifetime counter.
754    crash_restarts: VecDeque<Instant>,
755    lifetime_restarts: u32,
756    /// Successful child spawns in this daemon incarnation.
757    ///
758    /// `lifetime_restarts` was considered and rejected: it starts at zero
759    /// (line 640), successful initial/operator spawns in `set_running` do not
760    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
761    /// increments before a successful replacement exists (lines 604, 3846,
762    /// and 3921), so a failed spawn can consume it. This counter moves only
763    /// when a live PID is accepted below.
764    spawn_generation: u64,
765    pid: Option<u32>,
766    spawned_at_ms: Option<u64>,
767    spawned_from: Option<PathBuf>,
768    spawned_file_identity: Option<SpawnedFileIdentity>,
769    process_start_time: Option<u64>,
770    deliberate_severance: Option<ProcessIdentity>,
771    last_exit: Option<ExitReport>,
772    health: ModuleHealthStatus,
773    /// Whether the current process was started as a swap candidate and so
774    /// lives in the module's alternate cgroup. The next swap's candidate takes
775    /// the other one, so the two processes of a swap never share a cgroup. A
776    /// plain spawn always uses the primary cgroup.
777    in_alternate_slot: bool,
778    /// Whether the current `Draining` state ends in a replacement process
779    /// (restart, reload, health restart) rather than a stop. Only meaningful
780    /// while `state` is `Draining`; every entry into that state rewrites it.
781    /// It is what lets route.open answer the retryable `module_reloading` to a
782    /// consumer that reaches a still-registered process mid-restart, instead of
783    /// the `supervisor_not_live` a stop or disable deserves.
784    draining_to_replace: bool,
785    /// Whether a configuration update has been applied since the current
786    /// process was spawned, so that process runs an older spec than the one
787    /// the supervisor now holds. A queued restart is only coalesced into a
788    /// fresher process when this is false: a restart requested to pick up a
789    /// new configuration must not be satisfied by a process that predates it.
790    configuration_updated_since_spawn: bool,
791}
792
793impl SupervisorSnapshot {
794    fn starting() -> Self {
795        Self::new(ModuleState::Starting, true)
796    }
797
798    fn disabled() -> Self {
799        Self::new(ModuleState::Disabled, false)
800    }
801
802    fn failed() -> Self {
803        Self::new(ModuleState::Failed, true)
804    }
805
806    /// Crash restarts still inside `window`, having dropped the ones that are
807    /// not. Pruning on read is what makes the budget a rate: an instant older
808    /// than the window stops holding a slot the moment anybody counts.
809    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
810        while let Some(oldest) = self.crash_restarts.front() {
811            if now.duration_since(*oldest) > window {
812                self.crash_restarts.pop_front();
813            } else {
814                break;
815            }
816        }
817        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
818    }
819
820    /// Spend one unit of the crash budget and record the restart in the ledger.
821    ///
822    /// The ring is bounded by the cap because more than `max_restarts` in-window
823    /// instants can never be reached (the caller refuses the restart first), so
824    /// anything beyond that is an unbounded queue waiting to happen.
825    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
826        self.crash_restarts.push_back(now);
827        while self.crash_restarts.len() > policy.max_restarts as usize {
828            self.crash_restarts.pop_front();
829        }
830        self.lifetime_restarts += 1;
831    }
832
833    /// Reserve one crash-restart slot and calculate the delay before respawning.
834    /// The count is captured before recording this restart, so the first retry
835    /// uses the base delay and each later in-window retry escalates once.
836    fn next_crash_restart(
837        &mut self,
838        policy: &RestartPolicy,
839        now: Instant,
840    ) -> Option<CrashRestartSchedule> {
841        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
842        if restart_in_window >= policy.max_restarts {
843            return None;
844        }
845        self.record_crash_restart(policy, now);
846        Some(CrashRestartSchedule {
847            restart_in_window,
848            delay: policy.delay_for_restart(restart_in_window),
849        })
850    }
851
852    /// Give the module its full budget back, as an operator restart, reload, or
853    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
854    /// ledger of what actually happened, and an operator action does not unmake
855    /// the crashes.
856    fn clear_crash_restarts(&mut self) {
857        self.crash_restarts.clear();
858    }
859
860    fn new(state: ModuleState, enabled: bool) -> Self {
861        Self {
862            state,
863            enabled,
864            process_alive: false,
865            crash_restarts: VecDeque::new(),
866            lifetime_restarts: 0,
867            spawn_generation: 0,
868            pid: None,
869            spawned_at_ms: None,
870            spawned_from: None,
871            spawned_file_identity: None,
872            process_start_time: None,
873            deliberate_severance: None,
874            last_exit: None,
875            health: ModuleHealthStatus::default(),
876            in_alternate_slot: false,
877            draining_to_replace: false,
878            configuration_updated_since_spawn: false,
879        }
880    }
881}
882
883type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
884
885type SpawnSubscriberKey = (ConnectionId, u64);
886
887#[derive(Debug)]
888struct SpawnSubscriber {
889    version: u8,
890    frames: mpsc::Sender<Frame>,
891    /// Tells this subscriber's forwarder that it was dropped for lagging, and
892    /// from which event. The full frame channel cannot carry that news, so it
893    /// travels beside it; see `SpawnEventFeed::subscribe`.
894    lagged: Option<oneshot::Sender<SpawnCursor>>,
895}
896
897#[derive(Debug)]
898struct SpawnEventState {
899    daemon_incarnation: String,
900    seq: u64,
901    capacity: usize,
902    live: HashMap<String, LiveSpawn>,
903    generations: HashMap<String, u64>,
904    events: VecDeque<SpawnEvent>,
905    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
906}
907
908impl Default for SpawnEventState {
909    fn default() -> Self {
910        Self {
911            daemon_incarnation: "unconfigured".to_string(),
912            seq: 0,
913            capacity: SPAWN_EVENT_RING_CAPACITY,
914            live: HashMap::new(),
915            generations: HashMap::new(),
916            events: VecDeque::new(),
917            subscribers: HashMap::new(),
918        }
919    }
920}
921
922#[derive(Debug, Clone, Default)]
923struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
924
925#[derive(Debug, Clone, PartialEq, Eq)]
926pub(crate) enum SpawnSubscribeRefusal {
927    ForeignIncarnation { current: String },
928    TooOld { oldest: SpawnCursor },
929    Frame(String),
930}
931
932impl SpawnEventFeed {
933    fn configure_incarnation(&self, daemon_incarnation: String) {
934        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
935        state.daemon_incarnation = daemon_incarnation;
936        state.seq = 0;
937        state.live.clear();
938        state.generations.clear();
939        state.events.clear();
940        state.subscribers.clear();
941    }
942
943    fn cursor(state: &SpawnEventState) -> SpawnCursor {
944        SpawnCursor {
945            daemon_incarnation: state.daemon_incarnation.clone(),
946            seq: state.seq,
947        }
948    }
949
950    fn snapshot(&self) -> SpawnSnapshot {
951        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
952        let mut live = state.live.values().cloned().collect::<Vec<_>>();
953        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
954        SpawnSnapshot {
955            cursor: Self::cursor(&state),
956            ring_bound: state.capacity as u64,
957            live,
958        }
959    }
960
961    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
962        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
963        let generation = state
964            .generations
965            .get(module_id)
966            .copied()
967            .unwrap_or(0)
968            .checked_add(1)
969            .expect("spawn generation exhausted");
970        state.generations.insert(module_id.to_string(), generation);
971        let live = LiveSpawn {
972            module_id: module_id.to_string(),
973            spawn_generation: generation,
974            pid,
975            spawned_at_ms,
976        };
977        state.live.insert(module_id.to_string(), live);
978        Self::emit_locked(
979            &mut state,
980            SpawnEventKind::Spawned,
981            module_id.to_string(),
982            generation,
983            pid,
984            None,
985            None,
986        );
987        generation
988    }
989
990    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
991        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
992        let Some(live) = state.live.remove(module_id) else {
993            warn!(
994                module_id,
995                "terminal record had no live spawn event identity"
996            );
997            return;
998        };
999        Self::emit_locked(
1000            &mut state,
1001            SpawnEventKind::Exited,
1002            module_id.to_string(),
1003            live.spawn_generation,
1004            live.pid,
1005            exit_code,
1006            exit_signal,
1007        );
1008    }
1009
1010    /// Report the exit of a process that a swap has already replaced.
1011    ///
1012    /// `emit_exited` removes the module's live entry, which after a swap's
1013    /// cutover describes the promoted candidate, not the old process now
1014    /// exiting. This emits the old generation's exit and leaves the live entry
1015    /// alone unless it still names that generation.
1016    fn emit_superseded_exited(
1017        &self,
1018        module_id: &str,
1019        spawn_generation: u64,
1020        pid: u32,
1021        exit_code: Option<i32>,
1022        exit_signal: Option<i32>,
1023    ) {
1024        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1025        if state
1026            .live
1027            .get(module_id)
1028            .is_some_and(|live| live.spawn_generation == spawn_generation)
1029        {
1030            state.live.remove(module_id);
1031        }
1032        Self::emit_locked(
1033            &mut state,
1034            SpawnEventKind::Exited,
1035            module_id.to_string(),
1036            spawn_generation,
1037            pid,
1038            exit_code,
1039            exit_signal,
1040        );
1041    }
1042
1043    #[allow(clippy::too_many_arguments)]
1044    fn emit_locked(
1045        state: &mut SpawnEventState,
1046        kind: SpawnEventKind,
1047        module_id: String,
1048        spawn_generation: u64,
1049        pid: u32,
1050        exit_code: Option<i32>,
1051        exit_signal: Option<i32>,
1052    ) {
1053        state.seq = state
1054            .seq
1055            .checked_add(1)
1056            .expect("spawn event sequence exhausted");
1057        let event = SpawnEvent {
1058            cursor: Self::cursor(state),
1059            kind,
1060            module_id,
1061            spawn_generation,
1062            pid,
1063            exit_code,
1064            exit_signal,
1065        };
1066        state.events.push_back(event.clone());
1067        while state.events.len() > state.capacity {
1068            state.events.pop_front();
1069        }
1070        let body = match serde_json::to_vec(&event) {
1071            Ok(body) => body,
1072            Err(error) => {
1073                error!(%error, "failed to serialize supervisor spawn event");
1074                return;
1075            }
1076        };
1077        state.subscribers.retain(|(connection_id, corr), subscriber| {
1078            let frame = Frame::build_with_version(
1079                subscriber.version,
1080                FrameType::StreamData,
1081                control_flags(),
1082                0,
1083                0,
1084                *corr,
1085                body.clone(),
1086            );
1087            match frame {
1088                Ok(frame) => {
1089                    if subscriber.frames.try_send(frame).is_ok() {
1090                        true
1091                    } else {
1092                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1093                        if let Some(lagged) = subscriber.lagged.take() {
1094                            let _ = lagged.send(event.cursor.clone());
1095                        }
1096                        false
1097                    }
1098                }
1099                Err(error) => {
1100                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1101                    false
1102                }
1103            }
1104        });
1105    }
1106
1107    fn subscribe(
1108        &self,
1109        connection_id: ConnectionId,
1110        corr: u64,
1111        version: u8,
1112        since: Option<SpawnCursor>,
1113        sink: FrameSink,
1114    ) -> Result<(), SpawnSubscribeRefusal> {
1115        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1116        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1117        {
1118            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1119            let replay = if let Some(since) = since {
1120                if since.daemon_incarnation != state.daemon_incarnation {
1121                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1122                        current: state.daemon_incarnation.clone(),
1123                    });
1124                }
1125                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1126                    if since.seq < oldest.seq.saturating_sub(1) {
1127                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1128                    }
1129                }
1130                state
1131                    .events
1132                    .iter()
1133                    .filter(|event| event.cursor.seq > since.seq)
1134                    .cloned()
1135                    .collect::<Vec<_>>()
1136            } else {
1137                Vec::new()
1138            };
1139            for event in replay {
1140                let body = serde_json::to_vec(&event)
1141                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1142                let frame = Frame::build_with_version(
1143                    version,
1144                    FrameType::StreamData,
1145                    control_flags(),
1146                    0,
1147                    0,
1148                    corr,
1149                    body,
1150                )
1151                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1152                frames
1153                    .try_send(frame)
1154                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1155            }
1156            state.subscribers.insert(
1157                (connection_id, corr),
1158                SpawnSubscriber {
1159                    version,
1160                    frames,
1161                    lagged: Some(lagged),
1162                },
1163            );
1164        }
1165        // The lagged terminal is sent here, by the forwarder, rather than by
1166        // the emitter: at the moment of the drop the subscriber's own channel
1167        // is full, and writing to the connection sink directly from the emitter
1168        // would put the Error AHEAD of the events still queued in that channel
1169        // (and the emitter holds the feed lock, so it cannot await the sink).
1170        // Dropping the subscriber drops the only sender, so `recv` drains every
1171        // queued event and then returns `None`; only then is the Error sent, so
1172        // the client sees each event it can keep, then the reason it was cut.
1173        // Cancel and connection removal drop the oneshot unsent, so they end
1174        // the stream with no Error.
1175        tokio::spawn(async move {
1176            while let Some(frame) = receiver.recv().await {
1177                if sink.send(frame).await.is_err() {
1178                    return;
1179                }
1180            }
1181            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1182                return;
1183            };
1184            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1185                Ok(frame) => {
1186                    let _ = sink.send(frame).await;
1187                }
1188                Err(error) => {
1189                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1190                }
1191            }
1192        });
1193        Ok(())
1194    }
1195
1196    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1197        let Some(subscriber) = self
1198            .0
1199            .lock()
1200            .unwrap_or_else(|p| p.into_inner())
1201            .subscribers
1202            .remove(&(connection_id, corr))
1203        else {
1204            return false;
1205        };
1206        if let Ok(frame) = Frame::build_with_version(
1207            subscriber.version,
1208            FrameType::StreamEnd,
1209            control_flags(),
1210            0,
1211            0,
1212            corr,
1213            Vec::new(),
1214        ) {
1215            tokio::spawn(async move {
1216                let _ = subscriber.frames.send(frame).await;
1217            });
1218        }
1219        true
1220    }
1221
1222    fn remove_connection(&self, connection_id: ConnectionId) {
1223        self.0
1224            .lock()
1225            .unwrap_or_else(|p| p.into_inner())
1226            .subscribers
1227            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1228    }
1229
1230    #[cfg(any(test, feature = "test-support"))]
1231    fn set_capacity(&self, capacity: usize) {
1232        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1233    }
1234
1235    #[cfg(any(test, feature = "test-support"))]
1236    fn subscriber_count(&self) -> usize {
1237        self.0
1238            .lock()
1239            .unwrap_or_else(|p| p.into_inner())
1240            .subscribers
1241            .len()
1242    }
1243}
1244
1245/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1246/// The terminal Error a lagged spawn subscriber receives after its queued events.
1247fn spawn_subscriber_lagged_frame(
1248    version: u8,
1249    corr: u64,
1250    first_undelivered: SpawnCursor,
1251) -> Result<Frame, String> {
1252    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1253        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1254        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1255            .to_string(),
1256        detail: Some(serde_json::json!({
1257            "first_undelivered_cursor": first_undelivered
1258        })),
1259    })
1260    .map_err(|error| error.to_string())?;
1261    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1262        .map_err(|error| error.to_string())
1263}
1264
1265pub trait ModuleProcessLiveness: Send + Sync {
1266    fn process_live(&self, module_id: &str) -> Option<bool>;
1267
1268    /// Whether the supervisor is replacing this module's process right now: an
1269    /// operator restart or reload, a health restart, or a crash respawn whose
1270    /// backoff is running. A module in that state is not live, but a consumer
1271    /// refused now should retry shortly rather than treat the target as gone.
1272    /// Stopped, failed, and disabled modules are not replacing.
1273    fn process_replacing(&self, _module_id: &str) -> bool {
1274        false
1275    }
1276}
1277
1278/// Shared process-liveness registry keyed by supervised `module_id`.
1279#[derive(Debug, Clone, Default)]
1280pub struct SupervisorProcessLiveness {
1281    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1282}
1283
1284impl SupervisorProcessLiveness {
1285    pub fn new() -> Self {
1286        Self::default()
1287    }
1288
1289    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1290        let mut snapshots = self
1291            .snapshots
1292            .lock()
1293            .unwrap_or_else(|poisoned| poisoned.into_inner());
1294        snapshots.insert(module_id, snapshot);
1295    }
1296
1297    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1298        let mut snapshots = self
1299            .snapshots
1300            .lock()
1301            .unwrap_or_else(|poisoned| poisoned.into_inner());
1302        let is_current = snapshots
1303            .get(module_id)
1304            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1305            .unwrap_or(false);
1306        if is_current {
1307            snapshots.remove(module_id);
1308        }
1309    }
1310}
1311
1312impl ModuleProcessLiveness for SupervisorProcessLiveness {
1313    fn process_live(&self, module_id: &str) -> Option<bool> {
1314        let snapshot = {
1315            let snapshots = self
1316                .snapshots
1317                .lock()
1318                .unwrap_or_else(|poisoned| poisoned.into_inner());
1319            snapshots.get(module_id).cloned()
1320        }?;
1321        let snapshot = snapshot
1322            .lock()
1323            .unwrap_or_else(|poisoned| poisoned.into_inner());
1324        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1325    }
1326
1327    fn process_replacing(&self, module_id: &str) -> bool {
1328        let Some(snapshot) = self
1329            .snapshots
1330            .lock()
1331            .unwrap_or_else(|poisoned| poisoned.into_inner())
1332            .get(module_id)
1333            .cloned()
1334        else {
1335            return false;
1336        };
1337        let snapshot = snapshot
1338            .lock()
1339            .unwrap_or_else(|poisoned| poisoned.into_inner());
1340        snapshot.enabled
1341            && match snapshot.state {
1342                ModuleState::Restarting => true,
1343                ModuleState::Draining => snapshot.draining_to_replace,
1344                ModuleState::Starting
1345                | ModuleState::Running
1346                | ModuleState::Unresponsive
1347                | ModuleState::Stopped
1348                | ModuleState::Failed
1349                | ModuleState::Disabled => false,
1350            }
1351    }
1352}
1353
1354#[derive(Debug, Clone)]
1355struct SupervisorRuntimeConfig {
1356    restart_policy: RestartPolicy,
1357    /// This module's RESOLVED drain budget: per-module config when present,
1358    /// else `default_drain_timeout`.
1359    drain_timeout: Duration,
1360    /// Shared with the status handle so the attested value changes atomically
1361    /// when a rescan updates the running drain policy.
1362    effective_drain_timeout: Arc<Mutex<Duration>>,
1363    /// The supervisor-wide fallback, kept so a configuration update that
1364    /// REMOVES the per-module override can re-resolve to it.
1365    default_drain_timeout: Duration,
1366    health: HealthConfig,
1367    connection_file_path: Option<PathBuf>,
1368    capture_logs_dir: Option<PathBuf>,
1369    forwarding: Option<Arc<ForwardingTable>>,
1370    /// The shared handle, so every spawn path (initial, restart, reload) records the
1371    /// reserved-module launch nonce the HELLO verifier checks against.
1372    supervisor_handle: Option<SupervisorHandle>,
1373    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1374    /// status queries.
1375    ///
1376    /// One ring per module, held across every respawn. The lines explaining an exit
1377    /// are written BEFORE that exit, so a ring recreated per process would be empty
1378    /// exactly when it is asked for.
1379    stderr_ring: Arc<Mutex<StderrRing>>,
1380    terminal_ring: Arc<Mutex<TerminalRing>>,
1381    spawn_events: SpawnEventFeed,
1382    child_roster: ChildRoster,
1383    #[cfg(target_os = "linux")]
1384    cgroup_placement: Option<subc_cgroup::Placement>,
1385    #[cfg(test)]
1386    test_seed_stale_facts_before_enable_spawn: bool,
1387}
1388
1389#[derive(Debug, Clone, PartialEq, Eq)]
1390struct SupervisedConfiguration {
1391    spec: ModuleSpec,
1392    health: HealthConfig,
1393}
1394
1395/// Shared daemon lookup table for supervised module handles.
1396///
1397/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1398/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1399/// launch nonces recorded at spawn are checked by the same daemon instance.
1400#[derive(Debug, Clone, Default)]
1401pub struct SupervisorHandle {
1402    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1403    spawn_events: SpawnEventFeed,
1404    /// The current expected launch nonce for each reserved module_id. Set when the
1405    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1406    /// non-reserved module never has an entry here and is never nonce-checked.
1407    /// Reserved module ids and the nonce that authorizes their next HELLO.
1408    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1409    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1410    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1411    /// had NO entry and admitted anyone: the reservation protected the nonce
1412    /// holder, not the NAME (found live by CKCRED's canary probe registering
1413    /// against a reserved scratch id).
1414    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1415    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1416    ///
1417    /// This is deliberately in-memory only: subc is state-free across daemon
1418    /// restarts, and the tombstone only explains the hours-after-removal window
1419    /// while this executing daemon is still alive. Do not persist it in a store.
1420    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1421    /// The current launch nonce for every supervised spawn. This is separate from
1422    /// reserved_nonces because consumer route.open attestation applies to all spawned
1423    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1424    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1425    /// Reserved namespace prefixes mapped to the supervised owner module whose
1426    /// current spawn nonce authorizes HELLO claims below the prefix.
1427    ///
1428    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1429    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1430    /// accidental collisions and lower-trust processes from squatting protected
1431    /// namespaces.
1432    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1433    /// Blue/green swaps in progress, by module id. An entry exists from just
1434    /// before the candidate process is spawned until the swap has failed, or
1435    /// has cut over and the old process is gone. While it exists, HELLO for the
1436    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1437    /// consumer attestation accepts both processes' nonces.
1438    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1439    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1440    promotion_observer: PromotionObserverSlot,
1441    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1442    /// this daemon-wide ordering, a rescan could retire or update a module while a
1443    /// concurrent reload still held its old handle and launch specification.
1444    operation_lock: Arc<AsyncMutex<()>>,
1445}
1446
1447/// Told when a swap has promoted its candidate to be the module's active
1448/// registration.
1449///
1450/// An ordinary HELLO runs the control plane's registration side effects (the
1451/// capability cache, the deny census, the requirement recompute) as it
1452/// registers. A swap candidate's HELLO does not, because it is not routable;
1453/// promotion is when those must run instead, and promotion happens in the
1454/// supervisor, which has no other way into the control handler.
1455pub(crate) trait SwapPromotionObserver: Send + Sync {
1456    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1457}
1458
1459/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1460/// control handler) owns this handle, so a strong reference back would be a
1461/// cycle that keeps both alive.
1462#[derive(Clone, Default)]
1463struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1464
1465impl fmt::Debug for PromotionObserverSlot {
1466    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1467        f.write_str("PromotionObserverSlot")
1468    }
1469}
1470
1471/// The nonces of one open swap.
1472#[derive(Debug, Clone)]
1473struct OpenSwap {
1474    /// The launch nonce minted for the candidate process. It is the swap
1475    /// token: the only thing that admits a HELLO into the candidate slot.
1476    candidate_nonce: String,
1477    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1478    /// here because cutover moves the module's recorded spawn nonce to the
1479    /// candidate while the incumbent is still draining and its consumers are
1480    /// still attesting with this one.
1481    incumbent_nonce: Option<String>,
1482    /// Set once a HELLO has been admitted with the swap token, so the token
1483    /// admits one registration and cannot be replayed after cutover empties
1484    /// the candidate slot.
1485    candidate_admitted: bool,
1486}
1487
1488/// What the swap gate says about a HELLO. See
1489/// [`SupervisorHandle::swap_hello_admission`].
1490#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1491pub(crate) enum SwapHelloAdmission {
1492    /// No swap is open for the id (or the HELLO carries the incumbent's own
1493    /// nonce); the ordinary gates decide.
1494    NotSwapping,
1495    /// The HELLO carries the swap token: register it into the candidate slot.
1496    Candidate,
1497    /// A swap is open and the HELLO carries a nonce the supervisor did not
1498    /// mint for this id, no nonce, or a token already used.
1499    Refused,
1500}
1501
1502#[derive(Debug, Clone, PartialEq, Eq)]
1503pub(crate) enum ReservedHelloRejection {
1504    Exact {
1505        module_id: String,
1506    },
1507    Prefix {
1508        prefix: String,
1509        owner_module_id: String,
1510    },
1511}
1512
1513impl SupervisorHandle {
1514    pub fn new() -> Self {
1515        Self::default()
1516    }
1517
1518    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1519        self.spawn_events.snapshot()
1520    }
1521
1522    pub(crate) fn subscribe_spawns(
1523        &self,
1524        connection_id: ConnectionId,
1525        corr: u64,
1526        version: u8,
1527        since: Option<SpawnCursor>,
1528        sink: FrameSink,
1529    ) -> Result<(), SpawnSubscribeRefusal> {
1530        self.spawn_events
1531            .subscribe(connection_id, corr, version, since, sink)
1532    }
1533
1534    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1535        self.spawn_events.cancel(connection_id, corr)
1536    }
1537
1538    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1539        self.spawn_events.remove_connection(connection_id);
1540    }
1541
1542    #[cfg(any(test, feature = "test-support"))]
1543    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1544        assert!(capacity > 0, "spawn event capacity must be non-zero");
1545        self.spawn_events.set_capacity(capacity);
1546    }
1547
1548    #[cfg(any(test, feature = "test-support"))]
1549    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1550        self.spawn_events.subscriber_count()
1551    }
1552
1553    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1554    /// a respawn invalidates stale consumer identities.
1555    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1556        self.spawn_nonces
1557            .lock()
1558            .unwrap_or_else(|poisoned| poisoned.into_inner())
1559            .insert(module_id.to_string(), nonce);
1560    }
1561
1562    /// Record the launch nonce expected from the next HELLO for a reserved module,
1563    /// replacing any prior nonce (a respawn invalidates the previous one).
1564    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1565        self.reserved_nonces
1566            .lock()
1567            .unwrap_or_else(|poisoned| poisoned.into_inner())
1568            .insert(module_id.to_string(), Some(nonce));
1569    }
1570
1571    /// Record namespace prefixes owned by a supervised module.
1572    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1573        let mut owners = self
1574            .reserved_prefix_owners
1575            .lock()
1576            .unwrap_or_else(|poisoned| poisoned.into_inner());
1577        owners.retain(|_, owner| owner != owner_module_id);
1578        for prefix in prefixes {
1579            owners.insert(prefix.clone(), owner_module_id.to_string());
1580        }
1581    }
1582
1583    /// The launch nonce most recently minted for a module's spawn, if any.
1584    #[cfg(test)]
1585    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1586        self.spawn_nonces
1587            .lock()
1588            .unwrap_or_else(|poisoned| poisoned.into_inner())
1589            .get(module_id)
1590            .cloned()
1591    }
1592
1593    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1594        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1595        let spawn_nonce = self
1596            .spawn_nonces
1597            .lock()
1598            .unwrap_or_else(|poisoned| poisoned.into_inner())
1599            .get(&spec.module_id)
1600            .cloned();
1601        let mut reserved_nonces = self
1602            .reserved_nonces
1603            .lock()
1604            .unwrap_or_else(|poisoned| poisoned.into_inner());
1605        if spec.reserved {
1606            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1607            // reserved name whose module has never spawned has no legitimate
1608            // holder, and the entry's absence is what used to leave the name
1609            // open to the first claimant.
1610            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1611        }
1612        drop(reserved_nonces);
1613        // A later unreserved declaration must not silently unreserve an id that
1614        // was retained after its reserved configuration was removed. The explicit
1615        // release ceremony is the only operation that retires that gate.
1616        self.removal_tombstones
1617            .lock()
1618            .unwrap_or_else(|poisoned| poisoned.into_inner())
1619            .remove(&spec.module_id);
1620    }
1621
1622    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
1623    /// authorized only by its expected nonce; otherwise a matching reserved prefix
1624    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
1625    /// with no matching prefix are always authorized.
1626    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1627        self.reserved_hello_rejection(module_id, presented)
1628            .is_none()
1629    }
1630
1631    pub(crate) fn reserved_hello_rejection(
1632        &self,
1633        module_id: &str,
1634        presented: Option<&str>,
1635    ) -> Option<ReservedHelloRejection> {
1636        let nonces = self
1637            .reserved_nonces
1638            .lock()
1639            .unwrap_or_else(|poisoned| poisoned.into_inner());
1640        if let Some(expected) = nonces.get(module_id) {
1641            // `None` = reserved with no legitimate holder: refuse every
1642            // presentation, because no process can hold a nonce that was never
1643            // minted. Only a real minted nonce admits, in constant time.
1644            let authorized = match expected {
1645                Some(expected) => {
1646                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1647                }
1648                None => false,
1649            };
1650            if authorized {
1651                return None;
1652            }
1653            return Some(ReservedHelloRejection::Exact {
1654                module_id: module_id.to_string(),
1655            });
1656        }
1657        drop(nonces);
1658
1659        let matched_prefix = self
1660            .reserved_prefix_owners
1661            .lock()
1662            .unwrap_or_else(|poisoned| poisoned.into_inner())
1663            .iter()
1664            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
1665            .max_by_key(|(prefix, _)| prefix.len())
1666            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
1667        let (prefix, owner_module_id) = matched_prefix?;
1668
1669        let authorized = presented.is_some_and(|presented| {
1670            self.spawn_nonces
1671                .lock()
1672                .unwrap_or_else(|poisoned| poisoned.into_inner())
1673                .get(&owner_module_id)
1674                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
1675                // While the owner is being swapped, children started by
1676                // either of its two processes hold that process's nonce.
1677                || self.swap_nonce_matches(&owner_module_id, presented)
1678        });
1679        if authorized {
1680            None
1681        } else {
1682            Some(ReservedHelloRejection::Prefix {
1683                prefix,
1684                owner_module_id,
1685            })
1686        }
1687    }
1688
1689    /// Whether a consumer connection proved it came from a daemon-spawned module.
1690    ///
1691    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
1692    /// accepted only for module ids the supervisor has spawned.
1693    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
1694        if presented.is_empty() {
1695            return false;
1696        }
1697        let nonces = self
1698            .spawn_nonces
1699            .lock()
1700            .unwrap_or_else(|poisoned| poisoned.into_inner());
1701        let current = nonces
1702            .get(module_id)
1703            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
1704        drop(nonces);
1705        // During a swap two processes of the module are alive, and a consumer
1706        // started by either one presents that process's nonce. Accepting only
1707        // the recorded one would fail the incumbent's consumers for the whole
1708        // overlap once cutover moves the record to the candidate.
1709        current || self.swap_nonce_matches(module_id, presented)
1710    }
1711
1712    /// Whether `presented` is either nonce of an open swap for `module_id`.
1713    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
1714        let swaps = self
1715            .swaps
1716            .lock()
1717            .unwrap_or_else(|poisoned| poisoned.into_inner());
1718        swaps.get(module_id).is_some_and(|swap| {
1719            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
1720                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
1721                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
1722                })
1723        })
1724    }
1725
1726    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
1727    /// Called before the candidate process exists.
1728    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
1729        let incumbent_nonce = self
1730            .spawn_nonces
1731            .lock()
1732            .unwrap_or_else(|poisoned| poisoned.into_inner())
1733            .get(module_id)
1734            .cloned();
1735        self.swaps
1736            .lock()
1737            .unwrap_or_else(|poisoned| poisoned.into_inner())
1738            .insert(
1739                module_id.to_string(),
1740                OpenSwap {
1741                    candidate_nonce,
1742                    incumbent_nonce,
1743                    candidate_admitted: false,
1744                },
1745            );
1746    }
1747
1748    /// Close the swap for `module_id`, releasing whichever nonce is no longer
1749    /// the module's recorded one.
1750    pub(crate) fn close_swap(&self, module_id: &str) {
1751        self.swaps
1752            .lock()
1753            .unwrap_or_else(|poisoned| poisoned.into_inner())
1754            .remove(module_id);
1755    }
1756
1757    /// Install the observer told about swap promotions, replacing any earlier
1758    /// one.
1759    pub(crate) fn set_swap_promotion_observer(
1760        &self,
1761        observer: std::sync::Weak<dyn SwapPromotionObserver>,
1762    ) {
1763        *self
1764            .promotion_observer
1765            .0
1766            .lock()
1767            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
1768    }
1769
1770    /// Tell the installed observer, if it is still alive, that a swap promoted
1771    /// `registration`.
1772    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
1773        let observer = self
1774            .promotion_observer
1775            .0
1776            .lock()
1777            .unwrap_or_else(|poisoned| poisoned.into_inner())
1778            .as_ref()
1779            .and_then(std::sync::Weak::upgrade);
1780        if let Some(observer) = observer {
1781            observer.swap_promoted(registration);
1782        }
1783    }
1784
1785    /// Whether a swap is open for `module_id`.
1786    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
1787        self.swaps
1788            .lock()
1789            .unwrap_or_else(|poisoned| poisoned.into_inner())
1790            .contains_key(module_id)
1791    }
1792
1793    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
1794    /// respawn would, once cutover has made the candidate the module's process.
1795    /// The swap stays open so the incumbent's nonce keeps attesting until the
1796    /// incumbent has drained and exited.
1797    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
1798        let candidate_nonce = self
1799            .swaps
1800            .lock()
1801            .unwrap_or_else(|poisoned| poisoned.into_inner())
1802            .get(module_id)
1803            .map(|swap| swap.candidate_nonce.clone());
1804        let Some(nonce) = candidate_nonce else {
1805            return;
1806        };
1807        self.set_spawn_nonce(module_id, nonce.clone());
1808        if reserved {
1809            self.set_reserved_nonce(module_id, nonce);
1810        }
1811    }
1812
1813    /// The swap gate for a HELLO claiming `module_id`.
1814    ///
1815    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
1816    /// presents the candidate nonce, which the reserved gate (holding the
1817    /// incumbent's nonce) would refuse as `reserved_module` before swap
1818    /// admission was ever reached. And it applies to unreserved ids too: for an
1819    /// unreserved id the only thing that ever stopped a second process claiming
1820    /// a live id was the `duplicate_module_id` refusal, which is exactly the
1821    /// refusal a swap lifts for its candidate.
1822    ///
1823    /// The incumbent's own nonce falls through to the ordinary gates, which
1824    /// treat it as they always have (a live incumbent is refused as a
1825    /// duplicate). Anything else while a swap is open is refused, including an
1826    /// absent nonce.
1827    pub(crate) fn swap_hello_admission(
1828        &self,
1829        module_id: &str,
1830        presented: Option<&str>,
1831    ) -> SwapHelloAdmission {
1832        let swaps = self
1833            .swaps
1834            .lock()
1835            .unwrap_or_else(|poisoned| poisoned.into_inner());
1836        let Some(swap) = swaps.get(module_id) else {
1837            return SwapHelloAdmission::NotSwapping;
1838        };
1839        let Some(presented) = presented else {
1840            return SwapHelloAdmission::Refused;
1841        };
1842        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
1843            return if swap.candidate_admitted {
1844                SwapHelloAdmission::Refused
1845            } else {
1846                SwapHelloAdmission::Candidate
1847            };
1848        }
1849        if swap
1850            .incumbent_nonce
1851            .as_deref()
1852            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
1853        {
1854            return SwapHelloAdmission::NotSwapping;
1855        }
1856        SwapHelloAdmission::Refused
1857    }
1858
1859    /// Record that the swap token has registered a candidate, so it admits no
1860    /// second HELLO.
1861    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
1862        if let Some(swap) = self
1863            .swaps
1864            .lock()
1865            .unwrap_or_else(|poisoned| poisoned.into_inner())
1866            .get_mut(module_id)
1867        {
1868            swap.candidate_admitted = true;
1869        }
1870    }
1871
1872    /// Test/support lookup for the current launch nonce of a supervised spawn.
1873    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1874        self.spawn_nonces
1875            .lock()
1876            .unwrap_or_else(|poisoned| poisoned.into_inner())
1877            .get(module_id)
1878            .cloned()
1879    }
1880
1881    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
1882    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1883        self.reserved_nonces
1884            .lock()
1885            .unwrap_or_else(|poisoned| poisoned.into_inner())
1886            .get(module_id)
1887            .cloned()
1888            .flatten()
1889    }
1890
1891    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
1892        let mut modules = self
1893            .modules
1894            .lock()
1895            .unwrap_or_else(|poisoned| poisoned.into_inner());
1896        modules.insert(module.module_id().to_string(), module)
1897    }
1898
1899    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
1900        let modules = self
1901            .modules
1902            .lock()
1903            .unwrap_or_else(|poisoned| poisoned.into_inner());
1904        modules.get(module_id).cloned()
1905    }
1906
1907    pub(crate) fn record_late_health_answer(
1908        &self,
1909        module_id: &str,
1910        latency_ms: u64,
1911    ) -> Result<bool, SuperviseError> {
1912        let Some(module) = self.get(module_id) else {
1913            return Ok(false);
1914        };
1915        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
1916            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
1917            state.health.last_late_answer_latency_ms = Some(latency_ms);
1918            // A late answer is an answer: the module served the probe, just past
1919            // the deadline. Leaving the miss streak in place while logging
1920            // "proves the module is alive" is how a CPU-starved module that
1921            // answers every probe a few seconds late still marches to the
1922            // threshold and gets killed — the exact kill class `NoAnswer` is
1923            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
1924            // is degradation, and degradation reports; it does not restart.
1925            state.health.consecutive_failures = 0;
1926        })?;
1927        Ok(true)
1928    }
1929
1930    /// Arm the one-shot marker for the module process that this caller
1931    /// deliberately initiated severance against. Generic connection teardown
1932    /// must not call this:
1933    /// a surviving process would otherwise retain an exemption for a later
1934    /// genuine crash.
1935    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
1936        let Some(module) = self.get(module_id) else {
1937            return Ok(false);
1938        };
1939        let status = module.status()?;
1940        let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
1941            return Ok(false);
1942        };
1943        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
1944    }
1945
1946    pub fn list(&self) -> Vec<SupervisedModule> {
1947        let modules = self
1948            .modules
1949            .lock()
1950            .unwrap_or_else(|poisoned| poisoned.into_inner());
1951        let mut modules = modules.values().cloned().collect::<Vec<_>>();
1952        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
1953        modules
1954    }
1955
1956    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
1957        self.spawn_nonces
1958            .lock()
1959            .unwrap_or_else(|poisoned| poisoned.into_inner())
1960            .remove(module_id);
1961        self.close_swap(module_id);
1962        let mut reserved_nonces = self
1963            .reserved_nonces
1964            .lock()
1965            .unwrap_or_else(|poisoned| poisoned.into_inner());
1966        if reserved_nonces.contains_key(module_id) {
1967            // The old nonce must die with the removed process, but the exact-id
1968            // gate remains until an operator explicitly releases it.
1969            reserved_nonces.insert(module_id.to_string(), None);
1970        }
1971        drop(reserved_nonces);
1972        self.reserved_prefix_owners
1973            .lock()
1974            .unwrap_or_else(|poisoned| poisoned.into_inner())
1975            .retain(|_, owner| owner != module_id);
1976        self.modules
1977            .lock()
1978            .unwrap_or_else(|poisoned| poisoned.into_inner())
1979            .remove(module_id)
1980    }
1981
1982    /// Remember a module removed by a non-preview rescan so route.open can
1983    /// distinguish that intentional removal from an unknown id.
1984    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
1985        self.removal_tombstones
1986            .lock()
1987            .unwrap_or_else(|poisoned| poisoned.into_inner())
1988            .insert(module_id.to_string(), unix_ms_now());
1989    }
1990
1991    /// Return how long ago a rescan removed this module in milliseconds.
1992    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
1993        self.removal_tombstones
1994            .lock()
1995            .unwrap_or_else(|poisoned| poisoned.into_inner())
1996            .get(module_id)
1997            .copied()
1998            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
1999    }
2000
2001    /// Retire a reserved-id gate only after its module has left supervision.
2002    ///
2003    /// A retained gate has no live nonce (`None`), so releasing any other entry
2004    /// would weaken a currently configured or otherwise active reservation.
2005    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2006        if self.get(module_id).is_some() {
2007            return false;
2008        }
2009        let mut reserved_nonces = self
2010            .reserved_nonces
2011            .lock()
2012            .unwrap_or_else(|poisoned| poisoned.into_inner());
2013        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2014            return false;
2015        }
2016        reserved_nonces.remove(module_id);
2017        true
2018    }
2019
2020    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2021        Arc::clone(&self.operation_lock)
2022    }
2023}
2024
2025/// Process supervisor for subc-owned singleton modules.
2026#[derive(Debug, Clone)]
2027pub struct Supervisor {
2028    registry: Arc<Registry>,
2029    restart_policy: RestartPolicy,
2030    drain_timeout: Duration,
2031    connection_file_path: Option<PathBuf>,
2032    capture_logs_dir: Option<PathBuf>,
2033    forwarding: Option<Arc<ForwardingTable>>,
2034    process_liveness: Arc<SupervisorProcessLiveness>,
2035    supervisor_handle: Option<SupervisorHandle>,
2036    health: HealthConfig,
2037    daemon_start_clock: crate::clock::StartClock,
2038    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2039    spawn_events: SpawnEventFeed,
2040    provenance_probe: ExecutableIdentityProbe,
2041    /// Every process spawned through this supervisor (and its clones) and not
2042    /// yet reaped, so daemon shutdown can end them.
2043    child_roster: ChildRoster,
2044    #[cfg(target_os = "linux")]
2045    cgroup_placement: Option<subc_cgroup::Placement>,
2046}
2047
2048impl Supervisor {
2049    /// The first step of an announced daemon shutdown, before the notice and
2050    /// before any connection is closed.
2051    ///
2052    /// Sets the daemon-shutdown flag first: from here on no module is
2053    /// respawned (crash restart, operator restart, or swap), and every child
2054    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2055    /// the module exits on the EOF this shutdown gives it or is signalled by a
2056    /// service manager that kills the whole cgroup. Then writes the journal's
2057    /// shutdown marker, which records the instant and closes this daemon
2058    /// incarnation's stretch of the journal.
2059    #[cfg(unix)]
2060    pub(crate) fn begin_daemon_shutdown(&self) {
2061        self.child_roster.close();
2062        if let Some(journal) = &self.terminal_journal {
2063            journal.stamp_shutdown();
2064        }
2065    }
2066
2067    /// Announce a cut while established connections can still carry replies.
2068    /// These budgets promise notice and a bounded wait, not child completion;
2069    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2070    #[cfg(unix)]
2071    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2072        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2073        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2074        let Some(forwarding) = &self.forwarding else {
2075            return Ok(());
2076        };
2077        let module_ids = forwarding
2078            .begin_daemon_drain()
2079            .map_err(SuperviseError::Forwarding)?;
2080        let deadline_ms =
2081            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2082        let mut notices = tokio::task::JoinSet::new();
2083        let mut drains = Vec::new();
2084        for module_id in module_ids {
2085            let Some(target) = forwarding
2086                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2087                .map_err(SuperviseError::Forwarding)?
2088            else {
2089                continue;
2090            };
2091            let routes = forwarding
2092                .endpoint_routes(target.endpoint)
2093                .map_err(SuperviseError::Forwarding)?;
2094            // Restart allows deployed consumers to reopen after the new daemon
2095            // appears. The wire reason stays `restart`; what tells a daemon cut
2096            // apart from a module restart afterwards is the terminal record
2097            // itself, whose disposition is `daemon_shutdown` for every exit
2098            // observed once `begin_daemon_shutdown` has run.
2099            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2100                reason: RouteCloseReason::Restart,
2101                deadline_ms,
2102            })
2103            .expect("module draining serializes");
2104            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2105            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2106            for route in routes {
2107                let client = route.goodbye_target;
2108                if let Some((_, channels)) = clients
2109                    .iter_mut()
2110                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2111                {
2112                    channels.push(client.channel);
2113                } else {
2114                    let channel = client.channel;
2115                    clients.push((client, vec![channel]));
2116                }
2117            }
2118            for (client, mut channels) in clients {
2119                channels.sort_unstable();
2120                channels.dedup();
2121                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2122                    module_id: module_id.clone(),
2123                    channels,
2124                    reason: RouteCloseReason::Restart,
2125                })
2126                .expect("route closing serializes");
2127                recipients.push((client.sink, client.negotiated_ver, closing));
2128            }
2129            for (sink, version, body) in recipients {
2130                notices.spawn(async move {
2131                    let frame = Frame::build_with_version(
2132                        version,
2133                        FrameType::Push,
2134                        control_flags(),
2135                        0,
2136                        0,
2137                        0,
2138                        body,
2139                    )
2140                    .expect("bounded lifecycle notice frame builds");
2141                    sink.send_flushed(frame).await
2142                });
2143            }
2144            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2145            drains.push((module_id, target.endpoint, gauges));
2146        }
2147        // A quiet forwarding table is not proof that queued notices reached the
2148        // socket. Wait for writer flush acknowledgements before testing quiescence.
2149        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2150        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2151            if !matches!(result, Ok(Ok(()))) {
2152                warn!(?result, "daemon shutdown notice delivery failed");
2153            }
2154        }
2155        notices.abort_all();
2156        let deadline = Instant::now() + DRAIN_BUDGET;
2157        let mut waits = tokio::task::JoinSet::new();
2158        for (module_id, endpoint, gauges) in drains {
2159            let forwarding = Arc::clone(forwarding);
2160            let mut runtime = self.runtime_config();
2161            runtime.health.cadence = Duration::from_millis(100);
2162            waits.spawn(async move {
2163                wait_for_forwarding_quiescence(
2164                    &forwarding,
2165                    &module_id,
2166                    &runtime,
2167                    endpoint,
2168                    deadline,
2169                    &gauges,
2170                    DrainScope::Active,
2171                )
2172                .await
2173            });
2174        }
2175        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2176            if !matches!(result, Ok(Ok(true))) {
2177                warn!(?result, "daemon shutdown drain did not reach quiescence");
2178            }
2179        }
2180        Ok(())
2181    }
2182
2183    /// The last step of an announced daemon shutdown, after the notice and the
2184    /// drain: send every registered module a module GOODBYE, the same planned
2185    /// stop signal `ck module stop` gives, then close every connection so each
2186    /// subc module sees EOF and starts its own teardown, then end every
2187    /// supervised child that has not exited
2188    /// by its own deadline (its drain budget, capped). Modules lead their own
2189    /// process groups, so a
2190    /// service manager's group kill no longer reaches them; without this a
2191    /// child that does not stop on EOF (every `protocol: "none"` child, which
2192    /// has no connection) would outlive the daemon. Every wait is bounded (see
2193    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2194    #[cfg(unix)]
2195    pub(crate) async fn end_children_for_daemon_shutdown(
2196        &self,
2197        already_escalated: bool,
2198        escalate: impl std::future::Future<Output = ()>,
2199    ) {
2200        tokio::pin!(escalate);
2201        let mut escalated = already_escalated;
2202        if let Some(forwarding) = &self.forwarding {
2203            let reason = CloseReason::new(
2204                "daemon_shutdown",
2205                "the daemon is exiting after its shutdown notice and drain",
2206            );
2207            if escalated {
2208                // The operator asked to stop waiting: queue the GOODBYEs but
2209                // do not wait for them to be written.
2210                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2211            } else {
2212                tokio::select! {
2213                    biased;
2214                    _ = escalate.as_mut() => {
2215                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2216                        escalated = true;
2217                    }
2218                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2219                }
2220            }
2221            let closed = forwarding.close_all_connections(&reason);
2222            debug!(closed, "closed established connections for daemon shutdown");
2223        }
2224        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2225        // already completed and must not be polled again; the child shutdown
2226        // wait is told it is escalated and gets a future that never fires.
2227        let escalated_here = escalated && !already_escalated;
2228        let remaining_escalate = async move {
2229            if escalated_here {
2230                std::future::pending::<()>().await;
2231            } else {
2232                escalate.await;
2233            }
2234        };
2235        crate::child_roster::end_children_for_daemon_shutdown(
2236            &self.child_roster,
2237            escalated,
2238            remaining_escalate,
2239        )
2240        .await;
2241    }
2242
2243    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2244        Self {
2245            registry,
2246            restart_policy,
2247            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2248            connection_file_path: None,
2249            capture_logs_dir: None,
2250            forwarding: None,
2251            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2252            supervisor_handle: None,
2253            health: HealthConfig::default(),
2254            daemon_start_clock: crate::clock::StartClock::capture(),
2255            terminal_journal: None,
2256            spawn_events: SpawnEventFeed::default(),
2257            provenance_probe: ExecutableIdentityProbe::default(),
2258            child_roster: ChildRoster::default(),
2259            #[cfg(target_os = "linux")]
2260            cgroup_placement: None,
2261        }
2262    }
2263
2264    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2265        self.drain_timeout = drain_timeout;
2266        self
2267    }
2268
2269    pub fn with_process_liveness(
2270        mut self,
2271        process_liveness: Arc<SupervisorProcessLiveness>,
2272    ) -> Self {
2273        self.process_liveness = process_liveness;
2274        self
2275    }
2276
2277    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2278        self.connection_file_path = Some(connection_file_path.into());
2279        self
2280    }
2281
2282    /// Enables daemon-owned capture files for supervised stdout and stderr.
2283    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2284        self.capture_logs_dir = Some(logs_dir.into());
2285        self
2286    }
2287
2288    /// Names this daemon lifetime in spawn events, independently of whether a
2289    /// terminal journal is configured.
2290    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2291        // A millisecond start stamp can repeat after clock rollback or a rapid
2292        // restart. Use the connection file's random daemon_id instead: it already
2293        // identifies this daemon lifetime independently of the wall clock.
2294        self.spawn_events.configure_incarnation(daemon_incarnation);
2295        self
2296    }
2297
2298    /// Enables best-effort history shared by every supervised module. Without
2299    /// it, terminal history is kept only in each module's in-memory ring.
2300    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2301        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2302        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2303            path,
2304            daemon_incarnation,
2305        )));
2306        this
2307    }
2308
2309    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2310        self.forwarding = Some(forwarding);
2311        self
2312    }
2313
2314    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2315        self.spawn_events = supervisor_handle.spawn_events.clone();
2316        self.supervisor_handle = Some(supervisor_handle);
2317        self
2318    }
2319
2320    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2321        self.health = health;
2322        self
2323    }
2324
2325    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2326    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2327    /// record is kept.
2328    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2329        self.child_roster.record_to(path.into());
2330        self
2331    }
2332
2333    #[cfg(target_os = "linux")]
2334    pub fn with_cgroup_placement(
2335        mut self,
2336        cgroup_placement: Option<subc_cgroup::Placement>,
2337    ) -> Self {
2338        self.cgroup_placement = cgroup_placement;
2339        self
2340    }
2341
2342    /// Spawn `spec.program` and start monitoring it.
2343    ///
2344    /// The child is expected to parse `--subc <connection-file-path>`, read the
2345    /// TCP+key connection file, authenticate to the already-running listener, and
2346    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2347    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2348        validate_spec(&spec)?;
2349
2350        let runtime = self.runtime_config();
2351        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2352        let child = spawn_child(
2353            &spec,
2354            runtime.connection_file_path.as_deref(),
2355            self.supervisor_handle.as_ref(),
2356            &runtime.stderr_ring,
2357            runtime.capture_logs_dir.as_deref(),
2358            &runtime.child_roster,
2359            #[cfg(target_os = "linux")]
2360            runtime.cgroup_placement.as_ref(),
2361        )?;
2362        set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2363        self.process_liveness
2364            .track(spec.module_id.clone(), Arc::clone(&snapshot));
2365
2366        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2367    }
2368
2369    /// Start supervising a module declared in daemon configuration.
2370    ///
2371    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2372    /// failures in the supervisor handle so operator-facing `supervisor.list`
2373    /// reflects every configured module while daemon startup continues.
2374    pub fn supervise_configured(
2375        &self,
2376        spec: ModuleSpec,
2377        enabled: bool,
2378    ) -> Result<SupervisedModule, SuperviseError> {
2379        validate_spec(&spec)?;
2380
2381        let runtime = self.runtime_config();
2382        if !enabled {
2383            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2384            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2385        }
2386
2387        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2388        match spawn_child(
2389            &spec,
2390            runtime.connection_file_path.as_deref(),
2391            self.supervisor_handle.as_ref(),
2392            &runtime.stderr_ring,
2393            runtime.capture_logs_dir.as_deref(),
2394            &runtime.child_roster,
2395            #[cfg(target_os = "linux")]
2396            runtime.cgroup_placement.as_ref(),
2397        ) {
2398            Ok(child) => {
2399                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2400                self.process_liveness
2401                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2402                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2403            }
2404            Err(err) => {
2405                error!(
2406                    module_id = %spec.module_id,
2407                    program = %spec.program.display(),
2408                    error = %err,
2409                    "configured module failed to spawn; marking failed and continuing"
2410                );
2411                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2412                Ok(self.supervised_module(spec, runtime, snapshot, None))
2413            }
2414        }
2415    }
2416
2417    /// Supervise a configured module with its own health, drain, and crash
2418    /// budget. The restart policy is per-module because the config file is:
2419    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2420    /// module that is expensive to restart should not be forced onto the same
2421    /// budget as one that is cheap.
2422    pub fn supervise_configured_with_health(
2423        &self,
2424        spec: ModuleSpec,
2425        enabled: bool,
2426        health: HealthConfig,
2427        drain_timeout_ms: Option<u64>,
2428        restart_policy: RestartPolicy,
2429    ) -> Result<SupervisedModule, SuperviseError> {
2430        validate_spec(&spec)?;
2431
2432        let mut runtime = self.runtime_config();
2433        runtime.health = health;
2434        runtime.restart_policy = restart_policy;
2435        if let Some(ms) = drain_timeout_ms {
2436            runtime.drain_timeout = Duration::from_millis(ms);
2437            *runtime
2438                .effective_drain_timeout
2439                .lock()
2440                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2441        }
2442        if !enabled {
2443            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2444            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2445        }
2446
2447        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2448        match spawn_child(
2449            &spec,
2450            runtime.connection_file_path.as_deref(),
2451            self.supervisor_handle.as_ref(),
2452            &runtime.stderr_ring,
2453            runtime.capture_logs_dir.as_deref(),
2454            &runtime.child_roster,
2455            #[cfg(target_os = "linux")]
2456            runtime.cgroup_placement.as_ref(),
2457        ) {
2458            Ok(child) => {
2459                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2460                self.process_liveness
2461                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2462                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2463            }
2464            Err(err) => {
2465                if health.critical {
2466                    error!(
2467                        module_id = %spec.module_id,
2468                        program = %spec.program.display(),
2469                        error = %err,
2470                        "critical configured module failed to spawn; marking failed and alerting"
2471                    );
2472                } else {
2473                    error!(
2474                        module_id = %spec.module_id,
2475                        program = %spec.program.display(),
2476                        error = %err,
2477                        "configured module failed to spawn; marking failed and continuing"
2478                    );
2479                }
2480                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2481                Ok(self.supervised_module(spec, runtime, snapshot, None))
2482            }
2483        }
2484    }
2485
2486    fn runtime_config(&self) -> SupervisorRuntimeConfig {
2487        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2488        SupervisorRuntimeConfig {
2489            restart_policy: self.restart_policy,
2490            drain_timeout: self.drain_timeout,
2491            // Shared with this module's roster copy: daemon shutdown waits on
2492            // each child for the module's own drain budget, as resolved now.
2493            child_roster: self
2494                .child_roster
2495                .for_module(Arc::clone(&effective_drain_timeout)),
2496            effective_drain_timeout,
2497            default_drain_timeout: self.drain_timeout,
2498            health: self.health,
2499            connection_file_path: self.connection_file_path.clone(),
2500            capture_logs_dir: self.capture_logs_dir.clone(),
2501            forwarding: self.forwarding.clone(),
2502            supervisor_handle: self.supervisor_handle.clone(),
2503            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
2504            terminal_ring: Arc::new(Mutex::new(
2505                TerminalRing::new(
2506                    TerminalRingConfig::default(),
2507                    self.daemon_start_clock.started_at_ms(),
2508                )
2509                .with_start_clock(self.daemon_start_clock)
2510                .with_journal(self.terminal_journal.clone())
2511                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
2512            )),
2513            spawn_events: self.spawn_events.clone(),
2514            #[cfg(target_os = "linux")]
2515            cgroup_placement: self.cgroup_placement.clone(),
2516            #[cfg(test)]
2517            test_seed_stale_facts_before_enable_spawn: false,
2518        }
2519    }
2520
2521    fn supervised_module(
2522        &self,
2523        spec: ModuleSpec,
2524        runtime: SupervisorRuntimeConfig,
2525        snapshot: SharedSnapshot,
2526        child: Option<SupervisedChild>,
2527    ) -> SupervisedModule {
2528        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
2529            spec: spec.clone(),
2530            health: runtime.health,
2531        }));
2532        let stderr_ring = Arc::clone(&runtime.stderr_ring);
2533        let terminal_ring = Arc::clone(&runtime.terminal_ring);
2534        // The module's OWN policy, which may be its per-module config rather than
2535        // the supervisor-wide one; status must report the budget the supervise
2536        // loop actually enforces.
2537        let restart_policy = runtime.restart_policy;
2538        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
2539        let (tx, rx) = mpsc::channel(4);
2540        let monitor = tokio::spawn(supervise_loop(
2541            spec.clone(),
2542            runtime,
2543            Arc::clone(&self.registry),
2544            Arc::clone(&self.process_liveness),
2545            Arc::clone(&snapshot),
2546            child,
2547            rx,
2548        ));
2549
2550        let module_id = spec.module_id.clone();
2551        let module = SupervisedModule {
2552            inner: Arc::new(SupervisedModuleInner {
2553                module_id: module_id.clone(),
2554                registry: Arc::clone(&self.registry),
2555                snapshot,
2556                configuration,
2557                stderr_ring,
2558                terminal_ring,
2559                commands: tx,
2560                monitor: Mutex::new(Some(monitor)),
2561                restart_policy,
2562                effective_drain_timeout,
2563                provenance_probe: self.provenance_probe.clone(),
2564            }),
2565        };
2566        if let Some(supervisor_handle) = &self.supervisor_handle {
2567            supervisor_handle.apply_identity_configuration(&spec);
2568            supervisor_handle.insert(module.clone());
2569        }
2570        module
2571    }
2572}
2573
2574impl Default for Supervisor {
2575    fn default() -> Self {
2576        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
2577    }
2578}
2579
2580/// Handle to one supervised child process.
2581#[derive(Clone)]
2582pub struct SupervisedModule {
2583    inner: Arc<SupervisedModuleInner>,
2584}
2585
2586struct SupervisedModuleInner {
2587    module_id: String,
2588    registry: Arc<Registry>,
2589    snapshot: SharedSnapshot,
2590    configuration: Arc<Mutex<SupervisedConfiguration>>,
2591    stderr_ring: Arc<Mutex<StderrRing>>,
2592    terminal_ring: Arc<Mutex<TerminalRing>>,
2593    commands: mpsc::Sender<SupervisorCommand>,
2594    monitor: Mutex<Option<JoinHandle<()>>>,
2595    /// Copied from the supervisor's runtime config at spawn so `status()` can
2596    /// report the restart budget without reaching back into the supervisor. The
2597    /// policy is fixed for the process's lifetime, so a copy cannot drift.
2598    restart_policy: RestartPolicy,
2599    effective_drain_timeout: Arc<Mutex<Duration>>,
2600    provenance_probe: ExecutableIdentityProbe,
2601}
2602
2603impl fmt::Debug for SupervisedModule {
2604    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2605        f.debug_struct("SupervisedModule")
2606            .field("module_id", &self.inner.module_id)
2607            .field("status", &self.status())
2608            .finish_non_exhaustive()
2609    }
2610}
2611
2612impl SupervisedModule {
2613    pub fn module_id(&self) -> &str {
2614        &self.inner.module_id
2615    }
2616
2617    /// Test-only: put one probe miss on the streak, the way
2618    /// `handle_health_probe_failure` does, so tests can assert what a later
2619    /// event does to the streak without driving the whole probe loop.
2620    #[cfg(test)]
2621    pub(crate) fn record_health_probe_failure_for_test(
2622        &self,
2623        detail: &str,
2624    ) -> Result<(), SuperviseError> {
2625        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
2626            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
2627            state.health.detail = Some(detail.to_string());
2628        })
2629    }
2630
2631    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
2632        Ok(lock_snapshot(&self.inner.snapshot)?.state)
2633    }
2634
2635    /// The module's retained stderr, newest lines last.
2636    ///
2637    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
2638    /// module, `supervisor.list` renders every module, and putting it in the
2639    /// shared snapshot would make each status read carry a payload almost nobody
2640    /// asked for. Callers that want the text ask for it.
2641    pub fn stderr_tail(
2642        &self,
2643        max_lines: Option<usize>,
2644        max_bytes: Option<usize>,
2645    ) -> StderrTailSnapshot {
2646        self.inner
2647            .stderr_ring
2648            .lock()
2649            .unwrap_or_else(|poisoned| poisoned.into_inner())
2650            .snapshot(max_lines, max_bytes)
2651    }
2652
2653    /// The module's bounded terminal history, oldest retained exit first.
2654    ///
2655    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
2656    /// daemon whose in-memory history was necessarily reset.
2657    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
2658        self.inner
2659            .terminal_ring
2660            .lock()
2661            .unwrap_or_else(|poisoned| poisoned.into_inner())
2662            .snapshot()
2663    }
2664
2665    /// Retained observations from the current ring and all journal generations.
2666    ///
2667    /// Blocking: this reads the journal files. Async callers use
2668    /// [`Self::read_durable_terminal_history`].
2669    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
2670        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
2671    }
2672
2673    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
2674    /// read (up to every retained generation) never occupies a runtime worker.
2675    /// Fails only if the blocking task could not finish (runtime shutdown or a
2676    /// panic in the read).
2677    pub(crate) async fn read_durable_terminal_history(
2678        &self,
2679    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
2680        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
2681        let module_id = self.inner.module_id.clone();
2682        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
2683            .await
2684    }
2685
2686    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
2687        self.status_with_snapshot_lock(&self.inner.snapshot, None)
2688    }
2689
2690    pub(crate) fn record_deliberate_severance(
2691        &self,
2692        identity: ProcessIdentity,
2693    ) -> Result<bool, SuperviseError> {
2694        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2695        if snapshot.pid != Some(identity.pid)
2696            || snapshot.process_start_time != Some(identity.start_time)
2697        {
2698            return Ok(false);
2699        }
2700        snapshot.deliberate_severance = Some(identity);
2701        Ok(true)
2702    }
2703
2704    /// Read status for a channel-0 renderer and report a contended snapshot lock.
2705    ///
2706    /// Internal supervision callers use [`Self::status`] so writer-side machinery
2707    /// does not produce reader-observability logs.
2708    pub(crate) fn status_for_control(
2709        &self,
2710        caller: &'static str,
2711    ) -> Result<ModuleStatus, SuperviseError> {
2712        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
2713    }
2714
2715    fn status_with_snapshot_lock(
2716        &self,
2717        snapshot: &SharedSnapshot,
2718        caller: Option<&'static str>,
2719    ) -> Result<ModuleStatus, SuperviseError> {
2720        let mut guard = match caller {
2721            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
2722            None => lock_snapshot(snapshot)?,
2723        };
2724        // Read the budget through the pruning path so a reader sees the same
2725        // in-window count the restart decision would use, not a stale total.
2726        let restart_count =
2727            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
2728        let snapshot = guard.clone();
2729        drop(guard);
2730        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
2731            SuperviseError::StatePoisoned {
2732                module_id: Some(self.inner.module_id.clone()),
2733            }
2734        })?;
2735        let registration_active = self
2736            .inner
2737            .registry
2738            .get_module(&self.inner.module_id)
2739            .map_err(SuperviseError::Registry)?
2740            .is_some();
2741        let protocol = self.declared_protocol()?;
2742        let running_process =
2743            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
2744        // Registration is the difference between the two protocols and the only
2745        // one: a subc module that has not registered cannot serve a request even
2746        // though its process is up, and a `none` module never registers at all,
2747        // so requiring it there would pin `live` to false for the whole life of
2748        // a perfectly healthy process.
2749        let live = match protocol {
2750            ModuleProtocol::Subc => running_process && registration_active,
2751            ModuleProtocol::None => running_process,
2752        };
2753
2754        Ok(ModuleStatus {
2755            module_id: self.inner.module_id.clone(),
2756            state: snapshot.state,
2757            enabled: snapshot.enabled,
2758            process_alive: snapshot.process_alive,
2759            registration_active,
2760            protocol,
2761            live,
2762            restart_count,
2763            lifetime_restarts: snapshot.lifetime_restarts,
2764            spawn_generation: snapshot.spawn_generation,
2765            max_restarts: self.inner.restart_policy.max_restarts,
2766            restart_window: self.inner.restart_policy.window,
2767            drain_timeout,
2768            restart_backoff: self.inner.restart_policy.backoff,
2769            restart_max_backoff: self.inner.restart_policy.max_backoff,
2770            pid: snapshot.pid,
2771            spawned_at_ms: snapshot.spawned_at_ms,
2772            spawned_from: snapshot.spawned_from,
2773            process_start_time: snapshot.process_start_time,
2774            last_exit: snapshot.last_exit,
2775            health: snapshot.health,
2776        })
2777    }
2778
2779    #[cfg(test)]
2780    pub(crate) fn hold_snapshot_for_test(
2781        &self,
2782        acquired: std::sync::mpsc::Sender<()>,
2783        hold: Duration,
2784    ) -> std::thread::JoinHandle<()> {
2785        let snapshot = Arc::clone(&self.inner.snapshot);
2786        std::thread::spawn(move || {
2787            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
2788            acquired
2789                .send(())
2790                .expect("test receiver waits for snapshot lock");
2791            std::thread::sleep(hold);
2792        })
2793    }
2794
2795    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
2796        let snapshot = match lock_snapshot(&self.inner.snapshot) {
2797            Ok(snapshot) => snapshot.clone(),
2798            Err(_) => {
2799                return subc_control::RunningImageAgreement::Unavailable {
2800                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
2801                };
2802            }
2803        };
2804        self.inner
2805            .provenance_probe
2806            .observe(
2807                snapshot.pid,
2808                snapshot.spawned_from.as_deref(),
2809                snapshot.spawned_file_identity,
2810                snapshot.process_start_time,
2811            )
2812            .await
2813    }
2814
2815    /// Memory and CPU time of the module's current process, read now. Only the
2816    /// process the supervisor spawned is read, not processes it has started.
2817    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
2818        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
2819            Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
2820            Err(_) => {
2821                return subc_control::ChildResourceUsage::Unavailable {
2822                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
2823                }
2824            }
2825        };
2826        crate::child_resources::read(pid, start_time)
2827    }
2828
2829    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
2830        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2831        Ok(match snapshot.state {
2832            ModuleState::Restarting => true,
2833            ModuleState::Failed | ModuleState::Disabled => false,
2834            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
2835        })
2836    }
2837
2838    #[cfg(test)]
2839    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
2840        self.is_warming_with_snapshot_lock(None)
2841    }
2842
2843    pub(crate) fn is_warming_for_control(
2844        &self,
2845        caller: &'static str,
2846    ) -> Result<bool, SuperviseError> {
2847        self.is_warming_with_snapshot_lock(Some(caller))
2848    }
2849
2850    fn is_warming_with_snapshot_lock(
2851        &self,
2852        caller: Option<&'static str>,
2853    ) -> Result<bool, SuperviseError> {
2854        let snapshot = match caller {
2855            Some(caller) => {
2856                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
2857            }
2858            None => lock_snapshot(&self.inner.snapshot)?,
2859        }
2860        .clone();
2861        Ok(matches!(
2862            snapshot.state,
2863            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
2864        ))
2865    }
2866
2867    /// Drain the module and stop monitoring it.
2868    pub async fn drain(&self) -> Result<(), SuperviseError> {
2869        self.stop().await
2870    }
2871
2872    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
2873        match self.state()? {
2874            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2875            ModuleState::Starting
2876            | ModuleState::Running
2877            | ModuleState::Unresponsive
2878            | ModuleState::Restarting
2879            | ModuleState::Draining
2880            | ModuleState::Disabled => {}
2881        }
2882
2883        let (reply_tx, reply_rx) = oneshot::channel();
2884        self.inner
2885            .commands
2886            .send(SupervisorCommand::Retire { reply: reply_tx })
2887            .await
2888            .map_err(|_| SuperviseError::CommandClosed {
2889                module_id: self.inner.module_id.clone(),
2890            })?;
2891        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2892            module_id: self.inner.module_id.clone(),
2893        })?
2894    }
2895
2896    pub async fn stop(&self) -> Result<(), SuperviseError> {
2897        match self.state()? {
2898            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2899            ModuleState::Starting
2900            | ModuleState::Running
2901            | ModuleState::Unresponsive
2902            | ModuleState::Restarting
2903            | ModuleState::Draining
2904            | ModuleState::Disabled => {}
2905        }
2906
2907        let (reply_tx, reply_rx) = oneshot::channel();
2908        self.inner
2909            .commands
2910            .send(SupervisorCommand::Drain { reply: reply_tx })
2911            .await
2912            .map_err(|_| SuperviseError::CommandClosed {
2913                module_id: self.inner.module_id.clone(),
2914            })?;
2915        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2916            module_id: self.inner.module_id.clone(),
2917        })?
2918    }
2919
2920    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
2921        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
2922        let (reply_tx, reply_rx) = oneshot::channel();
2923        self.inner
2924            .commands
2925            .send(SupervisorCommand::Restart {
2926                drain_timeout_ms,
2927                received_at_generation,
2928                queued_at: Instant::now(),
2929                reply: reply_tx,
2930            })
2931            .await
2932            .map_err(|_| SuperviseError::CommandClosed {
2933                module_id: self.inner.module_id.clone(),
2934            })?;
2935        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2936            module_id: self.inner.module_id.clone(),
2937        })?
2938    }
2939
2940    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
2941    /// `supervisor_swap` module. Returns once the swap has cut over (the old
2942    /// process then drains in the background of the supervise loop) or has
2943    /// failed, leaving the old process serving.
2944    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
2945        let (reply_tx, reply_rx) = oneshot::channel();
2946        self.inner
2947            .commands
2948            .send(SupervisorCommand::Swap {
2949                ready_timeout,
2950                reply: reply_tx,
2951            })
2952            .await
2953            .map_err(|_| SuperviseError::CommandClosed {
2954                module_id: self.inner.module_id.clone(),
2955            })?;
2956        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2957            module_id: self.inner.module_id.clone(),
2958        })?
2959    }
2960
2961    pub async fn reload(&self) -> Result<(), SuperviseError> {
2962        let (reply_tx, reply_rx) = oneshot::channel();
2963        self.inner
2964            .commands
2965            .send(SupervisorCommand::Reload { reply: reply_tx })
2966            .await
2967            .map_err(|_| SuperviseError::CommandClosed {
2968                module_id: self.inner.module_id.clone(),
2969            })?;
2970        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2971            module_id: self.inner.module_id.clone(),
2972        })?
2973    }
2974
2975    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
2976        let (reply_tx, reply_rx) = oneshot::channel();
2977        self.inner
2978            .commands
2979            .send(SupervisorCommand::SetEnabled {
2980                enabled,
2981                reply: reply_tx,
2982            })
2983            .await
2984            .map_err(|_| SuperviseError::CommandClosed {
2985                module_id: self.inner.module_id.clone(),
2986            })?;
2987        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2988            module_id: self.inner.module_id.clone(),
2989        })?
2990    }
2991
2992    /// This module's declared protocol, read from the same stored configuration
2993    /// the rescan diff compares and `update_configuration` rewrites, so a status
2994    /// read and the supervise loop can never disagree about which protocol is in
2995    /// force.
2996    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
2997        Ok(self
2998            .inner
2999            .configuration
3000            .lock()
3001            .map_err(|_| SuperviseError::StatePoisoned {
3002                module_id: Some(self.inner.module_id.clone()),
3003            })?
3004            .spec
3005            .protocol)
3006    }
3007
3008    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3009        let configuration =
3010            self.inner
3011                .configuration
3012                .lock()
3013                .map_err(|_| SuperviseError::StatePoisoned {
3014                    module_id: Some(self.inner.module_id.clone()),
3015                })?;
3016        Ok((configuration.spec.clone(), configuration.health))
3017    }
3018
3019    /// Replace this module's launch spec, keeping its health and drain policy,
3020    /// the way a rescan does for a changed config entry. The running process is
3021    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3022    #[cfg(any(test, feature = "test-support"))]
3023    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3024        let (_, health) = self.configuration()?;
3025        let drain_timeout_ms = u64::try_from(
3026            self.inner
3027                .effective_drain_timeout
3028                .lock()
3029                .unwrap_or_else(|poisoned| poisoned.into_inner())
3030                .as_millis(),
3031        )
3032        .ok();
3033        self.update_configuration(spec, health, drain_timeout_ms)
3034            .await
3035    }
3036
3037    pub(crate) async fn update_configuration(
3038        &self,
3039        spec: ModuleSpec,
3040        health: HealthConfig,
3041        drain_timeout_ms: Option<u64>,
3042    ) -> Result<(), SuperviseError> {
3043        if spec.module_id != self.inner.module_id {
3044            return Err(SuperviseError::InvalidSpec {
3045                reason: "a supervised module's module_id cannot be changed".to_string(),
3046            });
3047        }
3048        validate_spec(&spec)?;
3049        let (reply_tx, reply_rx) = oneshot::channel();
3050        self.inner
3051            .commands
3052            .send(SupervisorCommand::UpdateConfiguration {
3053                spec: spec.clone(),
3054                health,
3055                drain_timeout_ms,
3056                reply: reply_tx,
3057            })
3058            .await
3059            .map_err(|_| SuperviseError::CommandClosed {
3060                module_id: self.inner.module_id.clone(),
3061            })?;
3062        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3063            module_id: self.inner.module_id.clone(),
3064        })?;
3065        let mut configuration =
3066            self.inner
3067                .configuration
3068                .lock()
3069                .map_err(|_| SuperviseError::StatePoisoned {
3070                    module_id: Some(self.inner.module_id.clone()),
3071                })?;
3072        configuration.spec = spec;
3073        configuration.health = health;
3074        Ok(())
3075    }
3076}
3077
3078impl Drop for SupervisedModuleInner {
3079    fn drop(&mut self) {
3080        let Ok(mut monitor) = self.monitor.lock() else {
3081            return;
3082        };
3083        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3084            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3085                state.state = ModuleState::Stopped;
3086                clear_current_process_facts(state);
3087            });
3088            monitor.abort();
3089        }
3090        let _ = monitor.take();
3091    }
3092}
3093
3094#[derive(Debug)]
3095enum SupervisorCommand {
3096    Drain {
3097        reply: oneshot::Sender<Result<(), SuperviseError>>,
3098    },
3099    Retire {
3100        reply: oneshot::Sender<Result<(), SuperviseError>>,
3101    },
3102    Restart {
3103        /// Operator override for this one restart's drain budget, in ms. `None`
3104        /// uses the module's configured/default budget; `Some(0)` cuts
3105        /// immediately (wedge bounce: a stuck request never settles, so
3106        /// waiting only delays recovery).
3107        drain_timeout_ms: Option<u64>,
3108        /// The module's `spawn_generation` when the request was received, before
3109        /// it waited in the command queue. A queued restart whose module has
3110        /// since spawned a newer process is already satisfied (see the handler).
3111        received_at_generation: u64,
3112        /// When the request entered the command queue, so the handler can log
3113        /// how long it waited behind the loop's other work.
3114        queued_at: Instant,
3115        reply: oneshot::Sender<Result<(), SuperviseError>>,
3116    },
3117    Reload {
3118        reply: oneshot::Sender<Result<(), SuperviseError>>,
3119    },
3120    SetEnabled {
3121        enabled: bool,
3122        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3123    },
3124    UpdateConfiguration {
3125        spec: ModuleSpec,
3126        health: HealthConfig,
3127        /// Per-module drain override from the new config; `None` re-resolves to
3128        /// the supervisor-wide default.
3129        drain_timeout_ms: Option<u64>,
3130        reply: oneshot::Sender<()>,
3131    },
3132    Swap {
3133        /// How long the candidate may take to register and declare itself
3134        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3135        ready_timeout: Option<Duration>,
3136        /// Answered at cutover or failure; the incumbent's drain follows.
3137        reply: oneshot::Sender<Result<(), SuperviseError>>,
3138    },
3139}
3140
3141#[derive(Debug)]
3142pub enum SuperviseError {
3143    InvalidSpec {
3144        reason: String,
3145    },
3146    Spawn {
3147        program: PathBuf,
3148        source: io::Error,
3149        cgroup_path: Option<PathBuf>,
3150    },
3151    Cgroup {
3152        module_id: String,
3153        source: io::Error,
3154    },
3155    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3156    /// than spawn a reserved module without its identity binding.
3157    LaunchNonce {
3158        reason: String,
3159    },
3160    Wait {
3161        module_id: String,
3162        source: io::Error,
3163    },
3164    Kill {
3165        module_id: String,
3166        source: io::Error,
3167    },
3168    Forwarding(ForwardingError),
3169    Registry(RegistryError),
3170    ReloadUnavailable {
3171        module_id: String,
3172        reason: String,
3173    },
3174    /// An operator restart/reload was requested for a module that is currently
3175    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3176    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3177    /// by a restart, so these commands are rejected instead of re-enabling it.
3178    Disabled {
3179        module_id: String,
3180    },
3181    ReloadFailed {
3182        module_id: String,
3183        reason: String,
3184    },
3185    RegistrationStillActive {
3186        module_id: String,
3187        waited: Duration,
3188    },
3189    StatePoisoned {
3190        module_id: Option<String>,
3191    },
3192    CommandClosed {
3193        module_id: String,
3194    },
3195    /// A restart or reload arrived while a swap's candidate was warming. The
3196    /// swap owns the module until it cuts over or fails; a stop or disable
3197    /// would have aborted it instead.
3198    SwapInProgress {
3199        module_id: String,
3200    },
3201    /// A swap was refused before anything was spawned.
3202    SwapRefused {
3203        module_id: String,
3204        reason: SwapRefusal,
3205    },
3206    /// A swap spawned a candidate and gave up on it. The candidate has been
3207    /// killed and its slot freed; the incumbent was left serving and was never
3208    /// drained, except in the one `CutoverLost` case described on that arm.
3209    SwapFailed {
3210        module_id: String,
3211        arm: SwapFailureArm,
3212        detail: String,
3213        /// How the candidate exited, when it exited on its own before the
3214        /// supervisor gave up on it.
3215        candidate_exit: Option<ExitReport>,
3216    },
3217}
3218
3219/// Why a swap was refused before a candidate was spawned.
3220#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3221pub enum SwapRefusal {
3222    /// The module's config does not declare `overlap: "safe"`.
3223    OverlapExclusive,
3224    /// The module is not registered, so there is no incumbent to keep serving
3225    /// and nothing a swap would improve on; a plain restart is the tool.
3226    NotRegistered,
3227    /// The module does not speak the subc wire, so a candidate could never
3228    /// register or declare itself ready.
3229    ProtocolNone,
3230    /// The supervisor lacks the forwarding table (to cut routes over) or the
3231    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3232    NotConfigured,
3233    /// A swap is already open for this module.
3234    AlreadySwapping,
3235}
3236
3237impl SwapRefusal {
3238    pub fn as_str(self) -> &'static str {
3239        match self {
3240            Self::OverlapExclusive => "overlap_exclusive",
3241            Self::NotRegistered => "not_registered",
3242            Self::ProtocolNone => "protocol_none",
3243            Self::NotConfigured => "not_configured",
3244            Self::AlreadySwapping => "already_swapping",
3245        }
3246    }
3247}
3248
3249/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3250/// serving and undrained; see `CutoverLost`.
3251#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3252pub enum SwapFailureArm {
3253    /// The candidate process could not be started.
3254    SpawnFailed,
3255    /// The candidate did not register within the readiness budget.
3256    NeverRegistered,
3257    /// The candidate registered but did not declare itself ready in time.
3258    NeverReady,
3259    /// The candidate exited before cutover.
3260    CandidateExited,
3261    /// The candidate declared itself ready but failed its health probe.
3262    CandidateUnhealthy,
3263    /// An operator stop, disable or retire arrived while the candidate warmed.
3264    /// The candidate was killed and the operator's command then carried out on
3265    /// the incumbent.
3266    Interrupted,
3267    /// The candidate's connection closed at the moment of cutover. If it
3268    /// closed before forwarding moved, the incumbent is untouched. If it closed
3269    /// between the forwarding and registry halves of cutover, forwarding can no
3270    /// longer route to the incumbent, so the module is restarted plainly.
3271    CutoverLost,
3272}
3273
3274impl SwapFailureArm {
3275    pub fn as_str(self) -> &'static str {
3276        match self {
3277            Self::SpawnFailed => "spawn_failed",
3278            Self::NeverRegistered => "never_registered",
3279            Self::NeverReady => "never_ready",
3280            Self::CandidateExited => "candidate_exited",
3281            Self::CandidateUnhealthy => "candidate_unhealthy",
3282            Self::Interrupted => "interrupted",
3283            Self::CutoverLost => "cutover_lost",
3284        }
3285    }
3286}
3287
3288impl fmt::Display for SuperviseError {
3289    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3290        match self {
3291            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3292            Self::Spawn {
3293                program,
3294                source,
3295                cgroup_path: Some(cgroup_path),
3296            } => write!(
3297                f,
3298                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3299                cgroup_path.display(),
3300                program.display()
3301            ),
3302            Self::Spawn {
3303                program,
3304                source,
3305                cgroup_path: None,
3306            } => write!(
3307                f,
3308                "failed to spawn module '{}': {source}",
3309                program.display()
3310            ),
3311            Self::Cgroup { module_id, source } => {
3312                write!(
3313                    f,
3314                    "failed to prepare cgroup for module '{module_id}': {source}"
3315                )
3316            }
3317            Self::LaunchNonce { reason } => {
3318                write!(
3319                    f,
3320                    "failed to generate reserved-module launch nonce: {reason}"
3321                )
3322            }
3323            Self::Wait { module_id, source } => {
3324                write!(f, "failed to wait for module '{module_id}': {source}")
3325            }
3326            Self::Kill { module_id, source } => {
3327                write!(f, "failed to kill module '{module_id}': {source}")
3328            }
3329            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3330            Self::Registry(err) => write!(f, "registry error: {err}"),
3331            Self::ReloadUnavailable { module_id, reason } => {
3332                write!(f, "reload unavailable for module '{module_id}': {reason}")
3333            }
3334            Self::Disabled { module_id } => {
3335                write!(
3336                    f,
3337                    "module '{module_id}' is disabled; enable it before restart or reload"
3338                )
3339            }
3340            Self::ReloadFailed { module_id, reason } => {
3341                write!(f, "reload failed for module '{module_id}': {reason}")
3342            }
3343            Self::RegistrationStillActive { module_id, waited } => write!(
3344                f,
3345                "module '{module_id}' registration remained active after waiting {waited:?}"
3346            ),
3347            Self::StatePoisoned { module_id } => match module_id {
3348                Some(module_id) => {
3349                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3350                }
3351                None => write!(f, "supervisor state was poisoned"),
3352            },
3353            Self::CommandClosed { module_id } => {
3354                write!(
3355                    f,
3356                    "supervisor command channel for module '{module_id}' is closed"
3357                )
3358            }
3359            Self::SwapInProgress { module_id } => write!(
3360                f,
3361                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3362            ),
3363            Self::SwapRefused { module_id, reason } => match reason {
3364                SwapRefusal::OverlapExclusive => write!(
3365                    f,
3366                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3367                ),
3368                SwapRefusal::NotRegistered => write!(
3369                    f,
3370                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3371                ),
3372                SwapRefusal::ProtocolNone => write!(
3373                    f,
3374                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3375                ),
3376                SwapRefusal::NotConfigured => write!(
3377                    f,
3378                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3379                ),
3380                SwapRefusal::AlreadySwapping => {
3381                    write!(f, "module '{module_id}' is already being swapped")
3382                }
3383            },
3384            Self::SwapFailed {
3385                module_id,
3386                arm,
3387                detail,
3388                ..
3389            } => write!(
3390                f,
3391                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3392                arm.as_str()
3393            ),
3394        }
3395    }
3396}
3397
3398impl Error for SuperviseError {
3399    fn source(&self) -> Option<&(dyn Error + 'static)> {
3400        match self {
3401            Self::Spawn { source, .. }
3402            | Self::Cgroup { source, .. }
3403            | Self::Wait { source, .. }
3404            | Self::Kill { source, .. } => Some(source),
3405            Self::Forwarding(err) => Some(err),
3406            Self::Registry(err) => Some(err),
3407            Self::LaunchNonce { .. }
3408            | Self::InvalidSpec { .. }
3409            | Self::ReloadUnavailable { .. }
3410            | Self::Disabled { .. }
3411            | Self::ReloadFailed { .. }
3412            | Self::RegistrationStillActive { .. }
3413            | Self::StatePoisoned { .. }
3414            | Self::CommandClosed { .. }
3415            | Self::SwapInProgress { .. }
3416            | Self::SwapRefused { .. }
3417            | Self::SwapFailed { .. } => None,
3418        }
3419    }
3420}
3421
3422pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3423    if spec.module_id.trim().is_empty() {
3424        return Err(SuperviseError::InvalidSpec {
3425            reason: "module_id must not be empty".to_string(),
3426        });
3427    }
3428
3429    Ok(())
3430}
3431
3432#[derive(Debug, Default)]
3433struct HealthProbeRuntime {
3434    registered_connection: Option<crate::ConnectionId>,
3435    advertised: bool,
3436    next_probe_at: Option<Instant>,
3437    probe_index: u64,
3438}
3439
3440impl HealthProbeRuntime {
3441    fn refresh_registration(
3442        &mut self,
3443        spec: &ModuleSpec,
3444        runtime: &SupervisorRuntimeConfig,
3445        registry: &Registry,
3446        snapshot: &SharedSnapshot,
3447    ) {
3448        // THE PROBE GATE FOR A MODULE THAT SPEAKS NO SUBC WIRE, placed here
3449        // because this is the only place that ever arms a probe: leaving
3450        // `advertised` false and `next_probe_at` empty makes `due()` false
3451        // forever, so `run_health_probe_cycle` -- and with it every arm of
3452        // `probe_module_health`, including the one that reads an absent
3453        // registration as proof the module is gone and escalates to a restart --
3454        // is unreachable for this module.
3455        //
3456        // That arm is right for a subc module and is exactly wrong here: a
3457        // `protocol: "none"` module never registers by declaration, so the
3458        // absence it would classify is the module working as configured.
3459        if spec.protocol == ModuleProtocol::None {
3460            self.registered_connection = None;
3461            self.advertised = false;
3462            self.next_probe_at = None;
3463            return;
3464        }
3465
3466        let registration = match registry.get_module(&spec.module_id) {
3467            Ok(registration) => registration,
3468            Err(err) => {
3469                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3470                self.advertised = false;
3471                self.next_probe_at = None;
3472                return;
3473            }
3474        };
3475
3476        let Some(registration) = registration else {
3477            self.registered_connection = None;
3478            self.advertised = false;
3479            self.next_probe_at = None;
3480            return;
3481        };
3482
3483        let advertised = registration
3484            .control_ops
3485            .iter()
3486            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
3487        if !advertised {
3488            self.registered_connection = Some(registration.connection_id);
3489            self.advertised = false;
3490            self.next_probe_at = None;
3491            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3492                state.health.status = SupervisorHealthStatus::Unknown;
3493                state.health.consecutive_failures = 0;
3494                state.health.last_probe_ms = None;
3495                state.health.detail = None;
3496                state.health.metrics = None;
3497            });
3498            return;
3499        }
3500
3501        let reregistered = self.registered_connection != Some(registration.connection_id);
3502        self.registered_connection = Some(registration.connection_id);
3503        self.advertised = true;
3504        if reregistered || self.next_probe_at.is_none() {
3505            self.probe_index = 0;
3506            self.next_probe_at = Some(
3507                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3508            );
3509            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3510                state.health.status = SupervisorHealthStatus::Unknown;
3511                state.health.consecutive_failures = 0;
3512                state.health.detail = None;
3513                state.health.metrics = None;
3514            });
3515        }
3516    }
3517
3518    fn wake_after(&self) -> Duration {
3519        if !self.advertised {
3520            return REGISTRY_RELEASE_POLL;
3521        }
3522        self.next_probe_at
3523            .map(|next| next.saturating_duration_since(Instant::now()))
3524            .unwrap_or(REGISTRY_RELEASE_POLL)
3525    }
3526
3527    fn due(&self) -> bool {
3528        self.advertised
3529            && self
3530                .next_probe_at
3531                .is_some_and(|next| Instant::now() >= next)
3532    }
3533
3534    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
3535        self.probe_index = self.probe_index.wrapping_add(1);
3536        self.next_probe_at = Some(
3537            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
3538        );
3539    }
3540}
3541
3542/// What a failed health probe actually OBSERVED, kept apart from how it reads.
3543///
3544/// This was a struct with a single `message: String`, and every one of the
3545/// fifteen construction sites collapsed into it. Each site knows exactly what it
3546/// saw -- the lane is gone, the module did not answer in time, the module
3547/// answered with the wrong thing -- and `handle_health_probe_failure` then
3548/// treated all of them identically: increment a counter, compare to a threshold,
3549/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
3550/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
3551///
3552/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
3553///
3554/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
3555///   answer on it again.
3556/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
3557///   AND with a perfectly healthy one that lost a CPU race -- which is what
3558///   happens under machine load, and is how this supervisor killed a healthy
3559///   module three times in one day.
3560/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
3561///   Restarting on it is defensible, but it is not the silence case and should
3562///   never be counted as one.
3563/// * `Misconfigured` is a daemon-side fault. The module has not been asked
3564///   anything, so it cannot be evidence about the module at all.
3565///
3566/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
3567/// one that fires most often, and while every variant collapsed into one string
3568/// it carried the same weight as the strongest.
3569///
3570/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
3571/// DESIGN and a reader stopping at it gets the build backwards: the restart
3572/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
3573/// probes still increment the failure streak and drive escalation at the
3574/// threshold (see `is_proof_of_death` below for why that is deliberate and
3575/// what gates the change). Absence of evidence restarts modules today.
3576#[derive(Debug)]
3577enum HealthProbeEvidence {
3578    /// The module's control lane is gone. Proof of death.
3579    LaneDead,
3580    /// No reply within the deadline. Proves nothing about the module's state.
3581    NoAnswer,
3582    /// The module replied, but not with a usable health report. Proves it is alive.
3583    BadAnswer,
3584    /// The daemon could not ask. Says nothing about the module.
3585    Misconfigured,
3586}
3587
3588#[derive(Debug)]
3589struct HealthProbeError {
3590    evidence: HealthProbeEvidence,
3591    message: String,
3592}
3593
3594impl HealthProbeError {
3595    fn lane_dead(message: impl Into<String>) -> Self {
3596        Self::with(HealthProbeEvidence::LaneDead, message)
3597    }
3598
3599    fn no_answer(message: impl Into<String>) -> Self {
3600        Self::with(HealthProbeEvidence::NoAnswer, message)
3601    }
3602
3603    fn bad_answer(message: impl Into<String>) -> Self {
3604        Self::with(HealthProbeEvidence::BadAnswer, message)
3605    }
3606
3607    fn misconfigured(message: impl Into<String>) -> Self {
3608        Self::with(HealthProbeEvidence::Misconfigured, message)
3609    }
3610
3611    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
3612        Self {
3613            evidence,
3614            message: message.into(),
3615        }
3616    }
3617
3618    /// Whether this observation is proof the module cannot serve.
3619    ///
3620    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
3621    /// variant that fires under CPU starvation, and treating it as proof is the
3622    /// defect this enum exists to make impossible to reintroduce silently.
3623    ///
3624    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
3625    /// to restart also needs a bound for the case it excludes -- a genuinely
3626    /// wedged module, alive but never answering -- and that bound must come from
3627    /// the distribution of real late-answer latencies, which nothing measures
3628    /// yet. Landing the classification first makes the later change a one-line
3629    /// decision against evidence that already exists, rather than two unproven
3630    /// changes at once.
3631    #[allow(dead_code)]
3632    fn is_proof_of_death(&self) -> bool {
3633        matches!(self.evidence, HealthProbeEvidence::LaneDead)
3634    }
3635
3636    /// Short stable label for logs and the health snapshot.
3637    ///
3638    /// An operator reading `ck health` currently cannot tell "the module is gone"
3639    /// from "the module did not answer in five seconds", because both render as
3640    /// prose in the same field. These labels are what make the two
3641    /// distinguishable at a glance, and they are what a later restart-policy
3642    /// change will be argued from.
3643    fn label(&self) -> &'static str {
3644        match self.evidence {
3645            HealthProbeEvidence::LaneDead => "lane-dead",
3646            HealthProbeEvidence::NoAnswer => "no-answer",
3647            HealthProbeEvidence::BadAnswer => "bad-answer",
3648            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
3649        }
3650    }
3651}
3652
3653impl fmt::Display for HealthProbeError {
3654    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3655        f.write_str(&self.message)
3656    }
3657}
3658
3659async fn run_health_probe_cycle(
3660    spec: &ModuleSpec,
3661    runtime: &SupervisorRuntimeConfig,
3662    registry: &Registry,
3663    process_liveness: &SupervisorProcessLiveness,
3664    snapshot: &SharedSnapshot,
3665    child: &mut Option<SupervisedChild>,
3666) {
3667    let now_ms = unix_ms_now();
3668    match probe_module_health(&spec.module_id, runtime, None).await {
3669        Ok(report) => {
3670            handle_health_report(
3671                spec,
3672                runtime,
3673                registry,
3674                process_liveness,
3675                snapshot,
3676                child,
3677                report,
3678                now_ms,
3679            )
3680            .await;
3681        }
3682        Err(err) => {
3683            handle_health_probe_failure(
3684                spec,
3685                runtime,
3686                registry,
3687                process_liveness,
3688                snapshot,
3689                child,
3690                err,
3691                now_ms,
3692            )
3693            .await;
3694        }
3695    }
3696}
3697
3698async fn probe_module_health(
3699    module_id: &str,
3700    runtime: &SupervisorRuntimeConfig,
3701    drain_deadline: Option<Instant>,
3702) -> Result<HealthReport, HealthProbeError> {
3703    let Some(forwarding) = runtime.forwarding.as_ref() else {
3704        return Err(HealthProbeError::misconfigured(
3705            "supervisor was not configured with a forwarding table",
3706        ));
3707    };
3708    let probe_started_at = Instant::now();
3709    let mut deadline = probe_started_at + runtime.health.deadline;
3710    if let Some(drain_deadline) = drain_deadline {
3711        deadline = deadline.min(drain_deadline);
3712    }
3713    let pending = if drain_deadline.is_some() {
3714        forwarding.begin_drain_health_probe_rpc_for(
3715            module_id,
3716            MODULE_CONTROL_OP_HEALTH_CHECK,
3717            probe_started_at,
3718            deadline,
3719        )
3720    } else {
3721        forwarding.begin_health_probe_rpc_for(
3722            module_id,
3723            MODULE_CONTROL_OP_HEALTH_CHECK,
3724            probe_started_at,
3725            deadline,
3726        )
3727    }
3728    .map_err(|err| {
3729        // The endpoint is not registered, so there is no live control lane to
3730        // ask. That is the module being absent, not slow.
3731        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3732    })?;
3733    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3734}
3735
3736/// [`probe_module_health`] for one endpoint rather than the id's active one.
3737///
3738/// A swap probes two processes that no by-id lookup reaches: its candidate
3739/// before cutover, and its superseded incumbent (for busy gauges) while the
3740/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
3741/// bounds the by-id drain probe.
3742async fn probe_endpoint_health(
3743    endpoint: crate::ModuleEndpointId,
3744    runtime: &SupervisorRuntimeConfig,
3745    deadline_cap: Option<Instant>,
3746) -> Result<HealthReport, HealthProbeError> {
3747    let Some(forwarding) = runtime.forwarding.as_ref() else {
3748        return Err(HealthProbeError::misconfigured(
3749            "supervisor was not configured with a forwarding table",
3750        ));
3751    };
3752    let probe_started_at = Instant::now();
3753    let mut deadline = probe_started_at + runtime.health.deadline;
3754    if let Some(cap) = deadline_cap {
3755        deadline = deadline.min(cap);
3756    }
3757    let pending = forwarding
3758        .begin_endpoint_health_probe_rpc_for(
3759            endpoint,
3760            MODULE_CONTROL_OP_HEALTH_CHECK,
3761            probe_started_at,
3762            deadline,
3763        )
3764        .map_err(|err| {
3765            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3766        })?;
3767    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3768}
3769
3770/// Send a begun health probe and classify its answer.
3771async fn await_health_probe(
3772    forwarding: &ForwardingTable,
3773    pending: PendingModuleControlRpc,
3774    deadline: Instant,
3775    probe_budget: Duration,
3776) -> Result<HealthReport, HealthProbeError> {
3777    let PendingModuleControlRpc {
3778        endpoint,
3779        module_sink,
3780        negotiated_ver,
3781        corr,
3782        receiver,
3783    } = pending;
3784    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
3785        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
3786    })?;
3787    let frame = Frame::build_with_version(
3788        negotiated_ver,
3789        FrameType::Request,
3790        control_flags(),
3791        0,
3792        0,
3793        corr,
3794        body,
3795    )
3796    .map_err(|err| {
3797        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
3798    })?;
3799
3800    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
3801    // blocks waiting for capacity when the module's egress queue is full, and an
3802    // unbounded await here freezes the whole supervision actor (it stops polling
3803    // Child::wait and supervisor commands), making the module unrecoverable
3804    // in-band. On timeout the probe fails like any transport failure.
3805    match timeout_at(deadline, module_sink.send(frame)).await {
3806        Ok(Ok(())) => {}
3807        Ok(Err(err)) => {
3808            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3809            // A closed sink means the module's egress channel is gone -- the
3810            // receiving half is dropped when its connection tears down. Proof.
3811            return Err(HealthProbeError::lane_dead(format!(
3812                "failed to send health.check: {err}"
3813            )));
3814        }
3815        Err(_elapsed) => {
3816            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3817            // A full egress queue means the module is not draining its socket, which
3818            // is consistent with a wedged module AND with one whose reader is merely
3819            // starved. Silence, not proof.
3820            return Err(HealthProbeError::no_answer(
3821                "health.check send timed out before enqueue (module egress full)",
3822            ));
3823        }
3824    }
3825
3826    match timeout_at(deadline, receiver).await {
3827        // Each arm records WHAT WAS OBSERVED. Four of them are the module
3828        // demonstrably answering -- rejected, non-health, malformed, wrong op --
3829        // and those prove it is alive even though the probe failed.
3830        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
3831            response.health_report().ok_or_else(|| {
3832                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
3833            })
3834        }
3835        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
3836            format!("health.check rejected: {}", body.message),
3837        )),
3838        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
3839            Err(HealthProbeError::lane_dead(message))
3840        }
3841        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
3842            Err(HealthProbeError::bad_answer(message))
3843        }
3844        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
3845            Err(HealthProbeError::bad_answer(format!(
3846                "expected module-control op '{expected}', got '{actual}'"
3847            )))
3848        }
3849        // A reply that crosses the deadline before this waiter observes it is
3850        // still proof of life. The forwarding path records its end-to-end latency
3851        // before delivering this classification.
3852        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
3853            "module answered health.check after its daemon deadline",
3854        )),
3855        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
3856            "health.check waiter was canceled before the module responded",
3857        )),
3858        Err(_) => {
3859            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
3860            Err(HealthProbeError::no_answer(format!(
3861                "module did not answer health.check within {probe_budget:?}"
3862            )))
3863        }
3864    }
3865}
3866
3867#[allow(clippy::too_many_arguments)]
3868async fn handle_health_report(
3869    spec: &ModuleSpec,
3870    runtime: &SupervisorRuntimeConfig,
3871    registry: &Registry,
3872    process_liveness: &SupervisorProcessLiveness,
3873    snapshot: &SharedSnapshot,
3874    child: &mut Option<SupervisedChild>,
3875    report: HealthReport,
3876    now_ms: u64,
3877) {
3878    let status = supervisor_health_status(report.status);
3879    let detail = report.detail.clone();
3880    let metrics = truncate_health_metrics(report.metrics);
3881    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3882        state.health.status = status;
3883        state.health.last_probe_ms = Some(now_ms);
3884        state.health.detail = detail.clone();
3885        state.health.metrics = metrics.clone();
3886        state.health.consecutive_failures = 0;
3887    });
3888
3889    let action = match report.status {
3890        HealthStatus::Ok => return,
3891        HealthStatus::Degraded => runtime.health.on_degraded,
3892        HealthStatus::Failing => runtime.health.on_failing,
3893    };
3894    apply_l3_health_action(
3895        spec,
3896        runtime,
3897        registry,
3898        process_liveness,
3899        snapshot,
3900        child,
3901        status,
3902        detail.as_deref(),
3903        action,
3904        now_ms,
3905    )
3906    .await;
3907}
3908
3909#[allow(clippy::too_many_arguments)]
3910async fn handle_health_probe_failure(
3911    spec: &ModuleSpec,
3912    runtime: &SupervisorRuntimeConfig,
3913    registry: &Registry,
3914    process_liveness: &SupervisorProcessLiveness,
3915    snapshot: &SharedSnapshot,
3916    child: &mut Option<SupervisedChild>,
3917    err: HealthProbeError,
3918    now_ms: u64,
3919) {
3920    let threshold = runtime.health.failure_threshold.max(1);
3921    let mut failures = 0;
3922    // Carry the evidence class into the operator-visible detail. Without it,
3923    // "module did not answer within 5s" and "the control lane is gone" are two
3924    // prose strings in the same field, and the reader has to know the codebase to
3925    // tell which one is proof of anything.
3926    let detail = format!("[{}] {err}", err.label());
3927    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3928        state.health.last_probe_ms = Some(now_ms);
3929        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3930        state.health.detail = Some(detail.clone());
3931        state.health.metrics = None;
3932        failures = state.health.consecutive_failures;
3933    });
3934
3935    if failures < threshold {
3936        warn!(
3937            module_id = %spec.module_id,
3938            consecutive_failures = failures,
3939            threshold,
3940            evidence = err.label(),
3941            detail = %detail,
3942            "health.check probe failed"
3943        );
3944        return;
3945    }
3946
3947    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3948        state.state = ModuleState::Unresponsive;
3949        state.health.status = SupervisorHealthStatus::Unresponsive;
3950    });
3951    // The evidence class is logged at the kill site because this is the line an
3952    // operator reads after an unexplained restart. A streak of `no-answer` under
3953    // machine load is the known false-positive shape; a `lane-dead` is not.
3954    if runtime.health.critical {
3955        error!(
3956            module_id = %spec.module_id,
3957            status = "unresponsive",
3958            evidence = err.label(),
3959            detail = %detail,
3960            "critical module health alert"
3961        );
3962    } else {
3963        warn!(
3964            module_id = %spec.module_id,
3965            status = "unresponsive",
3966            evidence = err.label(),
3967            detail = %detail,
3968            "module health threshold breached"
3969        );
3970    }
3971    if let Err(err) = health_restart_child(
3972        spec,
3973        runtime,
3974        registry,
3975        process_liveness,
3976        snapshot,
3977        child,
3978        SupervisorHealthStatus::Unresponsive,
3979        Some(&detail),
3980        now_ms,
3981    )
3982    .await
3983    {
3984        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
3985    }
3986}
3987
3988#[allow(clippy::too_many_arguments)]
3989async fn apply_l3_health_action(
3990    spec: &ModuleSpec,
3991    runtime: &SupervisorRuntimeConfig,
3992    registry: &Registry,
3993    process_liveness: &SupervisorProcessLiveness,
3994    snapshot: &SharedSnapshot,
3995    child: &mut Option<SupervisedChild>,
3996    status: SupervisorHealthStatus,
3997    detail: Option<&str>,
3998    action: HealthAction,
3999    now_ms: u64,
4000) {
4001    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4002    match action {
4003        HealthAction::Report => {
4004            info!(
4005                module_id = %spec.module_id,
4006                status = ?status,
4007                detail,
4008                "module reported non-ok health"
4009            );
4010        }
4011        HealthAction::Alert => {
4012            error!(
4013                module_id = %spec.module_id,
4014                status = ?status,
4015                detail,
4016                "module health alert"
4017            );
4018        }
4019        HealthAction::Restart => {
4020            if let Err(err) = health_restart_child(
4021                spec,
4022                runtime,
4023                registry,
4024                process_liveness,
4025                snapshot,
4026                child,
4027                status,
4028                detail,
4029                now_ms,
4030            )
4031            .await
4032            {
4033                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4034            }
4035        }
4036    }
4037}
4038
4039#[allow(clippy::too_many_arguments)]
4040async fn health_restart_child(
4041    spec: &ModuleSpec,
4042    runtime: &SupervisorRuntimeConfig,
4043    registry: &Registry,
4044    process_liveness: &SupervisorProcessLiveness,
4045    snapshot: &SharedSnapshot,
4046    child: &mut Option<SupervisedChild>,
4047    status: SupervisorHealthStatus,
4048    detail: Option<&str>,
4049    now_ms: u64,
4050) -> Result<(), SuperviseError> {
4051    let (enabled, schedule) = {
4052        let mut state = lock_snapshot(snapshot)?;
4053        let enabled = state.enabled;
4054        let schedule = if enabled {
4055            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4056        } else {
4057            None
4058        };
4059        (enabled, schedule)
4060    };
4061
4062    if !enabled {
4063        return Err(SuperviseError::Disabled {
4064            module_id: spec.module_id.clone(),
4065        });
4066    }
4067
4068    if schedule.is_none() {
4069        record_health_action(snapshot, &spec.module_id, "disabled".to_string(), now_ms);
4070        error!(
4071            module_id = %spec.module_id,
4072            status = ?status,
4073            detail,
4074            max_restarts = runtime.restart_policy.max_restarts,
4075            window_secs = runtime.restart_policy.window.as_secs(),
4076            "health restart budget exhausted; disabling module"
4077        );
4078        let stop_notice = begin_forwarding_drain_if_configured(
4079            spec,
4080            runtime,
4081            registry,
4082            snapshot,
4083            Some(false),
4084            RouteCloseReason::Disable,
4085        )
4086        .await?;
4087        drain_optional_child(
4088            &spec.module_id,
4089            spec.protocol,
4090            stop_notice,
4091            registry,
4092            snapshot,
4093            &runtime.terminal_ring,
4094            &runtime.spawn_events,
4095            child,
4096            runtime.drain_timeout,
4097            ModuleState::Disabled,
4098            Some(false),
4099        )
4100        .await?;
4101        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4102        return Ok(());
4103    }
4104
4105    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4106    let mut restart_count = 0;
4107    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4108        restart_count = state.crash_restarts.len();
4109        state.state = ModuleState::Unresponsive;
4110        state.health.status = status;
4111        state.health.last_action = Some(HealthAction::Restart.to_string());
4112        state.health.last_action_ms = Some(now_ms);
4113    })?;
4114    warn!(
4115        module_id = %spec.module_id,
4116        status = ?status,
4117        detail,
4118        restart_count,
4119        restart_in_window = schedule.restart_in_window,
4120        delay_ms = schedule.delay.as_millis() as u64,
4121        "health-triggered module restart"
4122    );
4123
4124    let stop_notice = begin_forwarding_drain_if_configured(
4125        spec,
4126        runtime,
4127        registry,
4128        snapshot,
4129        Some(true),
4130        RouteCloseReason::Restart,
4131    )
4132    .await?;
4133    drain_optional_child(
4134        &spec.module_id,
4135        spec.protocol,
4136        stop_notice,
4137        registry,
4138        snapshot,
4139        &runtime.terminal_ring,
4140        &runtime.spawn_events,
4141        child,
4142        runtime.drain_timeout,
4143        ModuleState::Restarting,
4144        Some(true),
4145    )
4146    .await?;
4147    sleep(schedule.delay).await;
4148    // The backoff may have outlasted the restart it was counting down to: an
4149    // operator disable or drain in between moves the snapshot out of
4150    // `Restarting`, and that stop must win over this respawn.
4151    if !respawn_still_pending(snapshot) {
4152        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4153        return Ok(());
4154    }
4155    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
4156    match spawn_and_mark_running(spec, runtime, snapshot) {
4157        Ok(next_child) => {
4158            *child = Some(next_child);
4159            Ok(())
4160        }
4161        Err(err) => {
4162            fail_snapshot(snapshot, Some(&spec.module_id), None);
4163            process_liveness.untrack_if_current(&spec.module_id, snapshot);
4164            *child = None;
4165            Err(err)
4166        }
4167    }
4168}
4169
4170fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4171    let _ = update_snapshot(snapshot, Some(module_id), |state| {
4172        state.health.last_action = Some(action);
4173        state.health.last_action_ms = Some(now_ms);
4174    });
4175}
4176
4177fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4178    match status {
4179        HealthStatus::Ok => SupervisorHealthStatus::Ok,
4180        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4181        HealthStatus::Failing => SupervisorHealthStatus::Failing,
4182    }
4183}
4184
4185/// Caps the metrics blob stored in the cached supervisor snapshot, which is
4186/// returned to every `supervisor.list` and `supervisor.health` caller.
4187///
4188/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
4189/// path: that request exists to return a module's complete metrics object, and
4190/// `ck health <module-id>` documents it as the way to see what the cached view
4191/// truncates. The asymmetry is the feature.
4192///
4193/// So a new caller must decide which side it is on rather than assume the cap is
4194/// universal. Reaching for it on a fresh-probe path would silently reintroduce
4195/// the truncation that path exists to avoid.
4196fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4197    let metrics = metrics?;
4198    match serde_json::to_vec(&metrics) {
4199        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4200            "truncated": true,
4201            "original_bytes": encoded.len(),
4202        })),
4203        Ok(_) | Err(_) => Some(metrics),
4204    }
4205}
4206
4207/// Spread health probes so a fleet-wide restart does not converge them.
4208///
4209/// The delay is derived from the module id and probe index rather than a random
4210/// source, so it is deterministic per module: a module keeps its own offset
4211/// across daemon restarts instead of re-rolling into a collision.
4212fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4213    if cadence.is_zero() {
4214        return Duration::ZERO;
4215    }
4216    let cadence_ms = cadence.as_millis() as u64;
4217    // This early return is REDUNDANT, deliberately, and a mutation run will show
4218    // it surviving removal. Recording why here so the next person to notice does
4219    // not have to re-derive it:
4220    //
4221    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
4222    //   a zero cadence and builds the Duration from whole milliseconds, so a
4223    //   sub-millisecond cadence cannot come from config.
4224    // - Even if reached it changes no answer. The `.max(1)` below makes the span
4225    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
4226    //   -- exactly what this returns.
4227    //
4228    // Kept as a guard against a future widening of the config parser (accepting
4229    // microseconds, say), which would make the sub-millisecond case reachable.
4230    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
4231    // divides by zero. Remove this and nothing changes.
4232    if cadence_ms == 0 {
4233        return cadence;
4234    }
4235    // Note that this never returns less than one cadence, including for the FIRST
4236    // probe. So a freshly registered module reports health `unknown` for a full
4237    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
4238    // ready to answer.
4239    //
4240    // That is a property of the supervisor's schedule, not of any module: an
4241    // operator watching a restart sees `unknown` and cannot tell it from a module
4242    // that is slow to warm. Measured on two unrelated modules, both flipping to
4243    // `ok` between 22s and 32s after restart.
4244    //
4245    // Left as-is because spreading the first probe is what keeps a fleet-wide
4246    // restart from firing fourteen simultaneous probes into a cold machine. The
4247    // alternative -- probe at t+0 and jitter only from the second onward -- trades
4248    // that thundering herd for a faster first reading.
4249    let jitter_span = (cadence_ms / 10).max(1);
4250    let hash = module_id.as_bytes().iter().fold(
4251        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
4252        |acc, byte| {
4253            acc.wrapping_mul(1099511628211)
4254                .wrapping_add(u64::from(*byte))
4255        },
4256    );
4257    cadence + Duration::from_millis(hash % jitter_span)
4258}
4259
4260#[cfg(test)]
4261mod tests {
4262    use super::*;
4263
4264    #[test]
4265    fn readding_a_module_clears_its_rescan_removal_tombstone() {
4266        let handle = SupervisorHandle::new();
4267        let module_id = "readded-tombstone";
4268        handle.record_rescan_removal(module_id);
4269        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
4270
4271        handle.apply_identity_configuration(&ModuleSpec {
4272            module_id: module_id.to_string(),
4273            program: PathBuf::from("/test/module"),
4274            args: Vec::new(),
4275            env: Vec::new(),
4276            reserved: false,
4277            reserved_prefixes: Vec::new(),
4278            protocol: ModuleProtocol::Subc,
4279            overlap: Default::default(),
4280        });
4281
4282        assert!(
4283            handle.removal_tombstone_age_ms(module_id).is_none(),
4284            "a re-added module must not retain a stale removal tombstone"
4285        );
4286    }
4287
4288    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
4289        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
4290        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
4291            snapshot.process_alive = true;
4292            snapshot.pid = Some(41);
4293            snapshot.spawned_at_ms = Some(42);
4294            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
4295            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
4296                device: 43,
4297                inode: 44,
4298            });
4299        })
4300        .unwrap();
4301        snapshot
4302    }
4303
4304    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
4305        let snapshot = lock_snapshot(snapshot).unwrap();
4306        assert!(!snapshot.process_alive);
4307        assert_eq!(snapshot.pid, None);
4308        assert_eq!(snapshot.spawned_at_ms, None);
4309        assert_eq!(snapshot.spawned_from, None);
4310        assert_eq!(snapshot.spawned_file_identity, None);
4311    }
4312
4313    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4314    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
4315        let supervisor = Supervisor::default();
4316        let mut runtime = supervisor.runtime_config();
4317        runtime.test_seed_stale_facts_before_enable_spawn = true;
4318        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
4319        let mut child = None;
4320        let spec = ModuleSpec {
4321            module_id: "failed-enable-clears-facts".to_string(),
4322            program: PathBuf::from("/definitely/missing/failed-enable-module"),
4323            args: Vec::new(),
4324            env: Vec::new(),
4325            reserved: false,
4326            reserved_prefixes: Vec::new(),
4327            protocol: ModuleProtocol::Subc,
4328            overlap: Default::default(),
4329        };
4330
4331        let result = set_child_enabled(
4332            &spec,
4333            &runtime,
4334            &supervisor.registry,
4335            &supervisor.process_liveness,
4336            &snapshot,
4337            &mut child,
4338            true,
4339        )
4340        .await;
4341
4342        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
4343        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4344        assert_snapshot_process_facts_cleared(&snapshot);
4345    }
4346
4347    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4348    async fn failed_reload_spawn_clears_current_process_facts() {
4349        let supervisor = Supervisor::default();
4350        let mut runtime = supervisor.runtime_config();
4351        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
4352        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4353        let mut child = None;
4354        let spec = ModuleSpec {
4355            module_id: "failed-reload-clears-facts".to_string(),
4356            program: PathBuf::from("/unused/failed-reload-module"),
4357            args: Vec::new(),
4358            env: Vec::new(),
4359            reserved: false,
4360            reserved_prefixes: Vec::new(),
4361            protocol: ModuleProtocol::Subc,
4362            overlap: Default::default(),
4363        };
4364
4365        let result = handle_reload_spawn_failure(
4366            &spec,
4367            &runtime,
4368            &supervisor.process_liveness,
4369            &snapshot,
4370            &mut child,
4371            "forced reload spawn failure".to_string(),
4372        )
4373        .await;
4374
4375        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
4376        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4377        assert_snapshot_process_facts_cleared(&snapshot);
4378    }
4379
4380    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4381    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
4382        let supervisor = Supervisor::default();
4383        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4384        let module = supervisor.supervised_module(
4385            ModuleSpec {
4386                module_id: "drop-clears-facts".to_string(),
4387                program: PathBuf::from("/unused/drop-module"),
4388                args: Vec::new(),
4389                env: Vec::new(),
4390                reserved: false,
4391                reserved_prefixes: Vec::new(),
4392                protocol: ModuleProtocol::Subc,
4393                overlap: Default::default(),
4394            },
4395            supervisor.runtime_config(),
4396            Arc::clone(&snapshot),
4397            None,
4398        );
4399        assert!(!module
4400            .inner
4401            .monitor
4402            .lock()
4403            .unwrap()
4404            .as_ref()
4405            .unwrap()
4406            .is_finished());
4407
4408        drop(module);
4409
4410        assert_eq!(
4411            lock_snapshot(&snapshot).unwrap().state,
4412            ModuleState::Stopped
4413        );
4414        assert_snapshot_process_facts_cleared(&snapshot);
4415    }
4416
4417    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4418    async fn configuration_update_does_not_replace_captured_running_process_facts() {
4419        let supervisor = Supervisor::default();
4420        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4421        let initial = ModuleSpec {
4422            module_id: "rescan-preserves-spawn-facts".to_string(),
4423            program: PathBuf::from("/spawned/module"),
4424            args: Vec::new(),
4425            env: Vec::new(),
4426            reserved: false,
4427            reserved_prefixes: Vec::new(),
4428            protocol: ModuleProtocol::Subc,
4429            overlap: Default::default(),
4430        };
4431        let module = supervisor.supervised_module(
4432            initial.clone(),
4433            supervisor.runtime_config(),
4434            snapshot,
4435            None,
4436        );
4437        let before = module.status().unwrap();
4438        let mut replacement = initial;
4439        replacement.program = PathBuf::from("/rescanned/replacement-module");
4440
4441        module
4442            .update_configuration(replacement, HealthConfig::default(), None)
4443            .await
4444            .unwrap();
4445
4446        let after = module.status().unwrap();
4447        assert_eq!(after.pid, before.pid);
4448        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
4449        assert_eq!(after.spawned_from, before.spawned_from);
4450        drop(module);
4451    }
4452}
4453
4454fn unix_ms_now() -> u64 {
4455    SystemTime::now()
4456        .duration_since(UNIX_EPOCH)
4457        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
4458        .unwrap_or(0)
4459}
4460
4461async fn supervise_loop(
4462    mut spec: ModuleSpec,
4463    mut runtime: SupervisorRuntimeConfig,
4464    registry: Arc<Registry>,
4465    process_liveness: Arc<SupervisorProcessLiveness>,
4466    snapshot: SharedSnapshot,
4467    mut child: Option<SupervisedChild>,
4468    mut commands: mpsc::Receiver<SupervisorCommand>,
4469) {
4470    let mut health_probe = HealthProbeRuntime::default();
4471    // Deadline of the crash respawn whose backoff is currently elapsing. While
4472    // it is set the loop serves commands instead of sleeping inside the exit
4473    // arm, so a disable or drain lands immediately and cancels the respawn.
4474    let mut pending_respawn: Option<Instant> = None;
4475    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
4476    // before anything else so a stop that interrupted a swap runs at once.
4477    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
4478    loop {
4479        if let Some(command) = requeued.pop_front() {
4480            if !handle_supervisor_command(
4481                command,
4482                &mut spec,
4483                &mut runtime,
4484                &registry,
4485                &process_liveness,
4486                &snapshot,
4487                &mut child,
4488                &mut commands,
4489                &mut requeued,
4490            )
4491            .await
4492            {
4493                return;
4494            }
4495            if child.is_some() || !respawn_still_pending(&snapshot) {
4496                pending_respawn = None;
4497            }
4498            continue;
4499        }
4500        if child.is_some() {
4501            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
4502            let probe_sleep = sleep(health_probe.wake_after());
4503            tokio::pin!(probe_sleep);
4504            let active_child = child.as_mut().expect("child checked above");
4505            tokio::select! {
4506                wait_result = active_child.wait() => {
4507                    // Every arm below that gives up on the CHILD must keep the
4508                    // supervision task itself alive (child = None, loop
4509                    // continues into command-serving mode). Returning here
4510                    // closes the command channel, which makes the module
4511                    // permanently unrestartable in-band: a clean child exit
4512                    // of an enabled module once wedged the fleet this way
4513                    // ('supervisor command channel is closed') and required a
4514                    // full daemon restart to recover.
4515                    let exit_report = match wait_result {
4516                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
4517                        Err(err) => {
4518                            active_child.drain_stderr(&spec.module_id).await;
4519                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4520                            // Every other exit path (on_child_exit's Clean/Crash arms,
4521                            // the reload-registration-failure path) records a terminal
4522                            // before moving on. Without one here, a module whose wait()
4523                            // itself errored (e.g. already reaped) leaves no terminal
4524                            // record at all -- an empty ring reads as "nothing died".
4525                            record_wait_error_terminal(
4526                                &spec.module_id,
4527                                &runtime.terminal_ring,
4528                                &runtime.spawn_events,
4529                            );
4530                            untrack_if_registration_released(
4531                                &process_liveness,
4532                                &registry,
4533                                &spec.module_id,
4534                                &snapshot,
4535                            );
4536                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
4537                            child = None;
4538                            continue;
4539                        }
4540                    };
4541                    active_child.drain_stderr(&spec.module_id).await;
4542
4543                    let next = on_child_exit(
4544                        &spec,
4545                        runtime.restart_policy,
4546                        &registry,
4547                        &snapshot,
4548                        &runtime.terminal_ring,
4549                        &runtime.spawn_events,
4550                        &runtime.child_roster,
4551                        exit_report,
4552                    ).await;
4553                    // The exit is recorded, so a daemon shutdown may stop
4554                    // waiting for this child (see `SupervisedChild::wait`).
4555                    active_child.release_roster();
4556                    match next {
4557                        NextAction::Stop { registration_released } => {
4558                            if registration_released {
4559                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4560                            }
4561                            child = None;
4562                        }
4563                        NextAction::Restart { schedule } => {
4564                            let delay = schedule.map_or(
4565                                runtime.restart_policy.delay_for_restart(0),
4566                                |schedule| schedule.delay,
4567                            );
4568                            if let Some(schedule) = schedule {
4569                                log_crash_respawn(&spec.module_id, schedule);
4570                            }
4571                            // The exited child is fully recorded at this point,
4572                            // so release it and count the backoff down in the
4573                            // command-serving branch below rather than sleeping
4574                            // here: commands cannot be received from inside this
4575                            // select arm, and an operator disable or drain that
4576                            // arrives during the backoff must cancel the pending
4577                            // respawn instead of waiting for it to spawn first.
4578                            child = None;
4579                            pending_respawn = Some(Instant::now() + delay);
4580                        }
4581                    }
4582                }
4583                command = commands.recv() => {
4584                    let Some(command) = command else {
4585                        return;
4586                    };
4587                    if !handle_supervisor_command(
4588                        command,
4589                        &mut spec,
4590                        &mut runtime,
4591                        &registry,
4592                        &process_liveness,
4593                        &snapshot,
4594                        &mut child,
4595                        &mut commands,
4596                        &mut requeued,
4597                    ).await {
4598                        return;
4599                    }
4600                }
4601                _ = &mut probe_sleep => {
4602                    if health_probe.due() {
4603                        run_health_probe_cycle(
4604                            &spec,
4605                            &runtime,
4606                            &registry,
4607                            &process_liveness,
4608                            &snapshot,
4609                            &mut child,
4610                        ).await;
4611                        if child.is_some() {
4612                            health_probe.schedule_next(&spec, runtime.health.cadence);
4613                        }
4614                    }
4615                }
4616            }
4617        } else if let Some(deadline) = pending_respawn {
4618            tokio::select! {
4619                _ = sleep_until(deadline) => {
4620                    pending_respawn = None;
4621                    // A command handled below while the backoff elapsed may
4622                    // have stopped the module; never respawn past an operator's
4623                    // disable or drain.
4624                    if !respawn_still_pending(&snapshot) {
4625                        continue;
4626                    }
4627                    // The daemon began shutting down during the backoff: the
4628                    // spawn would be refused anyway, and refusing it here
4629                    // leaves the module stopped instead of reporting a
4630                    // failed restart.
4631                    if runtime.child_roster.is_closed() {
4632                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4633                            state.state = ModuleState::Stopped;
4634                        });
4635                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
4636                        continue;
4637                    }
4638                    if let Err(err) = wait_for_registration_release(
4639                        &registry,
4640                        &spec.module_id,
4641                        REGISTRY_RELEASE_TIMEOUT,
4642                    ).await {
4643                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
4644                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
4645                        continue;
4646                    }
4647
4648                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
4649                        Ok(next_child) => {
4650                            child = Some(next_child);
4651                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
4652                        }
4653                        Err(err) => {
4654                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4655                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4656                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
4657                        }
4658                    }
4659                }
4660                command = commands.recv() => {
4661                    let Some(command) = command else {
4662                        return;
4663                    };
4664                    if !handle_supervisor_command(
4665                        command,
4666                        &mut spec,
4667                        &mut runtime,
4668                        &registry,
4669                        &process_liveness,
4670                        &snapshot,
4671                        &mut child,
4672                        &mut commands,
4673                        &mut requeued,
4674                    ).await {
4675                        return;
4676                    }
4677                    // Reconcile the pending respawn with what the command did:
4678                    // a restart or reload has already spawned a fresh child,
4679                    // while a disable or drain moved the snapshot out of the
4680                    // state the respawn was counting down from.
4681                    if child.is_some() || !respawn_still_pending(&snapshot) {
4682                        pending_respawn = None;
4683                    }
4684                }
4685            }
4686        } else {
4687            let Some(command) = commands.recv().await else {
4688                return;
4689            };
4690            if !handle_supervisor_command(
4691                command,
4692                &mut spec,
4693                &mut runtime,
4694                &registry,
4695                &process_liveness,
4696                &snapshot,
4697                &mut child,
4698                &mut commands,
4699                &mut requeued,
4700            )
4701            .await
4702            {
4703                return;
4704            }
4705        }
4706    }
4707}
4708
4709fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
4710    info!(
4711        module_id,
4712        restart_in_window = schedule.restart_in_window,
4713        delay_ms = schedule.delay.as_millis() as u64,
4714        "respawning after crash"
4715    );
4716}
4717
4718/// Whether the respawn a backoff was counting down to is still wanted. A
4719/// disable or drain handled while the backoff elapsed moves the snapshot out
4720/// of `Restarting`, and the operator's stop must win over the pending respawn,
4721/// so every sleep-then-spawn path re-validates against the live snapshot
4722/// instead of assuming the state it left behind still holds.
4723fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
4724    matches!(
4725        lock_snapshot(snapshot),
4726        Ok(state) if state.enabled && state.state == ModuleState::Restarting
4727    )
4728}
4729
4730enum NextAction {
4731    Stop {
4732        registration_released: bool,
4733    },
4734    Restart {
4735        schedule: Option<CrashRestartSchedule>,
4736    },
4737}
4738
4739#[allow(clippy::too_many_arguments)]
4740async fn handle_supervisor_command(
4741    command: SupervisorCommand,
4742    spec: &mut ModuleSpec,
4743    runtime: &mut SupervisorRuntimeConfig,
4744    registry: &Registry,
4745    process_liveness: &SupervisorProcessLiveness,
4746    snapshot: &SharedSnapshot,
4747    child: &mut Option<SupervisedChild>,
4748    commands: &mut mpsc::Receiver<SupervisorCommand>,
4749    requeued: &mut VecDeque<SupervisorCommand>,
4750) -> bool {
4751    match command {
4752        SupervisorCommand::Drain { reply } => {
4753            // A plain stop runs no forwarding drain, so nothing reaches the
4754            // module over its connection before the wait: ask by signal.
4755            let result = drain_optional_child(
4756                &spec.module_id,
4757                spec.protocol,
4758                StopNotice::NotSent,
4759                registry,
4760                snapshot,
4761                &runtime.terminal_ring,
4762                &runtime.spawn_events,
4763                child,
4764                runtime.drain_timeout,
4765                ModuleState::Stopped,
4766                None,
4767            )
4768            .await;
4769            let registration_released = result.is_ok();
4770            let _ = reply.send(result);
4771            if registration_released {
4772                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4773            }
4774            false
4775        }
4776        SupervisorCommand::Retire { reply } => {
4777            let result = async {
4778                let stop_notice = begin_forwarding_drain_if_configured(
4779                    spec,
4780                    runtime,
4781                    registry,
4782                    snapshot,
4783                    None,
4784                    RouteCloseReason::Disable,
4785                )
4786                .await?;
4787                drain_optional_child(
4788                    &spec.module_id,
4789                    spec.protocol,
4790                    stop_notice,
4791                    registry,
4792                    snapshot,
4793                    &runtime.terminal_ring,
4794                    &runtime.spawn_events,
4795                    child,
4796                    runtime.drain_timeout,
4797                    ModuleState::Stopped,
4798                    None,
4799                )
4800                .await
4801            }
4802            .await;
4803            let registration_released = result.is_ok();
4804            let _ = reply.send(result);
4805            if registration_released {
4806                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4807            }
4808            false
4809        }
4810        SupervisorCommand::Restart {
4811            drain_timeout_ms,
4812            received_at_generation,
4813            queued_at,
4814            reply,
4815        } => {
4816            // Without this line a restart that waited in the queue (behind a
4817            // health probe cycle or another command) was invisible: the log
4818            // showed only the drain timing out, minutes after the operator's call.
4819            info!(
4820                module_id = %spec.module_id,
4821                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
4822                "restart command dequeued"
4823            );
4824            // ACK AT INITIATION, not completion. The blocking form deadlocked any
4825            // caller whose own request lane rides the module being restarted: the
4826            // caller's in-flight request keeps the drain from quiescing, the drain
4827            // keeps the restart from completing, and the completion keeps the reply
4828            // from releasing the caller — so the drain always timed out and cut the
4829            // initiator with a GOODBYE, even on a healthy module. Replying once the
4830            // restart is validated lets a self-lane caller settle, which is exactly
4831            // what makes the drain succeed. Completion is observable via
4832            // supervisor.list / module status; a post-ack failure lands the module
4833            // in a visible terminal state below rather than in a reply nobody can
4834            // receive.
4835            let validation = match lock_snapshot(snapshot) {
4836                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
4837                    module_id: spec.module_id.clone(),
4838                }),
4839                Ok(_) => Ok(()),
4840                Err(err) => Err(err),
4841            };
4842            let initiated = validation.is_ok();
4843            let _ = reply.send(validation);
4844            // A restart asks for a fresh process. Commands run one at a time,
4845            // so a restart queued behind another restart (two operator calls
4846            // in quick succession) is dequeued the moment the first one has
4847            // spawned its replacement -- before that process has sent HELLO.
4848            // Running it would drain and kill the process the first restart
4849            // just produced, which is the opposite of what both callers asked
4850            // for. If a process spawned after this request was received is
4851            // still supervised, the request is already satisfied. Not when the
4852            // configuration changed since that spawn: then the newer process
4853            // predates the spec this restart may exist to apply.
4854            let satisfied_by_generation = if initiated && child.is_some() {
4855                lock_snapshot(snapshot).ok().and_then(|state| {
4856                    (state.spawn_generation > received_at_generation
4857                        && !state.configuration_updated_since_spawn)
4858                        .then_some(state.spawn_generation)
4859                })
4860            } else {
4861                None
4862            };
4863            if let Some(generation) = satisfied_by_generation {
4864                info!(
4865                    module_id = %spec.module_id,
4866                    received_at_generation,
4867                    "restart already satisfied by generation {generation}; not restarting again"
4868                );
4869            } else if initiated {
4870                // Precedence: this restart's operator override, else the module's
4871                // configured budget (already resolved into the runtime).
4872                let drain_timeout = drain_timeout_ms
4873                    .map(Duration::from_millis)
4874                    .unwrap_or(runtime.drain_timeout);
4875                if let Err(err) = restart_child(
4876                    spec,
4877                    runtime,
4878                    registry,
4879                    process_liveness,
4880                    snapshot,
4881                    child,
4882                    drain_timeout,
4883                )
4884                .await
4885                {
4886                    warn!(
4887                        module_id = %spec.module_id,
4888                        error = %err,
4889                        "operator restart failed after initiation ack; module state carries the outcome"
4890                    );
4891                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4892                        state.state = ModuleState::Failed;
4893                        clear_current_process_facts(state);
4894                    });
4895                }
4896            }
4897            true
4898        }
4899        SupervisorCommand::Reload { reply } => {
4900            let result =
4901                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
4902            let _ = reply.send(result);
4903            true
4904        }
4905        SupervisorCommand::SetEnabled { enabled, reply } => {
4906            let result = set_child_enabled(
4907                spec,
4908                runtime,
4909                registry,
4910                process_liveness,
4911                snapshot,
4912                child,
4913                enabled,
4914            )
4915            .await;
4916            let _ = reply.send(result);
4917            true
4918        }
4919        SupervisorCommand::UpdateConfiguration {
4920            spec: next_spec,
4921            health,
4922            drain_timeout_ms,
4923            reply,
4924        } => {
4925            if let Some(handle) = &runtime.supervisor_handle {
4926                handle.apply_identity_configuration(&next_spec);
4927            }
4928            *spec = next_spec;
4929            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4930                state.configuration_updated_since_spawn = true;
4931            });
4932            runtime.health = health;
4933            runtime.drain_timeout = drain_timeout_ms
4934                .map(Duration::from_millis)
4935                .unwrap_or(runtime.default_drain_timeout);
4936            *runtime
4937                .effective_drain_timeout
4938                .lock()
4939                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
4940            let _ = reply.send(());
4941            true
4942        }
4943        SupervisorCommand::Swap {
4944            ready_timeout,
4945            reply,
4946        } => {
4947            let end = swap::run_swap(
4948                spec,
4949                runtime,
4950                registry,
4951                process_liveness,
4952                snapshot,
4953                child,
4954                commands,
4955                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
4956                reply,
4957            )
4958            .await;
4959            requeued.extend(end.requeue);
4960            true
4961        }
4962    }
4963}
4964
4965async fn restart_child(
4966    spec: &ModuleSpec,
4967    runtime: &SupervisorRuntimeConfig,
4968    registry: &Registry,
4969    process_liveness: &SupervisorProcessLiveness,
4970    snapshot: &SharedSnapshot,
4971    child: &mut Option<SupervisedChild>,
4972    drain_timeout: Duration,
4973) -> Result<(), SuperviseError> {
4974    // Restart cycles a running module; it must not silently start a disabled one.
4975    if !lock_snapshot(snapshot)?.enabled {
4976        return Err(SuperviseError::Disabled {
4977            module_id: spec.module_id.clone(),
4978        });
4979    }
4980    let stop_notice = begin_forwarding_drain_with_timeout(
4981        spec,
4982        runtime,
4983        registry,
4984        snapshot,
4985        None,
4986        RouteCloseReason::Restart,
4987        drain_timeout,
4988    )
4989    .await?;
4990
4991    if child.is_some() {
4992        drain_optional_child(
4993            &spec.module_id,
4994            spec.protocol,
4995            stop_notice,
4996            registry,
4997            snapshot,
4998            &runtime.terminal_ring,
4999            &runtime.spawn_events,
5000            child,
5001            drain_timeout,
5002            ModuleState::Restarting,
5003            Some(true),
5004        )
5005        .await?;
5006    } else {
5007        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5008            state.enabled = true;
5009            state.state = ModuleState::Restarting;
5010            clear_current_process_facts(state);
5011        })?;
5012        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5013    }
5014
5015    reset_restart_count(snapshot, &spec.module_id)?;
5016    sleep(runtime.restart_policy.backoff).await;
5017    // A disable or drain that landed during the backoff cancels this respawn:
5018    // the operator's stop must win over the restart the sleep counted down to.
5019    if !respawn_still_pending(snapshot) {
5020        process_liveness.untrack_if_current(&spec.module_id, snapshot);
5021        return Ok(());
5022    }
5023    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5024    // Mirror health_restart_child's spawn-failure handling: of the four
5025    // spawn-failure sites this was the only one that propagated with the
5026    // snapshot still reading `Restarting` -- neither running nor failed, and
5027    // unrevivable by `set_enabled(true)` (issue #34). `Failed` is the state the
5028    // operator can see and heal.
5029    match spawn_and_mark_running(spec, runtime, snapshot) {
5030        Ok(next_child) => {
5031            *child = Some(next_child);
5032            debug!(module_id = %spec.module_id, "supervised module restarted by operator request");
5033            Ok(())
5034        }
5035        Err(err) => {
5036            fail_snapshot(snapshot, Some(&spec.module_id), None);
5037            process_liveness.untrack_if_current(&spec.module_id, snapshot);
5038            *child = None;
5039            Err(err)
5040        }
5041    }
5042}
5043
5044async fn reload_child(
5045    spec: &ModuleSpec,
5046    runtime: &SupervisorRuntimeConfig,
5047    registry: &Registry,
5048    process_liveness: &SupervisorProcessLiveness,
5049    snapshot: &SharedSnapshot,
5050    child: &mut Option<SupervisedChild>,
5051) -> Result<(), SuperviseError> {
5052    // Reload cycles a running module; it must not silently start a disabled one.
5053    if !lock_snapshot(snapshot)?.enabled {
5054        return Err(SuperviseError::Disabled {
5055            module_id: spec.module_id.clone(),
5056        });
5057    }
5058    let stop_notice = begin_forwarding_drain(
5059        spec,
5060        runtime,
5061        registry,
5062        snapshot,
5063        Some(true),
5064        RouteCloseReason::Reload,
5065    )
5066    .await?;
5067
5068    if child.is_some() {
5069        drain_optional_child(
5070            &spec.module_id,
5071            spec.protocol,
5072            stop_notice,
5073            registry,
5074            snapshot,
5075            &runtime.terminal_ring,
5076            &runtime.spawn_events,
5077            child,
5078            runtime.drain_timeout,
5079            ModuleState::Restarting,
5080            Some(true),
5081        )
5082        .await?;
5083    } else {
5084        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5085            state.enabled = true;
5086            state.state = ModuleState::Restarting;
5087            clear_current_process_facts(state);
5088        })?;
5089        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5090    }
5091
5092    reset_restart_count(snapshot, &spec.module_id)?;
5093    sleep(runtime.restart_policy.backoff).await;
5094    // A disable or drain that landed during the backoff cancels this respawn:
5095    // the operator's stop must win over the restart the sleep counted down to.
5096    if !respawn_still_pending(snapshot) {
5097        process_liveness.untrack_if_current(&spec.module_id, snapshot);
5098        return Ok(());
5099    }
5100    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5101    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5102        Ok(next_child) => next_child,
5103        Err(err) => {
5104            return handle_reload_spawn_failure(
5105                spec,
5106                runtime,
5107                process_liveness,
5108                snapshot,
5109                child,
5110                format!("new child failed to spawn: {err}"),
5111            )
5112            .await;
5113        }
5114    };
5115    *child = Some(next_child);
5116
5117    let wait_outcome = {
5118        let active_child = child.as_mut().expect("new reload child was just stored");
5119        wait_for_registration_after_reload(
5120            registry,
5121            &spec.module_id,
5122            snapshot,
5123            active_child,
5124            REGISTRY_RELEASE_TIMEOUT,
5125        )
5126        .await?
5127    };
5128
5129    match wait_outcome {
5130        RegistrationWaitOutcome::Registered => {
5131            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
5132            Ok(())
5133        }
5134        RegistrationWaitOutcome::Exited(exit_report) => {
5135            if let Some(active_child) = child.as_mut() {
5136                active_child.drain_stderr(&spec.module_id).await;
5137            }
5138            *child = None;
5139            handle_reload_child_registration_failure(
5140                spec,
5141                runtime,
5142                registry,
5143                process_liveness,
5144                snapshot,
5145                child,
5146                ReloadRegistrationFailure {
5147                    exit_report: registration_failure_exit_report(exit_report),
5148                    reason: "new child exited before registering".to_string(),
5149                },
5150            )
5151            .await
5152        }
5153        RegistrationWaitOutcome::TimedOut => {
5154            let mut timed_out_child = child
5155                .take()
5156                .expect("timed-out reload child is still running");
5157            timed_out_child
5158                .start_kill()
5159                .map_err(|source| SuperviseError::Kill {
5160                    module_id: spec.module_id.clone(),
5161                    source,
5162                })?;
5163            let status = timed_out_child
5164                .wait()
5165                .await
5166                .map_err(|source| SuperviseError::Wait {
5167                    module_id: spec.module_id.clone(),
5168                    source,
5169                })?;
5170            timed_out_child.drain_stderr(&spec.module_id).await;
5171            handle_reload_child_registration_failure(
5172                spec,
5173                runtime,
5174                registry,
5175                process_liveness,
5176                snapshot,
5177                child,
5178                ReloadRegistrationFailure {
5179                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
5180                        snapshot,
5181                        &timed_out_child,
5182                        &status,
5183                    )),
5184                    reason: format!(
5185                        "new child did not register within {:?}",
5186                        REGISTRY_RELEASE_TIMEOUT
5187                    ),
5188                },
5189            )
5190            .await
5191        }
5192    }
5193}
5194
5195async fn set_child_enabled(
5196    spec: &ModuleSpec,
5197    runtime: &SupervisorRuntimeConfig,
5198    registry: &Registry,
5199    process_liveness: &SupervisorProcessLiveness,
5200    snapshot: &SharedSnapshot,
5201    child: &mut Option<SupervisedChild>,
5202    enabled: bool,
5203) -> Result<bool, SuperviseError> {
5204    let (current_enabled, current_state) = {
5205        let state = lock_snapshot(snapshot)?;
5206        (state.enabled, state.state)
5207    };
5208    // `start` (enable on an already-enabled module) heals TERMINAL states instead
5209    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
5210    // clean (Stopped) has no live process and no other in-band recovery — the
5211    // operator's start is the explicit recovery act and resets the budget. Without
5212    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
5213    // which the 2026-07-14 aft outage proved is a trap when the failed module is
5214    // the one providing every agent's shell.
5215    let revive_terminal = enabled
5216        && current_enabled
5217        && child.is_none()
5218        && matches!(current_state, ModuleState::Failed | ModuleState::Stopped);
5219    if current_enabled == enabled && !revive_terminal {
5220        return Ok(false);
5221    }
5222
5223    if enabled {
5224        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5225            state.enabled = true;
5226            state.state = ModuleState::Starting;
5227            clear_current_process_facts(state);
5228        })?;
5229        #[cfg(test)]
5230        if runtime.test_seed_stale_facts_before_enable_spawn {
5231            update_snapshot(snapshot, Some(&spec.module_id), |state| {
5232                state.process_alive = true;
5233                state.pid = Some(41);
5234                state.spawned_at_ms = Some(42);
5235                state.spawned_from = Some(PathBuf::from("/spawned/module"));
5236                state.spawned_file_identity = Some(SpawnedFileIdentity {
5237                    device: 43,
5238                    inode: 44,
5239                });
5240            })?;
5241        }
5242        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5243        reset_restart_count(snapshot, &spec.module_id)?;
5244        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5245        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5246            Ok(next_child) => next_child,
5247            Err(err) => {
5248                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5249                    state.state = ModuleState::Failed;
5250                    clear_current_process_facts(state);
5251                }) {
5252                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
5253                }
5254                process_liveness.untrack_if_current(&spec.module_id, snapshot);
5255                return Err(err);
5256            }
5257        };
5258        *child = Some(next_child);
5259        debug!(module_id = %spec.module_id, "supervised module enabled");
5260        Ok(true)
5261    } else {
5262        let stop_notice = begin_forwarding_drain_if_configured(
5263            spec,
5264            runtime,
5265            registry,
5266            snapshot,
5267            Some(false),
5268            RouteCloseReason::Disable,
5269        )
5270        .await?;
5271        drain_optional_child(
5272            &spec.module_id,
5273            spec.protocol,
5274            stop_notice,
5275            registry,
5276            snapshot,
5277            &runtime.terminal_ring,
5278            &runtime.spawn_events,
5279            child,
5280            runtime.drain_timeout,
5281            ModuleState::Disabled,
5282            Some(false),
5283        )
5284        .await?;
5285        debug!(module_id = %spec.module_id, "supervised module disabled");
5286        Ok(true)
5287    }
5288}
5289
5290#[allow(clippy::too_many_arguments)]
5291async fn on_child_exit(
5292    spec: &ModuleSpec,
5293    policy: RestartPolicy,
5294    registry: &Registry,
5295    snapshot: &SharedSnapshot,
5296    terminal_ring: &Arc<Mutex<TerminalRing>>,
5297    spawn_events: &SpawnEventFeed,
5298    roster: &ChildRoster,
5299    exit_report: ExitReport,
5300) -> NextAction {
5301    // Once the daemon has begun shutting down, no exit is a crash to recover
5302    // from: the module is exiting because the daemon is going away (EOF on its
5303    // connection, or a service manager signalling the whole cgroup). Record it
5304    // as such and never schedule a respawn, which would only start a process
5305    // for the shutdown to end again.
5306    if roster.is_closed() {
5307        return on_child_exit_during_daemon_shutdown(
5308            spec,
5309            registry,
5310            snapshot,
5311            terminal_ring,
5312            spawn_events,
5313            exit_report,
5314        )
5315        .await;
5316    }
5317    // Every stop the supervisor itself asks for (operator stop, disable,
5318    // restart, reload, swap, a health restart, a drain that runs out of budget)
5319    // takes the child out of the supervise loop and reaps it in
5320    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
5321    // that reaches this point was not requested by the daemon.
5322    //
5323    // For a subc-wire module a clean exit is still a stop: those modules are
5324    // written to re-raise SIGTERM, so a stray outside signal already reads as a
5325    // crash, and exiting 0 is a deliberate choice the module made. A
5326    // `protocol: "none"` module is a stock program we cannot change, and many
5327    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
5328    // stop would leave the module down for good after any stray signal, so it
5329    // goes through the crash path instead: it spends restart budget, respawns
5330    // with the crash backoff, and ends `failed` when the budget runs out.
5331    let unrequested_clean_exit_of_protocol_none =
5332        exit_report.kind == ExitKind::Clean && spec.protocol == ModuleProtocol::None;
5333    match exit_report.kind {
5334        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
5335            info!(
5336                module_id = %spec.module_id,
5337                exit_code = ?exit_report.code,
5338                exit_signal = ?exit_report.signal,
5339                "supervised module exited cleanly"
5340            );
5341            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5342                state.state = ModuleState::Stopped;
5343                clear_current_process_facts(state);
5344                state.last_exit = Some(exit_report.clone());
5345            }) {
5346                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
5347            }
5348            record_terminal(
5349                &spec.module_id,
5350                terminal_ring,
5351                spawn_events,
5352                &exit_report,
5353                TerminalDisposition::Stopped,
5354            );
5355            let registration_released = match wait_for_registration_release(
5356                registry,
5357                &spec.module_id,
5358                REGISTRY_RELEASE_TIMEOUT,
5359            )
5360            .await
5361            {
5362                Ok(()) => true,
5363                Err(err) => {
5364                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
5365                    false
5366                }
5367            };
5368            NextAction::Stop {
5369                registration_released,
5370            }
5371        }
5372        ExitKind::Clean | ExitKind::Crash => {
5373            if unrequested_clean_exit_of_protocol_none {
5374                warn!(
5375                    module_id = %spec.module_id,
5376                    exit_code = ?exit_report.code,
5377                    exit_signal = ?exit_report.signal,
5378                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
5379                );
5380            } else {
5381                warn!(
5382                    module_id = %spec.module_id,
5383                    exit_code = ?exit_report.code,
5384                    exit_signal = ?exit_report.signal,
5385                    "supervised module exited abnormally (crash)"
5386                );
5387            }
5388            let mut restart_schedule = None;
5389            let mut disposition = TerminalDisposition::Disabled;
5390            // Set only when the budget is what stopped the module, so the
5391            // terminal record says which limit was hit rather than leaving
5392            // `failed` to be read as "crashed once, badly".
5393            let mut disposition_detail = None;
5394            let now = Instant::now();
5395            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5396                clear_current_process_facts(state);
5397                state.last_exit = Some(exit_report.clone());
5398                if state.enabled {
5399                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
5400                        state.state = ModuleState::Restarting;
5401                        restart_schedule = Some(schedule);
5402                        disposition = TerminalDisposition::Restarting;
5403                    } else {
5404                        state.state = ModuleState::Failed;
5405                        disposition = TerminalDisposition::Failed;
5406                        disposition_detail = Some(policy.budget_exhausted_detail());
5407                    }
5408                } else {
5409                    state.state = ModuleState::Disabled;
5410                    disposition = TerminalDisposition::Disabled;
5411                }
5412            }) {
5413                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
5414                return NextAction::Stop {
5415                    registration_released: false,
5416                };
5417            }
5418            if disposition_detail.is_some() {
5419                // The window is in the message, not only in the fields: this line
5420                // is read in a scrollback where a bare `max_restarts=3` reads as a
5421                // lifetime cap and sends the operator looking for three crashes
5422                // that never happened together.
5423                error!(
5424                    module_id = %spec.module_id,
5425                    max_restarts = policy.max_restarts,
5426                    window_secs = policy.window.as_secs(),
5427                    "module stopped: {}",
5428                    policy.budget_exhausted_detail()
5429                );
5430            }
5431            record_terminal_with_detail(
5432                &spec.module_id,
5433                terminal_ring,
5434                spawn_events,
5435                &exit_report,
5436                disposition,
5437                disposition_detail,
5438            );
5439
5440            if let Some(schedule) = restart_schedule {
5441                NextAction::Restart {
5442                    schedule: Some(schedule),
5443                }
5444            } else {
5445                let registration_released = match wait_for_registration_release(
5446                    registry,
5447                    &spec.module_id,
5448                    REGISTRY_RELEASE_TIMEOUT,
5449                )
5450                .await
5451                {
5452                    Ok(()) => true,
5453                    Err(err) => {
5454                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
5455                        false
5456                    }
5457                };
5458                NextAction::Stop {
5459                    registration_released,
5460                }
5461            }
5462        }
5463        ExitKind::DeliberateSeverance => {
5464            warn!(
5465                module_id = %spec.module_id,
5466                exit_code = ?exit_report.code,
5467                exit_signal = ?exit_report.signal,
5468                "supervised module exited after deliberate connection severance"
5469            );
5470            let mut should_restart = false;
5471            let mut disposition = TerminalDisposition::Disabled;
5472            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5473                clear_current_process_facts(state);
5474                state.last_exit = Some(exit_report.clone());
5475                state.lifetime_restarts += 1;
5476                if state.enabled {
5477                    state.state = ModuleState::Restarting;
5478                    should_restart = true;
5479                    disposition = TerminalDisposition::Restarting;
5480                } else {
5481                    state.state = ModuleState::Disabled;
5482                }
5483            }) {
5484                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
5485                return NextAction::Stop {
5486                    registration_released: false,
5487                };
5488            }
5489            record_terminal(
5490                &spec.module_id,
5491                terminal_ring,
5492                spawn_events,
5493                &exit_report,
5494                disposition,
5495            );
5496
5497            if should_restart {
5498                NextAction::Restart { schedule: None }
5499            } else {
5500                let registration_released = match wait_for_registration_release(
5501                    registry,
5502                    &spec.module_id,
5503                    REGISTRY_RELEASE_TIMEOUT,
5504                )
5505                .await
5506                {
5507                    Ok(()) => true,
5508                    Err(err) => {
5509                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
5510                        false
5511                    }
5512                };
5513                NextAction::Stop {
5514                    registration_released,
5515                }
5516            }
5517        }
5518    }
5519}
5520
5521async fn on_child_exit_during_daemon_shutdown(
5522    spec: &ModuleSpec,
5523    registry: &Registry,
5524    snapshot: &SharedSnapshot,
5525    terminal_ring: &Arc<Mutex<TerminalRing>>,
5526    spawn_events: &SpawnEventFeed,
5527    exit_report: ExitReport,
5528) -> NextAction {
5529    info!(
5530        module_id = %spec.module_id,
5531        exit_code = ?exit_report.code,
5532        exit_signal = ?exit_report.signal,
5533        exit_kind = ?exit_report.kind,
5534        "supervised module exited during daemon shutdown; not restarting it"
5535    );
5536    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5537        state.state = ModuleState::Stopped;
5538        clear_current_process_facts(state);
5539        state.last_exit = Some(exit_report.clone());
5540    }) {
5541        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
5542    }
5543    record_terminal(
5544        &spec.module_id,
5545        terminal_ring,
5546        spawn_events,
5547        &exit_report,
5548        TerminalDisposition::DaemonShutdown,
5549    );
5550    let registration_released =
5551        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
5552            .await
5553            .is_ok();
5554    NextAction::Stop {
5555        registration_released,
5556    }
5557}
5558
5559fn record_wait_error_terminal(
5560    module_id: &str,
5561    terminal_ring: &Arc<Mutex<TerminalRing>>,
5562    spawn_events: &SpawnEventFeed,
5563) {
5564    record_terminal(
5565        module_id,
5566        terminal_ring,
5567        spawn_events,
5568        &wait_error_exit_report(),
5569        TerminalDisposition::Failed,
5570    );
5571}
5572
5573fn record_terminal(
5574    module_id: &str,
5575    terminal_ring: &Arc<Mutex<TerminalRing>>,
5576    spawn_events: &SpawnEventFeed,
5577    exit_report: &ExitReport,
5578    disposition: TerminalDisposition,
5579) {
5580    record_terminal_with_detail(
5581        module_id,
5582        terminal_ring,
5583        spawn_events,
5584        exit_report,
5585        disposition,
5586        None,
5587    );
5588}
5589
5590/// The ring lock is held only to capture the read (see
5591/// `TerminalJournal::capture_read`), so this module's exits keep recording
5592/// while the journal files are read. Blocking: it reads files.
5593fn durable_terminal_history_of(
5594    terminal_ring: &Mutex<TerminalRing>,
5595    module_id: &str,
5596) -> subc_control::TerminalHistory {
5597    let read = terminal_ring
5598        .lock()
5599        .unwrap_or_else(|p| p.into_inner())
5600        .capture_durable_history();
5601    read.read(module_id)
5602}
5603
5604fn record_terminal_with_detail(
5605    module_id: &str,
5606    terminal_ring: &Arc<Mutex<TerminalRing>>,
5607    spawn_events: &SpawnEventFeed,
5608    exit_report: &ExitReport,
5609    disposition: TerminalDisposition,
5610    disposition_detail: Option<String>,
5611) {
5612    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
5613    let record = TerminalRecord {
5614        exit_code: exit_report.code,
5615        exit_signal: exit_report.signal,
5616        at_ms: exit_report.at_ms,
5617        disposition,
5618        exit_kind: exit_report.kind.into(),
5619        disposition_detail,
5620    };
5621    terminal_ring
5622        .lock()
5623        .unwrap_or_else(|poisoned| poisoned.into_inner())
5624        .record_exit(module_id, record);
5625}
5626
5627fn untrack_if_registration_released(
5628    process_liveness: &SupervisorProcessLiveness,
5629    registry: &Registry,
5630    module_id: &str,
5631    snapshot: &SharedSnapshot,
5632) {
5633    match registry.get_module(module_id) {
5634        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
5635        Ok(Some(_)) => {}
5636        Err(err) => {
5637            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
5638        }
5639    }
5640}
5641
5642/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
5643/// then apply the module's configured entries minus daemon-private capture keys.
5644///
5645/// Separated from `spawn_child` only so it can be asserted without spawning a
5646/// process — a duplicate of this logic in a test would pass while the real one
5647/// drifted, which is the defect class this function exists to avoid.
5648/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
5649/// nonce. A `protocol: "none"` module gets neither, because it cannot use
5650/// either and the argument would stop a stock binary from starting at all.
5651/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
5652///
5653/// The plain-spawn form, kept for the tests that assert its plan; spawns go
5654/// through [`apply_wire_spawn_args_for_role`].
5655#[cfg(test)]
5656fn apply_wire_spawn_args(
5657    command: &mut Command,
5658    spec: &ModuleSpec,
5659    connection_file_path: Option<&std::path::Path>,
5660    handle: Option<&SupervisorHandle>,
5661) -> Result<Option<NonceHandoff>, SuperviseError> {
5662    apply_wire_spawn_args_for_role(
5663        command,
5664        spec,
5665        connection_file_path,
5666        handle,
5667        SpawnRole::Plain,
5668    )
5669}
5670
5671/// The read end of a spawn's launch-nonce pipe, prepared by
5672/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
5673/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
5674/// handoff and keeps only the environment copy.
5675#[cfg(unix)]
5676type NonceHandoff = subc_os::LaunchNonceHandoff;
5677#[cfg(not(unix))]
5678type NonceHandoff = std::convert::Infallible;
5679
5680/// Prepare wire identity for a plain spawn or a swap candidate.
5681///
5682/// A plain spawn replaces the module's recorded nonce. A swap candidate records
5683/// a separate candidate token so the still-serving incumbent and its consumers
5684/// keep their nonce. Both records are installed before the process exists, so
5685/// the child's initial HELLO registration cannot arrive ahead of its nonce.
5686///
5687/// On Unix the nonce is delivered only through a pipe. It is written into
5688/// a pipe whose read end the child gets as descriptor 3, named by
5689/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
5690/// process of the same user cannot read it with `ps eww`. That handoff is
5691/// returned rather than installed here, because installing it replaces
5692/// whatever the child has at descriptor 3 and so must be the last pre-exec
5693/// step, after the Linux cgroup placement that the caller registers later.
5694/// Windows retains the environment handoff until restricted handle inheritance
5695/// can be implemented outside std's process primitives.
5696fn apply_wire_spawn_args_for_role(
5697    command: &mut Command,
5698    spec: &ModuleSpec,
5699    connection_file_path: Option<&std::path::Path>,
5700    handle: Option<&SupervisorHandle>,
5701    role: SpawnRole,
5702) -> Result<Option<NonceHandoff>, SuperviseError> {
5703    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
5704    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
5705    // included: a daemon started from a module's process tree inherits it,
5706    // and passing it on would point the child at a descriptor it does not
5707    // have.
5708    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
5709    // Remove inherited or configured copies too: withholding must mean absent.
5710    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
5711    if spec.protocol == ModuleProtocol::None {
5712        return Ok(None);
5713    }
5714    if let Some(connection_file_path) = connection_file_path {
5715        command.arg(SUBC_ARG).arg(connection_file_path);
5716    }
5717
5718    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
5719    // route.open attestation. Reserved modules additionally use the same nonce
5720    // for HELLO id-squatting protection. A respawn rotates both records.
5721    let nonce = generate_launch_nonce()?;
5722    if let Some(handle) = handle {
5723        match role {
5724            SpawnRole::Plain => {
5725                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
5726                if spec.reserved {
5727                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
5728                }
5729            }
5730            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
5731        }
5732    }
5733    #[cfg(unix)]
5734    let handoff = {
5735        let handoff =
5736            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
5737                program: spec.program.clone(),
5738                source,
5739                cgroup_path: None,
5740            })?;
5741        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
5742        Some(handoff)
5743    };
5744    #[cfg(not(unix))]
5745    let handoff = None;
5746    // Windows keeps the environment copy: std cannot restrict an inherited pipe
5747    // handle to this child without leaking it to concurrently spawned processes.
5748    #[cfg(not(unix))]
5749    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
5750    Ok(handoff)
5751}
5752
5753fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
5754    command.env_remove(CK_LOG_ENV);
5755    // The spawn role is the supervisor's to set, and only on a swap candidate
5756    // (see `apply_spawn_role`). Removing it here, rather than just not setting
5757    // it, is what makes it absent on a plain spawn: the daemon's own
5758    // environment could carry it, and so could a spec built outside daemon
5759    // config (config refuses it as an `env` key). A module reading it on a
5760    // plain restart would pick the long swap budget and leave callers waiting.
5761    command.env_remove(SUBC_SPAWN_ROLE_ENV);
5762    for (key, value) in &spec.env {
5763        // cortexkit-log currently exposes retention only as a Rust struct, not
5764        // environment names. These values are daemon-private sink metadata and
5765        // must never become a public child-process contract by being inherited.
5766        if matches!(
5767            key.as_str(),
5768            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
5769        ) || key == SUBC_SPAWN_ROLE_ENV
5770        {
5771            continue;
5772        }
5773        command.env(key, value);
5774    }
5775}
5776
5777/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
5778/// of a blue/green swap.
5779#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5780enum SpawnRole {
5781    Plain,
5782    SwapCandidate,
5783}
5784
5785/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
5786/// `apply_child_env` has already removed the variable for every spawn.
5787fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
5788    if role == SpawnRole::SwapCandidate {
5789        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
5790    }
5791}
5792
5793fn spawn_child(
5794    spec: &ModuleSpec,
5795    connection_file_path: Option<&std::path::Path>,
5796    handle: Option<&SupervisorHandle>,
5797    ring: &Arc<Mutex<StderrRing>>,
5798    capture_logs_dir: Option<&std::path::Path>,
5799    roster: &ChildRoster,
5800    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5801) -> Result<SupervisedChild, SuperviseError> {
5802    spawn_child_in_slot(
5803        spec,
5804        connection_file_path,
5805        handle,
5806        ring,
5807        capture_logs_dir,
5808        roster,
5809        #[cfg(target_os = "linux")]
5810        cgroup_placement,
5811        SpawnRole::Plain,
5812        false,
5813    )
5814}
5815
5816/// Spawn one process of `spec` into a slot.
5817///
5818/// `alternate_slot` picks the process's cgroup name (see `swap::cgroup_name`).
5819/// A swap candidate needs a different cgroup from the process it is replacing,
5820/// which is still alive: in the same cgroup the two would be one kill domain,
5821/// and killing a failed candidate could take the incumbent with it.
5822///
5823/// The stderr capture file is `<module_id>.stderr.log` for every process of
5824/// the module, whichever slot it is in, because that is the one file
5825/// `ck module logs` reads. During a swap's overlap both processes append to it;
5826/// the daemon writes whole lines, so the two interleave by line, which is also
5827/// the merged view an operator wants while a swap runs.
5828#[allow(clippy::too_many_arguments)]
5829fn spawn_child_in_slot(
5830    spec: &ModuleSpec,
5831    connection_file_path: Option<&std::path::Path>,
5832    handle: Option<&SupervisorHandle>,
5833    ring: &Arc<Mutex<StderrRing>>,
5834    capture_logs_dir: Option<&std::path::Path>,
5835    roster: &ChildRoster,
5836    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5837    role: SpawnRole,
5838    alternate_slot: bool,
5839) -> Result<SupervisedChild, SuperviseError> {
5840    if roster.is_closed() {
5841        return Err(SuperviseError::Spawn {
5842            program: spec.program.clone(),
5843            source: io::Error::other("the daemon is shutting down; not starting a new process"),
5844            cgroup_path: None,
5845        });
5846    }
5847    #[cfg(target_os = "linux")]
5848    let cgroup_name = swap::cgroup_name(&spec.module_id, alternate_slot);
5849    #[cfg(not(target_os = "linux"))]
5850    let _ = alternate_slot;
5851    let mut command = Command::new(&spec.program);
5852    command.args(&spec.args);
5853    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
5854    // that is the whole of the intent, so remove that one key rather than the
5855    // environment.
5856    //
5857    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
5858    // and took the POSIX environment with it. Modules spawned that way had no
5859    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
5860    // logging:
5861    //
5862    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
5863    //     both unset it fell back to the temp dir alone and `ck` could not find
5864    //     a daemon running on the same machine from inside any module's process
5865    //     tree — reporting a path the file has never lived at, which reads as
5866    //     "the daemon did not write its file".
5867    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
5868    //     the RELATIVE `.local/share`, so a module deriving its own store path
5869    //     resolved it against its own CWD. That is the store-fragmentation
5870    //     defect the daemon already refuses in config (`parse_doc` rejects a
5871    //     relative `storage.data_home`) arriving by derivation instead.
5872    //   * anything a module spawns inherited it: git without ~/.gitconfig,
5873    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
5874    //     quietly rather than erroring.
5875    //
5876    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
5877    // offered one candidate under /tmp while the file sat in /run/user/1000.
5878    //
5879    // A configured module is unaffected either way: `module_spec()` puts the
5880    // resolved CK_LOG into `spec.env`, which is applied below and therefore
5881    // wins over anything ambient.
5882    apply_child_env(&mut command, spec);
5883    apply_spawn_role(&mut command, role);
5884    let nonce_handoff =
5885        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
5886
5887    #[cfg(target_os = "linux")]
5888    let cgroup_path = cgroup_placement
5889        .map(|placement| placement.module_path(&cgroup_name))
5890        .transpose()
5891        .map_err(|source| SuperviseError::Cgroup {
5892            module_id: spec.module_id.clone(),
5893            source,
5894        })?;
5895    #[cfg(not(target_os = "linux"))]
5896    let cgroup_path: Option<PathBuf> = None;
5897    #[cfg(target_os = "linux")]
5898    if let Some(path) = &cgroup_path {
5899        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
5900            if let Some(placement) = cgroup_placement {
5901                remove_module_cgroup(placement, &cgroup_name);
5902            }
5903            return Err(error);
5904        }
5905    }
5906
5907    let output_sink = if let Some(logs_dir) = capture_logs_dir {
5908        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
5909        match ChildOutputSink::open(&path, capture_retention(spec)) {
5910            Ok(sink) => sink,
5911            Err(error) => {
5912                warn!(
5913                    module_id = %spec.module_id,
5914                    path = %path.display(),
5915                    error = %error,
5916                    "could not open child output capture file; forwarding to stderr"
5917                );
5918                ChildOutputSink::Stderr
5919            }
5920        }
5921    } else {
5922        ChildOutputSink::Stderr
5923    };
5924
5925    command.stdout(Stdio::piped());
5926    command.stderr(Stdio::piped());
5927    command.kill_on_drop(true);
5928    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
5929    // before exec). In the daemon's group, a service manager that kills the
5930    // job's process group when the daemon exits (launchd's default) killed
5931    // every module at the same moment its control connection closed, so no
5932    // module ever ran its EOF teardown on a daemon stop. Outside that group a
5933    // module is reached only by the daemon: the EOF it sees when its
5934    // connection closes, and the bounded stop in `child_roster` for anything
5935    // still running after that. On Linux this composes with the cgroup
5936    // placement above: that is a pre_exec write to cgroup.procs, std performs
5937    // setpgid in the child before running pre_exec callbacks, and the two
5938    // change independent process attributes.
5939    //
5940    // stdin is /dev/null because a process outside the terminal's foreground
5941    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
5942    // by hand would otherwise hand down. Under a service manager stdin is
5943    // already /dev/null.
5944    #[cfg(unix)]
5945    command.process_group(0);
5946    command.stdin(Stdio::null());
5947    // The LAST pre-exec step, after the cgroup placement above: installing the
5948    // nonce at descriptor 3 replaces whatever the child had there, which could
5949    // be the descriptor an earlier step writes through.
5950    #[cfg(unix)]
5951    if let Some(handoff) = nonce_handoff {
5952        handoff.install_last(command.as_std_mut());
5953    }
5954    #[cfg(not(unix))]
5955    let _ = nonce_handoff;
5956
5957    // Containment, step 1 of 3 (issue #109): create the child suspended so it
5958    // cannot run a single instruction -- and therefore cannot spawn a
5959    // grandchild -- before it is in the job. See `contain_spawned_child` for the
5960    // other two steps and why the window matters.
5961    #[cfg(windows)]
5962    subc_jobobject::suspend_on_create_async(&mut command);
5963    let mut child = match command.spawn() {
5964        Ok(child) => child,
5965        Err(source) => {
5966            #[cfg(target_os = "linux")]
5967            if let Some(placement) = cgroup_placement {
5968                remove_module_cgroup(placement, &cgroup_name);
5969            }
5970            return Err(SuperviseError::Spawn {
5971                program: spec.program.clone(),
5972                source,
5973                cgroup_path,
5974            });
5975        }
5976    };
5977
5978    // Containment, steps 2 and 3: assign while suspended, then resume.
5979    #[cfg(windows)]
5980    let job = contain_spawned_child(&child, spec)?;
5981    let spawned_at_ms = unix_ms_now();
5982    let spawned_from = spec.program.clone();
5983    let spawned_file_identity = spawned_file_identity(&spawned_from);
5984    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
5985        program: spec.program.clone(),
5986        source: io::Error::other("spawned child exposed no live pid"),
5987        cgroup_path: cgroup_path.clone(),
5988    })?;
5989    let process_start_time = crate::provenance::process_start_time(pid);
5990    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
5991    // The executable identity is the spawned path's, read above, not the
5992    // running image's: right after spawn the child may not have finished its
5993    // exec yet and would still report this daemon's own image.
5994    #[cfg(target_os = "linux")]
5995    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
5996    #[cfg(not(target_os = "linux"))]
5997    let recorded_cgroup_name = None;
5998    let roster_guard = roster.admit(
5999        spec.module_id.clone(),
6000        pid,
6001        spec.protocol,
6002        process_start_time,
6003        crate::child_roster::RecordedIdentity {
6004            start_time: subc_os::start_time(pid),
6005            executable: spawned_file_identity.map(|identity| {
6006                crate::live_children::ExecutableIdentity {
6007                    device: identity.device,
6008                    inode: identity.inode,
6009                }
6010            }),
6011            cgroup_name: recorded_cgroup_name,
6012        },
6013    );
6014    // The check at the top of this function can pass just before daemon
6015    // shutdown begins, and the process is only in the roster from here on.
6016    // The shutdown stop returns as soon as it finds the roster empty, so a
6017    // process admitted after that look would outlive the daemon. The roster
6018    // is closed before the stop first reads it and admission happens under
6019    // the roster's lock, so either the stop sees this process or this check
6020    // sees the roster closed: end the process now rather than start a module
6021    // the daemon is about to stop.
6022    if roster.is_closed() {
6023        if let Err(error) = child.start_kill() {
6024            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
6025        }
6026        drop(roster_guard);
6027        return Err(SuperviseError::Spawn {
6028            program: spec.program.clone(),
6029            source: io::Error::other(
6030                "the daemon began shutting down while this process was starting; ended it",
6031            ),
6032            cgroup_path,
6033        });
6034    }
6035
6036    let stdout_pump = match child.stdout.take() {
6037        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
6038        None => {
6039            warn!(
6040                module_id = %spec.module_id,
6041                "spawned child exposed no stdout pipe; file capture will be incomplete"
6042            );
6043            None
6044        }
6045    };
6046    let stderr_pump = match child.stderr.take() {
6047        Some(stderr) => {
6048            let generation = ring
6049                .lock()
6050                .unwrap_or_else(|poisoned| poisoned.into_inner())
6051                .begin_process();
6052            Some(StderrPump {
6053                task: tokio::spawn(pump_stderr_to(
6054                    stderr,
6055                    Arc::clone(ring),
6056                    generation,
6057                    output_sink,
6058                )),
6059                generation,
6060            })
6061        }
6062        None => {
6063            // Spawning succeeded but the pipe did not materialise. Recording it as
6064            // uncaptured keeps the tail honest: the alternative is an empty tail
6065            // that reads as a module which printed nothing.
6066            ring.lock()
6067                .unwrap_or_else(|poisoned| poisoned.into_inner())
6068                .mark_not_captured("stderr pipe was not available on spawn");
6069            warn!(
6070                module_id = %spec.module_id,
6071                "spawned child exposed no stderr pipe; tail will be unavailable"
6072            );
6073            None
6074        }
6075    };
6076
6077    Ok(SupervisedChild {
6078        child,
6079        #[cfg(target_os = "linux")]
6080        module_id: cgroup_name,
6081        #[cfg(target_os = "linux")]
6082        cgroup_placement: cgroup_placement.cloned(),
6083        #[cfg(windows)]
6084        job,
6085        stdout_pump,
6086        stderr_pump,
6087        stderr_ring: Arc::clone(ring),
6088        spawned_at_ms,
6089        spawned_from,
6090        spawned_file_identity,
6091        process_start_time,
6092        process_identity,
6093        pid,
6094        roster_guard: Some(roster_guard),
6095    })
6096}
6097
6098/// Contain a freshly spawned Windows child and start it.
6099///
6100/// Steps 2 and 3 of the suspended-create contract: the job is created and the
6101/// child assigned **while it is still suspended** (step 1 is
6102/// `suspend_on_create_async` at the spawn site), then the child is resumed.
6103///
6104/// A child that is never resumed hangs forever holding a pid, so a resume
6105/// failure kills the child and fails the spawn rather than returning a
6106/// `SupervisedChild` that can never run.
6107///
6108/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
6109/// it did before this existed, whereas refusing to start one would be a new
6110/// outage. It is logged at warn because it means a helper process could leak.
6111#[cfg(windows)]
6112fn contain_spawned_child(
6113    child: &Child,
6114    spec: &ModuleSpec,
6115) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
6116    let module_id = spec.module_id.as_str();
6117    let Some(pid) = child.id() else {
6118        // The child exited between spawn and here. Its tree, if it made one,
6119        // needs no containment: nothing is left to contain.
6120        warn!(
6121            module_id,
6122            "spawned child had already exited before containment; no job object attached"
6123        );
6124        return Ok(None);
6125    };
6126
6127    let job = match subc_jobobject::JobObject::new() {
6128        Ok(job) => job,
6129        Err(source) => {
6130            warn!(
6131                module_id,
6132                error = %source,
6133                "could not create a job object; this module's helper processes will not be \
6134                 reaped on teardown"
6135            );
6136            // Resume regardless: leaving the child suspended would turn a
6137            // containment gap into a hung module.
6138            resume_suspended_child(pid, spec)?;
6139            return Ok(None);
6140        }
6141    };
6142
6143    if let Err(source) = job.assign(child) {
6144        warn!(
6145            module_id,
6146            error = %source,
6147            "could not assign the child to its job object; this module's helper processes \
6148             will not be reaped on teardown"
6149        );
6150        resume_suspended_child(pid, spec)?;
6151        return Ok(None);
6152    }
6153
6154    resume_suspended_child(pid, spec)?;
6155    Ok(Some(job))
6156}
6157
6158/// Resume a suspended child, killing it if it cannot be started.
6159///
6160/// A suspended process holds a pid and does nothing, so there is no useful
6161/// state to return: the caller gets an error and the spawn fails.
6162#[cfg(windows)]
6163fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
6164    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
6165        // Kill it here rather than leaving a suspended process for the caller
6166        // to notice; `kill_on_drop` would eventually do this, but the module
6167        // would have been reported as running in between.
6168        let _ = std::process::Command::new("taskkill.exe")
6169            .args(["/PID", &pid.to_string(), "/T", "/F"])
6170            .stdin(Stdio::null())
6171            .stdout(Stdio::null())
6172            .stderr(Stdio::null())
6173            .status();
6174        return Err(SuperviseError::Spawn {
6175            program: spec.program.clone(),
6176            source,
6177            cgroup_path: None,
6178        });
6179    }
6180    Ok(())
6181}
6182
6183#[cfg(target_os = "linux")]
6184fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
6185    match placement.remove_module(module_id) {
6186        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
6187        Err(error) => warn!(
6188            module_id,
6189            error = %error,
6190            "could not remove module cgroup after process exit; continuing teardown"
6191        ),
6192    }
6193}
6194
6195#[cfg(target_os = "linux")]
6196fn apply_cgroup_placement(
6197    command: &mut Command,
6198    spec: &ModuleSpec,
6199    path: &std::path::Path,
6200) -> Result<(), SuperviseError> {
6201    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
6202        module_id: spec.module_id.clone(),
6203        source,
6204    })
6205}
6206
6207fn capture_retention(spec: &ModuleSpec) -> Retention {
6208    let defaults = Retention::default();
6209    let value = |name: &str| {
6210        spec.env
6211            .iter()
6212            .rev()
6213            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
6214    };
6215    Retention {
6216        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
6217            .and_then(|value| value.parse().ok())
6218            .unwrap_or(defaults.max_file_mb),
6219        keep: value(CAPTURE_KEEP_ENV)
6220            .and_then(|value| value.parse().ok())
6221            .unwrap_or(defaults.keep),
6222        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
6223            .and_then(|value| value.parse().ok())
6224            .unwrap_or(defaults.max_age_days),
6225    }
6226}
6227
6228/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
6229/// module's registration to the exact process the supervisor spawned.
6230fn generate_launch_nonce() -> Result<String, SuperviseError> {
6231    let mut bytes = [0u8; 32];
6232    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
6233        reason: source.to_string(),
6234    })?;
6235    let mut hex = String::with_capacity(64);
6236    for b in bytes {
6237        use std::fmt::Write;
6238        let _ = write!(hex, "{b:02x}");
6239    }
6240    Ok(hex)
6241}
6242
6243/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
6244/// signal about how many leading bytes matched.
6245fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
6246    if a.len() != b.len() {
6247        return false;
6248    }
6249    let mut diff = 0u8;
6250    for (x, y) in a.iter().zip(b.iter()) {
6251        diff |= x ^ y;
6252    }
6253    diff == 0
6254}
6255
6256fn spawn_and_mark_running(
6257    spec: &ModuleSpec,
6258    runtime: &SupervisorRuntimeConfig,
6259    snapshot: &SharedSnapshot,
6260) -> Result<SupervisedChild, SuperviseError> {
6261    let child = spawn_child(
6262        spec,
6263        runtime.connection_file_path.as_deref(),
6264        runtime.supervisor_handle.as_ref(),
6265        &runtime.stderr_ring,
6266        runtime.capture_logs_dir.as_deref(),
6267        &runtime.child_roster,
6268        #[cfg(target_os = "linux")]
6269        runtime.cgroup_placement.as_ref(),
6270    )?;
6271    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
6272    Ok(child)
6273}
6274
6275enum RegistrationWaitOutcome {
6276    Registered,
6277    Exited(ExitReport),
6278    TimedOut,
6279}
6280
6281struct ReloadRegistrationFailure {
6282    exit_report: ExitReport,
6283    reason: String,
6284}
6285
6286#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6287enum BusyGaugeObservation {
6288    Quiescent,
6289    Busy,
6290    Omitted,
6291}
6292
6293fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
6294    let Some(metrics) = metrics.and_then(Value::as_object) else {
6295        return BusyGaugeObservation::Omitted;
6296    };
6297    let mut sum = 0u128;
6298    for gauge in gauges {
6299        let Some(value) = metrics.get(gauge) else {
6300            return BusyGaugeObservation::Omitted;
6301        };
6302        let Some(value) = value.as_u64() else {
6303            return BusyGaugeObservation::Busy;
6304        };
6305        sum = sum.saturating_add(u128::from(value));
6306    }
6307    if sum == 0 {
6308        BusyGaugeObservation::Quiescent
6309    } else {
6310        BusyGaugeObservation::Busy
6311    }
6312}
6313
6314fn declared_busy_gauges(
6315    registry: &Registry,
6316    module_id: &str,
6317) -> Result<Vec<String>, SuperviseError> {
6318    busy_gauges_of(
6319        registry
6320            .get_module(module_id)
6321            .map_err(SuperviseError::Registry)?,
6322    )
6323}
6324
6325/// [`declared_busy_gauges`] for the registration a connection holds, in any
6326/// slot: after cutover the incumbent is no longer the id's active
6327/// registration, and its own manifest is the one that names its gauges.
6328fn declared_busy_gauges_for_connection(
6329    registry: &Registry,
6330    connection_id: ConnectionId,
6331) -> Result<Vec<String>, SuperviseError> {
6332    busy_gauges_of(
6333        registry
6334            .get_module_by_connection(connection_id)
6335            .map_err(SuperviseError::Registry)?,
6336    )
6337}
6338
6339fn busy_gauges_of(
6340    registration: Option<crate::registry::ModuleRegistration>,
6341) -> Result<Vec<String>, SuperviseError> {
6342    let Some(registration) = registration else {
6343        return Ok(Vec::new());
6344    };
6345    let Some(self_signals) = registration.manifest.self_signals else {
6346        return Ok(Vec::new());
6347    };
6348
6349    let mut gauges = Vec::new();
6350    for declaration in self_signals {
6351        if declaration.kind != SelfSignalKind::Busy {
6352            continue;
6353        }
6354        match declaration.anchored_to {
6355            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
6356                gauges.extend(declared)
6357            }
6358            _ => {
6359                // An invalid Busy anchor is fail-safe: the empty name cannot be
6360                // present in a conforming health report, so this drain stays busy.
6361                gauges.push(String::new());
6362            }
6363        }
6364    }
6365    Ok(gauges)
6366}
6367
6368/// Wait for `endpoint` to have nothing in flight and, when the module declares
6369/// busy gauges, for a health probe to report them quiet. The probe is addressed
6370/// by `scope`: a swap's superseded incumbent must be asked about its own
6371/// gauges, and by module id the probe would reach the promoted candidate.
6372async fn wait_for_forwarding_quiescence(
6373    forwarding: &ForwardingTable,
6374    module_id: &str,
6375    runtime: &SupervisorRuntimeConfig,
6376    endpoint: crate::ModuleEndpointId,
6377    deadline: Instant,
6378    busy_gauges: &[String],
6379    scope: DrainScope,
6380) -> Result<bool, SuperviseError> {
6381    let mut gauges_quiescent = busy_gauges.is_empty();
6382    let mut next_probe_at = Instant::now();
6383    let mut omission_counted = false;
6384
6385    loop {
6386        let now = Instant::now();
6387        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
6388            let report = match scope {
6389                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
6390                DrainScope::Endpoint(endpoint) => {
6391                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
6392                }
6393            };
6394            gauges_quiescent = match report {
6395                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
6396                    BusyGaugeObservation::Quiescent => true,
6397                    BusyGaugeObservation::Busy => false,
6398                    BusyGaugeObservation::Omitted => {
6399                        if !omission_counted {
6400                            forwarding
6401                                .counters()
6402                                .increment_drains_with_undeclared_gauge();
6403                            omission_counted = true;
6404                        }
6405                        false
6406                    }
6407                },
6408                Err(err) => {
6409                    warn!(
6410                        module_id,
6411                        error = %err,
6412                        "drain health.check did not produce declared busy gauges; treating module as busy"
6413                    );
6414                    false
6415                }
6416            };
6417            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
6418        }
6419
6420        let in_flight = forwarding
6421            .endpoint_in_flight_count(endpoint)
6422            .map_err(SuperviseError::Forwarding)?;
6423        if in_flight == 0 && gauges_quiescent {
6424            return Ok(true);
6425        }
6426
6427        let now = Instant::now();
6428        if now >= deadline {
6429            return Ok(false);
6430        }
6431        let mut wait = deadline
6432            .saturating_duration_since(now)
6433            .min(REGISTRY_RELEASE_POLL);
6434        if !busy_gauges.is_empty() {
6435            wait = wait.min(next_probe_at.saturating_duration_since(now));
6436        }
6437        sleep(wait).await;
6438    }
6439}
6440
6441/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
6442///
6443/// `Ok` is always honest and passed straight through -- the wait actually measured
6444/// in-flight state. `Err` means the wait produced no measurement at all (the
6445/// forwarding table's lock was poisoned), so `false` is reported as the one honest
6446/// constant: the drain did not complete. Never recomputed from route state, never a
6447/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
6448fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
6449    match wait_result {
6450        Ok(drained) => *drained,
6451        Err(_) => false,
6452    }
6453}
6454
6455fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
6456    for released in released_routes {
6457        let frame = match Frame::build_with_version(
6458            released.negotiated_ver,
6459            FrameType::Goodbye,
6460            control_flags(),
6461            released.channel,
6462            released.epoch,
6463            0,
6464            Vec::new(),
6465        ) {
6466            Ok(frame) => frame,
6467            Err(err) => {
6468                warn!(
6469                    route_channel = released.channel,
6470                    error = %err,
6471                    "failed to build supervisor drain route GOODBYE frame"
6472                );
6473                continue;
6474            }
6475        };
6476        if !released.close_on_delivery_failure() {
6477            crate::forwarding::send_module_route_goodbye(
6478                &forwarding.counters(),
6479                &released.sink,
6480                frame,
6481                released.module_id.as_deref(),
6482                "supervisor drain",
6483            );
6484            continue;
6485        }
6486        if let Err(err) = released.sink.try_send(frame) {
6487            warn!(
6488                target_connection_id = released.connection_id.get(),
6489                route_channel = released.channel,
6490                error = %err,
6491                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
6492            );
6493            let _ = forwarding.escalate_client_delivery_failure(
6494                released.connection_id,
6495                released.channel,
6496                released.epoch,
6497                CloseReason::new(
6498                    "route_goodbye_delivery_failed",
6499                    format!(
6500                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
6501                        released.channel
6502                    ),
6503                ),
6504                crate::forwarding::UndeliveredFrame {
6505                    module_id: released.module_id.as_deref(),
6506                    sink: &released.sink,
6507                },
6508            );
6509        }
6510    }
6511}
6512
6513fn send_module_draining(
6514    module_id: &str,
6515    reason: RouteCloseReason,
6516    deadline_ms: u64,
6517    target: &ModuleDrainTarget,
6518) {
6519    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
6520        reason,
6521        deadline_ms,
6522    }) {
6523        Ok(body) => body,
6524        Err(err) => {
6525            warn!(
6526                module_id,
6527                error = %err,
6528                "failed to encode module draining command"
6529            );
6530            return;
6531        }
6532    };
6533    let frame = match Frame::build_with_version(
6534        target.negotiated_ver,
6535        FrameType::Push,
6536        control_flags(),
6537        0,
6538        0,
6539        0,
6540        body,
6541    ) {
6542        Ok(frame) => frame,
6543        Err(err) => {
6544            warn!(
6545                module_id,
6546                error = %err,
6547                "failed to build module draining command frame"
6548            );
6549            return;
6550        }
6551    };
6552    if let Err(err) = target.sink.try_send(frame) {
6553        warn!(
6554            module_id,
6555            target_connection_id = target.endpoint.connection_id.get(),
6556            error = %err,
6557            "module draining command was not delivered to peer"
6558        );
6559    }
6560}
6561
6562/// The channel-0 GOODBYE that tells a module its stop is planned.
6563fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
6564    match Frame::build_with_version(
6565        negotiated_ver,
6566        FrameType::Goodbye,
6567        control_flags(),
6568        0,
6569        0,
6570        0,
6571        Vec::new(),
6572    ) {
6573        Ok(frame) => Some(frame),
6574        Err(err) => {
6575            warn!(
6576                module_id,
6577                error = %err,
6578                "failed to build module GOODBYE frame"
6579            );
6580            None
6581        }
6582    }
6583}
6584
6585/// Send every registered module connection its module GOODBYE at daemon
6586/// shutdown, then request that connection's close.
6587///
6588/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
6589/// before EOF, so the GOODBYE must reach the socket before the close. A close
6590/// request does not wait for the connection's queued frames: its writer gets a
6591/// bounded grace after the close, is aborted if it overruns it, and the daemon
6592/// process may exit before that grace ends. So with `wait_for_flush`, each
6593/// connection is closed only after its writer has acknowledged writing the
6594/// GOODBYE, or once a short shared budget runs out, so one module that is not
6595/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
6596/// are only queued, for a shutdown the operator has told to stop waiting.
6597/// A connection that is already gone is skipped.
6598#[cfg(unix)]
6599async fn send_module_goodbyes_for_daemon_shutdown(
6600    forwarding: &Arc<ForwardingTable>,
6601    reason: &CloseReason,
6602    wait_for_flush: bool,
6603) {
6604    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
6605    let targets = match forwarding.module_connections() {
6606        Ok(targets) => targets,
6607        Err(err) => {
6608            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
6609            return;
6610        }
6611    };
6612    let deadline = Instant::now() + GOODBYE_BUDGET;
6613    let mut sends = tokio::task::JoinSet::new();
6614    for target in targets {
6615        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
6616            continue;
6617        };
6618        if !wait_for_flush {
6619            if let Err(err) = target.sink.try_send(frame) {
6620                debug!(
6621                    module_id = %target.module_id,
6622                    error = %err,
6623                    "shutdown module GOODBYE was not queued"
6624                );
6625            }
6626            continue;
6627        }
6628        let forwarding = Arc::clone(forwarding);
6629        let reason = reason.clone();
6630        sends.spawn(async move {
6631            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
6632                Ok(Ok(())) => {}
6633                Ok(Err(err)) => debug!(
6634                    module_id = %target.module_id,
6635                    error = %err,
6636                    "module connection closed before its shutdown GOODBYE was written"
6637                ),
6638                Err(_) => warn!(
6639                    module_id = %target.module_id,
6640                    budget = ?GOODBYE_BUDGET,
6641                    "shutdown module GOODBYE was not written within its budget; closing anyway"
6642                ),
6643            }
6644            forwarding.request_connection_close(target.endpoint.connection_id, reason);
6645        });
6646    }
6647    // Every task ends by the shared deadline, so this wait is bounded too.
6648    while sends.join_next().await.is_some() {}
6649}
6650
6651fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
6652    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
6653        return;
6654    };
6655    if let Err(err) = target.sink.try_send(frame) {
6656        warn!(
6657            module_id,
6658            target_connection_id = target.endpoint.connection_id.get(),
6659            error = %err,
6660            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
6661        );
6662        forwarding.request_connection_close(
6663            target.endpoint.connection_id,
6664            CloseReason::new(
6665                "module_goodbye_delivery_failed",
6666                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
6667            ),
6668        );
6669    }
6670}
6671
6672#[derive(Clone, Copy)]
6673struct ForwardingDrainContext<'a> {
6674    spec: &'a ModuleSpec,
6675    runtime: &'a SupervisorRuntimeConfig,
6676    registry: &'a Registry,
6677    scope: DrainScope,
6678}
6679
6680/// Which process a forwarding drain addresses.
6681#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6682enum DrainScope {
6683    /// Whatever endpoint is active for the module id: every plain stop,
6684    /// restart and reload. Also moves the module's state to `Draining`.
6685    Active,
6686    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
6687    /// module id would resolve to the promoted candidate and leave neither
6688    /// process routable. The module's state is left alone, since the promoted
6689    /// candidate is what it describes and that process is running.
6690    Endpoint(crate::ModuleEndpointId),
6691}
6692
6693/// Whether a child being drained has already been asked to stop by the time
6694/// its drain wait starts.
6695///
6696/// The drain wait is the same budget whatever this says. What it decides is
6697/// whether the supervisor must ask by signal before that wait begins: a child
6698/// that nobody asked will sit out the whole budget and then be SIGKILLed,
6699/// healthy or not.
6700#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6701enum StopNotice {
6702    /// The module was sent `module.draining` and a module GOODBYE over its own
6703    /// registered connection, and stops itself.
6704    SentOverConnection,
6705    /// The forwarding drain found no registered connection for the module: a
6706    /// subc child spawned moments ago that has not sent HELLO yet, or a
6707    /// `protocol: "none"` child, which never registers.
6708    NoConnection,
6709    /// This path sends nothing over the module's connection: the supervisor has
6710    /// no forwarding table, or the caller stops the child without a forwarding
6711    /// drain.
6712    NotSent,
6713}
6714
6715async fn begin_forwarding_drain(
6716    spec: &ModuleSpec,
6717    runtime: &SupervisorRuntimeConfig,
6718    registry: &Registry,
6719    snapshot: &SharedSnapshot,
6720    enabled: Option<bool>,
6721    reason: RouteCloseReason,
6722) -> Result<StopNotice, SuperviseError> {
6723    let Some(forwarding) = runtime.forwarding.as_ref() else {
6724        return Err(SuperviseError::ReloadUnavailable {
6725            module_id: spec.module_id.clone(),
6726            reason: "supervisor was not configured with a forwarding table".to_string(),
6727        });
6728    };
6729
6730    begin_forwarding_drain_with(
6731        forwarding,
6732        ForwardingDrainContext {
6733            spec,
6734            runtime,
6735            registry,
6736            scope: DrainScope::Active,
6737        },
6738        snapshot,
6739        enabled,
6740        reason,
6741        runtime.drain_timeout,
6742    )
6743    .await
6744}
6745
6746async fn begin_forwarding_drain_if_configured(
6747    spec: &ModuleSpec,
6748    runtime: &SupervisorRuntimeConfig,
6749    registry: &Registry,
6750    snapshot: &SharedSnapshot,
6751    enabled: Option<bool>,
6752    reason: RouteCloseReason,
6753) -> Result<StopNotice, SuperviseError> {
6754    begin_forwarding_drain_with_timeout(
6755        spec,
6756        runtime,
6757        registry,
6758        snapshot,
6759        enabled,
6760        reason,
6761        runtime.drain_timeout,
6762    )
6763    .await
6764}
6765
6766/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
6767/// budget, for paths where the operator overrides the module's configured one
6768/// (`supervisor.restart{drain_timeout_ms}`).
6769async fn begin_forwarding_drain_with_timeout(
6770    spec: &ModuleSpec,
6771    runtime: &SupervisorRuntimeConfig,
6772    registry: &Registry,
6773    snapshot: &SharedSnapshot,
6774    enabled: Option<bool>,
6775    reason: RouteCloseReason,
6776    drain_timeout: Duration,
6777) -> Result<StopNotice, SuperviseError> {
6778    let Some(forwarding) = runtime.forwarding.as_ref() else {
6779        return Ok(StopNotice::NotSent);
6780    };
6781
6782    begin_forwarding_drain_with(
6783        forwarding,
6784        ForwardingDrainContext {
6785            spec,
6786            runtime,
6787            registry,
6788            scope: DrainScope::Active,
6789        },
6790        snapshot,
6791        enabled,
6792        reason,
6793        drain_timeout,
6794    )
6795    .await
6796}
6797
6798async fn begin_forwarding_drain_with(
6799    forwarding: &ForwardingTable,
6800    context: ForwardingDrainContext<'_>,
6801    snapshot: &SharedSnapshot,
6802    enabled: Option<bool>,
6803    reason: RouteCloseReason,
6804    drain_timeout: Duration,
6805) -> Result<StopNotice, SuperviseError> {
6806    let ForwardingDrainContext {
6807        spec,
6808        runtime,
6809        registry,
6810        scope,
6811    } = context;
6812    debug_assert_ne!(reason, RouteCloseReason::Crash);
6813    let terminal = matches!(reason, RouteCloseReason::Disable);
6814    let drain_started_at = Instant::now();
6815    let drain_deadline = drain_started_at + drain_timeout;
6816    let deadline_ms =
6817        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
6818    let busy_gauges = match scope {
6819        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
6820        DrainScope::Endpoint(endpoint) => {
6821            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
6822        }
6823    };
6824
6825    // Admission gate first: route.open/commit and route REQUEST admission are closed
6826    // before the first quiescence check, so the outstanding count can only fall.
6827    let gate_started = Instant::now();
6828    let drain_target = match scope {
6829        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
6830        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
6831    }
6832    .map_err(SuperviseError::Forwarding)?;
6833    // The instant admission closed, and how long taking the forwarding write
6834    // lock to close it took. The timeout line reports only the quiescence
6835    // wait, so without this a drain that started late looked like one that
6836    // started on time.
6837    info!(
6838        module_id = %spec.module_id,
6839        ?reason,
6840        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
6841        connected = drain_target.is_some(),
6842        "module drain began; route admission closed"
6843    );
6844    if scope == DrainScope::Active {
6845        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6846            state.state = ModuleState::Draining;
6847            state.draining_to_replace =
6848                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
6849            if let Some(enabled) = enabled {
6850                state.enabled = enabled;
6851            }
6852        })?;
6853    }
6854
6855    let Some(target) = drain_target.as_ref() else {
6856        // Nothing was sent: the module has no registered connection to carry
6857        // `module.draining` or a GOODBYE. The caller must not assume the child
6858        // was asked to stop.
6859        return Ok(StopNotice::NoConnection);
6860    };
6861    {
6862        send_module_draining(&spec.module_id, reason, deadline_ms, target);
6863        let routes = forwarding
6864            .endpoint_routes(target.endpoint)
6865            .map_err(SuperviseError::Forwarding)?;
6866        let routes_notified = routes.len();
6867        crate::control::send_route_control_pushes(
6868            forwarding,
6869            routes.clone(),
6870            ClientControlPush::RouteClosing {
6871                module_id: spec.module_id.clone(),
6872                channels: Vec::new(),
6873                reason,
6874            },
6875        );
6876        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
6877
6878        // `route.closing` was just sent above: from here on every return path,
6879        // including an early one, MUST send `route.closed` before propagating
6880        // anything else. A client holds `closing` as a promise that a verdict is
6881        // coming; leaving early without `closed` strands it waiting forever, since
6882        // `closing` carries no timeout of its own.
6883        let wait_result = wait_for_forwarding_quiescence(
6884            forwarding,
6885            &spec.module_id,
6886            runtime,
6887            target.endpoint,
6888            drain_deadline,
6889            &busy_gauges,
6890            scope,
6891        )
6892        .await;
6893        let drained = drained_after_quiescence_wait(&wait_result);
6894        if let Err(err) = &wait_result {
6895            error!(
6896                module_id = %spec.module_id,
6897                ?reason,
6898                error = %err,
6899                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
6900            );
6901        } else if !drained {
6902            // Name what the drain waited on. Without it the line says only that
6903            // something did not settle, and "one wedged call" and "every
6904            // session's held stream" read the same; the first is a module bug,
6905            // the second is a module that should end its streams on
6906            // module.draining. Read before teardown releases the routes.
6907            let holdouts = forwarding
6908                .endpoint_drain_holdouts(target.endpoint)
6909                .unwrap_or_default();
6910            warn!(
6911                module_id = %spec.module_id,
6912                waited = ?drain_timeout,
6913                ?reason,
6914                held_requests = holdouts.requests,
6915                held_routes = holdouts.routes,
6916                total_routes = holdouts.total_routes,
6917                top_connections = ?holdouts.top_connections,
6918                // `module_channel:corr`, so the module can find each held request
6919                // in its own log; capped, so `held_requests` is the full count.
6920                held = %holdouts
6921                    .held
6922                    .iter()
6923                    .map(|(channel, corr)| format!("{channel}:{corr}"))
6924                    .collect::<Vec<_>>()
6925                    .join(","),
6926                "route drain timed out before request quiescence; forcing teardown"
6927            );
6928        }
6929        crate::control::send_route_control_pushes(
6930            forwarding,
6931            routes,
6932            ClientControlPush::RouteClosed {
6933                module_id: spec.module_id.clone(),
6934                channels: Vec::new(),
6935                reason,
6936                drained,
6937                abandoned: target.abandoned_bindings.len() as u32,
6938                excluded_subscriptions: target.excluded_subscriptions,
6939                terminal: Some(terminal),
6940            },
6941        );
6942        wait_result?;
6943
6944        // `route.closed` has now been sent unconditionally above. From here the
6945        // remaining steps are cleanup (route + module GOODBYE) rather than a
6946        // promise the client is waiting on, but a lock-poisoned
6947        // `release_module_endpoint_routes` would otherwise skip the module
6948        // GOODBYE silently too -- send it before propagating the error.
6949        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
6950            Ok(routes) => routes,
6951            Err(err) => {
6952                warn!(
6953                    module_id = %spec.module_id,
6954                    ?reason,
6955                    error = %err,
6956                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
6957                );
6958                send_module_goodbye(&spec.module_id, forwarding, target);
6959                return Err(SuperviseError::Forwarding(err));
6960            }
6961        };
6962        let route_goodbye_count = released_routes.len();
6963        send_route_goodbyes(forwarding, released_routes);
6964        send_module_goodbye(&spec.module_id, forwarding, target);
6965
6966        // The drain's happy path was previously silent: every emission above is
6967        // best-effort with only its failure arm logged, so "were consumers told"
6968        // was unprovable from the daemon log (surfaced by a 30-minute consumer
6969        // hang where the open question was exactly whether teardown notice went
6970        // out). One summary line makes that class decidable in one grep.
6971        info!(
6972            module_id = %spec.module_id,
6973            ?reason,
6974            routes_notified,
6975            route_goodbyes = route_goodbye_count,
6976            abandoned_reservations = target.abandoned_bindings.len(),
6977            excluded_subscriptions = target.excluded_subscriptions,
6978            drained,
6979            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
6980        );
6981    }
6982
6983    Ok(StopNotice::SentOverConnection)
6984}
6985
6986/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
6987/// the only slot a plain (non-swap) spawn can register into.
6988async fn wait_for_registration_after_reload(
6989    registry: &Registry,
6990    module_id: &str,
6991    snapshot: &SharedSnapshot,
6992    child: &mut SupervisedChild,
6993    wait: Duration,
6994) -> Result<RegistrationWaitOutcome, SuperviseError> {
6995    wait_for_slot_registration(
6996        registry,
6997        crate::registry::RegistrationSlot::Active(module_id),
6998        module_id,
6999        snapshot,
7000        child,
7001        wait,
7002    )
7003    .await
7004}
7005
7006/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
7007///
7008/// Keyed on the slot rather than the bare module id because during a swap the
7009/// id's active slot is already held by the incumbent: an id-keyed wait would
7010/// report the incumbent's registration as the candidate's and a candidate that
7011/// never registers would look registered. A swap candidate waits on
7012/// `crate::registry::RegistrationSlot::Candidate`.
7013async fn wait_for_slot_registration(
7014    registry: &Registry,
7015    slot: crate::registry::RegistrationSlot<'_>,
7016    module_id: &str,
7017    snapshot: &SharedSnapshot,
7018    child: &mut SupervisedChild,
7019    wait: Duration,
7020) -> Result<RegistrationWaitOutcome, SuperviseError> {
7021    let deadline = Instant::now() + wait;
7022    loop {
7023        if registry
7024            .registration(slot)
7025            .map_err(SuperviseError::Registry)?
7026            .is_some()
7027        {
7028            return Ok(RegistrationWaitOutcome::Registered);
7029        }
7030
7031        let now = Instant::now();
7032        if now >= deadline {
7033            return Ok(RegistrationWaitOutcome::TimedOut);
7034        }
7035        let remaining = deadline.saturating_duration_since(now);
7036        let poll = remaining.min(REGISTRY_RELEASE_POLL);
7037
7038        tokio::select! {
7039            wait_result = child.wait() => {
7040                let status = wait_result.map_err(|source| SuperviseError::Wait {
7041                    module_id: module_id.to_string(),
7042                    source,
7043                })?;
7044                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
7045                    snapshot,
7046                    child,
7047                    &status,
7048                )));
7049            }
7050            _ = sleep(poll) => {}
7051        }
7052    }
7053}
7054
7055fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
7056    // A replacement process that exits before HELLO did not provide service, even
7057    // if it used status 0. Count it against the restart cap as a new-binary failure.
7058    if exit_report.kind != ExitKind::DeliberateSeverance {
7059        exit_report.kind = ExitKind::Crash;
7060    }
7061    exit_report
7062}
7063
7064async fn handle_reload_child_registration_failure(
7065    spec: &ModuleSpec,
7066    runtime: &SupervisorRuntimeConfig,
7067    registry: &Registry,
7068    process_liveness: &SupervisorProcessLiveness,
7069    snapshot: &SharedSnapshot,
7070    child: &mut Option<SupervisedChild>,
7071    failure: ReloadRegistrationFailure,
7072) -> Result<(), SuperviseError> {
7073    let ReloadRegistrationFailure {
7074        exit_report,
7075        reason,
7076    } = failure;
7077    match on_child_exit(
7078        spec,
7079        runtime.restart_policy,
7080        registry,
7081        snapshot,
7082        &runtime.terminal_ring,
7083        &runtime.spawn_events,
7084        &runtime.child_roster,
7085        exit_report,
7086    )
7087    .await
7088    {
7089        NextAction::Stop {
7090            registration_released,
7091        } => {
7092            if registration_released {
7093                process_liveness.untrack_if_current(&spec.module_id, snapshot);
7094            }
7095        }
7096        NextAction::Restart { schedule } => {
7097            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
7098                schedule.delay
7099            });
7100            if let Some(schedule) = schedule {
7101                log_crash_respawn(&spec.module_id, schedule);
7102            }
7103            sleep(delay).await;
7104            // A disable or drain that landed during the backoff cancels this
7105            // policy retry: the operator's stop must win over the respawn the
7106            // sleep counted down to.
7107            if respawn_still_pending(snapshot) {
7108                if let Err(err) = wait_for_registration_release(
7109                    registry,
7110                    &spec.module_id,
7111                    REGISTRY_RELEASE_TIMEOUT,
7112                )
7113                .await
7114                {
7115                    fail_snapshot(snapshot, Some(&spec.module_id), None);
7116                    process_liveness.untrack_if_current(&spec.module_id, snapshot);
7117                    return Err(SuperviseError::ReloadFailed {
7118                        module_id: spec.module_id.clone(),
7119                        reason: format!(
7120                            "{reason}; registration did not release before policy retry: {err}"
7121                        ),
7122                    });
7123                }
7124                process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7125                match spawn_and_mark_running(spec, runtime, snapshot) {
7126                    Ok(next_child) => {
7127                        *child = Some(next_child);
7128                    }
7129                    Err(err) => {
7130                        fail_snapshot(snapshot, Some(&spec.module_id), None);
7131                        process_liveness.untrack_if_current(&spec.module_id, snapshot);
7132                        return Err(SuperviseError::ReloadFailed {
7133                            module_id: spec.module_id.clone(),
7134                            reason: format!("{reason}; policy retry spawn failed: {err}"),
7135                        });
7136                    }
7137                }
7138            }
7139        }
7140    }
7141
7142    Err(SuperviseError::ReloadFailed {
7143        module_id: spec.module_id.clone(),
7144        reason,
7145    })
7146}
7147
7148async fn handle_reload_spawn_failure(
7149    spec: &ModuleSpec,
7150    runtime: &SupervisorRuntimeConfig,
7151    process_liveness: &SupervisorProcessLiveness,
7152    snapshot: &SharedSnapshot,
7153    child: &mut Option<SupervisedChild>,
7154    reason: String,
7155) -> Result<(), SuperviseError> {
7156    let mut should_retry = false;
7157    let now = Instant::now();
7158    update_snapshot(snapshot, Some(&spec.module_id), |state| {
7159        clear_current_process_facts(state);
7160        if daemon_will_restart(state, &runtime.restart_policy, now) {
7161            state.record_crash_restart(&runtime.restart_policy, now);
7162            state.state = ModuleState::Restarting;
7163            should_retry = true;
7164        } else if state.enabled {
7165            state.state = ModuleState::Failed;
7166        } else {
7167            state.state = ModuleState::Disabled;
7168        }
7169    })?;
7170
7171    if should_retry {
7172        sleep(runtime.restart_policy.backoff).await;
7173        // A disable or drain that landed during the backoff cancels this
7174        // policy retry: the operator's stop must win over the respawn the
7175        // sleep counted down to.
7176        if respawn_still_pending(snapshot) {
7177            process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7178            match spawn_and_mark_running(spec, runtime, snapshot) {
7179                Ok(next_child) => {
7180                    *child = Some(next_child);
7181                }
7182                Err(err) => {
7183                    fail_snapshot(snapshot, Some(&spec.module_id), None);
7184                    process_liveness.untrack_if_current(&spec.module_id, snapshot);
7185                    return Err(SuperviseError::ReloadFailed {
7186                        module_id: spec.module_id.clone(),
7187                        reason: format!("{reason}; policy retry spawn failed: {err}"),
7188                    });
7189                }
7190            }
7191        }
7192    } else {
7193        process_liveness.untrack_if_current(&spec.module_id, snapshot);
7194    }
7195
7196    Err(SuperviseError::ReloadFailed {
7197        module_id: spec.module_id.clone(),
7198        reason,
7199    })
7200}
7201
7202fn control_flags() -> Flags {
7203    Flags::new(false, Priority::Passive, false)
7204}
7205
7206#[allow(clippy::too_many_arguments)]
7207async fn drain_optional_child(
7208    module_id: &str,
7209    protocol: ModuleProtocol,
7210    stop_notice: StopNotice,
7211    registry: &Registry,
7212    snapshot: &SharedSnapshot,
7213    terminal_ring: &Arc<Mutex<TerminalRing>>,
7214    spawn_events: &SpawnEventFeed,
7215    child: &mut Option<SupervisedChild>,
7216    drain_timeout: Duration,
7217    final_state: ModuleState,
7218    enabled: Option<bool>,
7219) -> Result<(), SuperviseError> {
7220    if let Some(child) = child.take() {
7221        drain_child_to_state(
7222            module_id,
7223            protocol,
7224            stop_notice,
7225            registry,
7226            snapshot,
7227            terminal_ring,
7228            spawn_events,
7229            child,
7230            drain_timeout,
7231            final_state,
7232            enabled,
7233        )
7234        .await
7235    } else {
7236        update_snapshot(snapshot, Some(module_id), |state| {
7237            state.state = final_state;
7238            if let Some(enabled) = enabled {
7239                state.enabled = enabled;
7240            }
7241            clear_current_process_facts(state);
7242        })?;
7243        wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7244    }
7245}
7246
7247#[allow(clippy::too_many_arguments)]
7248async fn drain_child_to_state(
7249    module_id: &str,
7250    protocol: ModuleProtocol,
7251    stop_notice: StopNotice,
7252    registry: &Registry,
7253    snapshot: &SharedSnapshot,
7254    terminal_ring: &Arc<Mutex<TerminalRing>>,
7255    spawn_events: &SpawnEventFeed,
7256    mut child: SupervisedChild,
7257    drain_timeout: Duration,
7258    final_state: ModuleState,
7259    enabled: Option<bool>,
7260) -> Result<(), SuperviseError> {
7261    update_snapshot(snapshot, Some(module_id), |state| {
7262        state.state = ModuleState::Draining;
7263        state.draining_to_replace = final_state == ModuleState::Restarting;
7264        if let Some(enabled) = enabled {
7265            state.enabled = enabled;
7266        }
7267    })?;
7268
7269    // The wait below is the same budget in every case; what differs is
7270    // whether anything has ASKED the child to stop before it starts. Only a
7271    // forwarding drain that reached the module's registered connection has
7272    // (`module.draining`, then a module GOODBYE). Every other child was told
7273    // nothing: a `protocol: "none"` module, which never registers; a subc
7274    // module spawned moments ago that has not sent HELLO yet; or a stop that
7275    // runs no forwarding drain. Without a signal the budget is only a delay
7276    // in front of SIGKILL -- and the not-yet-registered child is the worst
7277    // case, because it registers into a module that is already draining,
7278    // is never told, and is killed while healthy.
7279    if stop_notice != StopNotice::SentOverConnection {
7280        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
7281            info!(
7282                module_id,
7283                pid = child.pid,
7284                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7285                "module has no connection yet; requesting stop by signal"
7286            );
7287        }
7288        request_graceful_stop(module_id, &child);
7289    }
7290
7291    let exit_report = match timeout(drain_timeout, child.wait()).await {
7292        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
7293        Ok(Err(source)) => {
7294            fail_snapshot(snapshot, Some(module_id), None);
7295            return Err(SuperviseError::Wait {
7296                module_id: module_id.to_string(),
7297                source,
7298            });
7299        }
7300        Err(_) => {
7301            // Mirror the sibling arm above: state is already `Draining`, and an
7302            // error propagated from here would strand it there -- a state
7303            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
7304            // `Failed | Stopped`), leaving an operator Restart as the only exit.
7305            // `Failed` before `?` keeps the module operator-visible and
7306            // revivable. Trigger is an ESRCH race (process exits between the
7307            // drain timeout firing and the kill) or a post-kill wait failure
7308            // (issue #34).
7309            //
7310            // Logged because the kill is otherwise visible only as signal 9 in
7311            // the terminal ring, and the budget it follows can be long enough
7312            // that consumers see a stretch of refusals with no stated cause.
7313            warn!(
7314                module_id,
7315                pid = child.pid,
7316                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7317                reason = ?final_state,
7318                ?stop_notice,
7319                "drain budget expired before the module exited; killing it"
7320            );
7321            child.start_kill().map_err(|source| {
7322                fail_snapshot(snapshot, Some(module_id), None);
7323                SuperviseError::Kill {
7324                    module_id: module_id.to_string(),
7325                    source,
7326                }
7327            })?;
7328            let status = child.wait().await.map_err(|source| {
7329                fail_snapshot(snapshot, Some(module_id), None);
7330                SuperviseError::Wait {
7331                    module_id: module_id.to_string(),
7332                    source,
7333                }
7334            })?;
7335            classify_reaped_child_exit(snapshot, &child, &status)
7336        }
7337    };
7338
7339    update_snapshot(snapshot, Some(module_id), |state| {
7340        state.state = final_state;
7341        if let Some(enabled) = enabled {
7342            state.enabled = enabled;
7343        }
7344        clear_current_process_facts(state);
7345        state.last_exit = Some(exit_report.clone());
7346        if exit_report.kind == ExitKind::DeliberateSeverance {
7347            state.lifetime_restarts += 1;
7348        }
7349    })?;
7350    record_terminal(
7351        module_id,
7352        terminal_ring,
7353        spawn_events,
7354        &exit_report,
7355        terminal_disposition(final_state),
7356    );
7357    child.drain_stderr(module_id).await;
7358
7359    wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7360}
7361
7362/// Ask a child that nothing else has asked to stop, by signal.
7363///
7364/// A registered subc module is asked over its own connection: the drain sends
7365/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
7366/// module GOODBYE, and the module stops itself. A module that speaks no subc
7367/// wire receives none of that, and neither does a subc module that has not
7368/// registered yet, so for them the drain budget would be pure delay in front of
7369/// a SIGKILL -- and for a process with a store to flush (JetStream is the
7370/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
7371/// into a recovery on the next start.
7372///
7373/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
7374/// rule rather than an optimisation: that module's graceful stop is already
7375/// running by the time its child is drained, and a signal would race it.
7376///
7377/// Best-effort by construction. A child that has already exited is the ordinary
7378/// case rather than an error (the kill lands on a reaped or exiting pid), so a
7379/// failure is logged at debug and the wait-then-kill below still decides the
7380/// outcome.
7381#[cfg(unix)]
7382fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
7383    let Some(pid) = child
7384        .id()
7385        .and_then(|pid| i32::try_from(pid).ok())
7386        .and_then(rustix::process::Pid::from_raw)
7387    else {
7388        debug!(
7389            module_id,
7390            "no pid to signal for teardown; falling through to the drain wait"
7391        );
7392        return;
7393    };
7394    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
7395        Ok(()) => debug!(
7396            module_id,
7397            "sent SIGTERM to a module nothing else asked to stop"
7398        ),
7399        Err(err) => debug!(
7400            module_id,
7401            error = %err,
7402            "SIGTERM to module failed; the drain wait and kill still apply"
7403        ),
7404    }
7405}
7406
7407/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
7408/// Windows does offer need cooperation this supervisor cannot assume: a console
7409/// control event requires sharing a console with the child, and `WM_CLOSE`
7410/// requires the child to pump a message loop. A supervised server process does
7411/// neither, so there is nothing to send and teardown is the wait followed by the
7412/// kill. Emulating a signal here would mean inventing a stop protocol, which is
7413/// the thing `protocol: "none"` exists to avoid.
7414#[cfg(not(unix))]
7415fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
7416    debug!(
7417        module_id,
7418        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
7419    );
7420}
7421
7422fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
7423    match final_state {
7424        ModuleState::Stopped => TerminalDisposition::Stopped,
7425        ModuleState::Disabled => TerminalDisposition::Disabled,
7426        ModuleState::Restarting => TerminalDisposition::Restarting,
7427        ModuleState::Failed => TerminalDisposition::Failed,
7428        ModuleState::Starting
7429        | ModuleState::Running
7430        | ModuleState::Unresponsive
7431        | ModuleState::Draining => {
7432            unreachable!("terminal exits only finish in terminal or restarting states")
7433        }
7434    }
7435}
7436
7437/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
7438/// plain stop or restart waits for before it spawns a replacement.
7439async fn wait_for_registration_release(
7440    registry: &Registry,
7441    module_id: &str,
7442    wait: Duration,
7443) -> Result<(), SuperviseError> {
7444    wait_for_slot_registration_release(
7445        registry,
7446        crate::registry::RegistrationSlot::Active(module_id),
7447        wait,
7448    )
7449    .await
7450}
7451
7452/// Wait for the registration in `slot` to go away.
7453///
7454/// Keyed on the slot rather than the bare module id because a successful swap
7455/// never empties the id's active slot (the promoted candidate is in it), so an
7456/// id-keyed wait for the incumbent's release would always time out. Draining a
7457/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
7458/// incumbent's connection instead.
7459async fn wait_for_slot_registration_release(
7460    registry: &Registry,
7461    slot: crate::registry::RegistrationSlot<'_>,
7462    wait: Duration,
7463) -> Result<(), SuperviseError> {
7464    let deadline = Instant::now() + wait;
7465    let mut release_events = registration_release_events().subscribe();
7466    let still_active = |registration: &crate::registry::ModuleRegistration| {
7467        SuperviseError::RegistrationStillActive {
7468            module_id: registration.manifest.module_id.clone(),
7469            waited: wait,
7470        }
7471    };
7472    loop {
7473        let _observed_generation = *release_events.borrow_and_update();
7474        let Some(registration) = registry
7475            .registration(slot)
7476            .map_err(SuperviseError::Registry)?
7477        else {
7478            return Ok(());
7479        };
7480
7481        let now = Instant::now();
7482        if now >= deadline {
7483            return Err(still_active(&registration));
7484        }
7485
7486        let remaining = deadline.saturating_duration_since(now);
7487        match timeout(remaining, release_events.changed()).await {
7488            Ok(Ok(())) | Ok(Err(_)) => {}
7489            Err(_) => return Err(still_active(&registration)),
7490        }
7491    }
7492}
7493
7494#[cfg(test)]
7495mod slot_registration_wait_tests {
7496    use super::*;
7497    use crate::registry::{ConnectionId, RegistrationSlot};
7498    use subc_protocol::manifest::ModuleManifest;
7499
7500    const INCUMBENT: u64 = 1;
7501    const CANDIDATE: u64 = 2;
7502
7503    fn swapped_registry() -> Arc<Registry> {
7504        let registry = Arc::new(Registry::default());
7505        let manifest = ModuleManifest::builder("m", "0.1.0").build();
7506        registry
7507            .register_with_control_ops(
7508                manifest.clone(),
7509                1,
7510                ConnectionId::new(INCUMBENT),
7511                Vec::new(),
7512            )
7513            .unwrap();
7514        registry
7515            .register_candidate_with_control_ops(
7516                manifest,
7517                1,
7518                ConnectionId::new(CANDIDATE),
7519                Vec::new(),
7520            )
7521            .unwrap();
7522        registry
7523    }
7524
7525    /// After a promotion the id's active slot is held by the new process, so an
7526    /// id-keyed wait for the incumbent's release can never succeed; the
7527    /// connection-keyed wait completes as soon as the incumbent deregisters.
7528    #[tokio::test]
7529    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
7530        let registry = swapped_registry();
7531        registry.promote_candidate("m").unwrap().unwrap();
7532
7533        assert!(matches!(
7534            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
7535            Err(SuperviseError::RegistrationStillActive { .. })
7536        ));
7537
7538        // Still held while the incumbent's connection has not deregistered.
7539        assert!(matches!(
7540            wait_for_slot_registration_release(
7541                &registry,
7542                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7543                Duration::from_millis(50),
7544            )
7545            .await,
7546            Err(SuperviseError::RegistrationStillActive { .. })
7547        ));
7548
7549        let releaser = Arc::clone(&registry);
7550        let release = tokio::spawn(async move {
7551            sleep(Duration::from_millis(20)).await;
7552            releaser
7553                .deregister_connection(ConnectionId::new(INCUMBENT))
7554                .unwrap();
7555            notify_registration_release();
7556        });
7557        wait_for_slot_registration_release(
7558            &registry,
7559            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7560            Duration::from_secs(5),
7561        )
7562        .await
7563        .expect("the incumbent's own registration is released");
7564        release.await.unwrap();
7565        assert!(registry.get_module("m").unwrap().is_some());
7566    }
7567
7568    /// The candidate slot is waited on separately from the active slot: the
7569    /// incumbent's registration neither holds up nor stands in for it.
7570    #[tokio::test]
7571    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
7572        let registry = swapped_registry();
7573        assert!(matches!(
7574            wait_for_slot_registration_release(
7575                &registry,
7576                RegistrationSlot::Candidate("m"),
7577                Duration::from_millis(50),
7578            )
7579            .await,
7580            Err(SuperviseError::RegistrationStillActive { .. })
7581        ));
7582        registry
7583            .deregister_connection(ConnectionId::new(CANDIDATE))
7584            .unwrap();
7585        wait_for_slot_registration_release(
7586            &registry,
7587            RegistrationSlot::Candidate("m"),
7588            Duration::from_millis(50),
7589        )
7590        .await
7591        .expect("a candidate slot with no candidate is released");
7592        assert!(registry
7593            .registration(RegistrationSlot::Active("m"))
7594            .unwrap()
7595            .is_some());
7596    }
7597}
7598
7599fn classify_exit(status: &ExitStatus) -> ExitReport {
7600    ExitReport {
7601        kind: if status.success() {
7602            ExitKind::Clean
7603        } else {
7604            ExitKind::Crash
7605        },
7606        code: status.code(),
7607        signal: exit_signal(status),
7608        at_ms: unix_ms_now(),
7609    }
7610}
7611
7612/// The terminal record for a module whose `wait()` call itself errored (e.g. the
7613/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
7614/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
7615/// disposition still must be `Failed` so the terminal ring is not silently missing
7616/// an entry, matching what `fail_snapshot` records for this same arm.
7617fn wait_error_exit_report() -> ExitReport {
7618    ExitReport {
7619        kind: ExitKind::Crash,
7620        code: None,
7621        signal: None,
7622        at_ms: unix_ms_now(),
7623    }
7624}
7625
7626#[cfg(unix)]
7627fn exit_signal(status: &ExitStatus) -> Option<i32> {
7628    use std::os::unix::process::ExitStatusExt;
7629
7630    status.signal()
7631}
7632
7633#[cfg(not(unix))]
7634fn exit_signal(_status: &ExitStatus) -> Option<i32> {
7635    None
7636}
7637
7638/// Give an operator-touched module its full crash budget back.
7639///
7640/// Named for the counter it used to zero; it now empties the in-window ring,
7641/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
7642/// ledger of what happened survives every operator action.
7643fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
7644    update_snapshot(snapshot, Some(module_id), |state| {
7645        state.clear_crash_restarts();
7646    })
7647}
7648
7649fn set_running(
7650    snapshot: &SharedSnapshot,
7651    child: &SupervisedChild,
7652    module_id: &str,
7653    spawn_events: &SpawnEventFeed,
7654) -> Result<(), SuperviseError> {
7655    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7656        module_id: Some(module_id.to_string()),
7657    })?;
7658    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
7659    // Every caller of this is a plain spawn, which always uses the primary key;
7660    // a promoted swap candidate sets the flag itself after this returns.
7661    state.in_alternate_slot = false;
7662    state.configuration_updated_since_spawn = false;
7663    state.state = ModuleState::Running;
7664    state.enabled = true;
7665    state.process_alive = true;
7666    state.pid = child.id();
7667    state.spawned_at_ms = Some(child.spawned_at_ms);
7668    state.spawned_from = Some(child.spawned_from.clone());
7669    state.spawned_file_identity = child.spawned_file_identity;
7670    state.process_start_time = child.process_start_time;
7671    Ok(())
7672}
7673
7674fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
7675    state.process_alive = false;
7676    state.pid = None;
7677    state.spawned_at_ms = None;
7678    state.spawned_from = None;
7679    state.spawned_file_identity = None;
7680    state.process_start_time = None;
7681    state.deliberate_severance = None;
7682}
7683
7684#[cfg(test)]
7685fn record_deliberate_severance(
7686    snapshot: &SharedSnapshot,
7687    identity: ProcessIdentity,
7688) -> Result<(), SuperviseError> {
7689    update_snapshot(snapshot, None, |state| {
7690        state.deliberate_severance = Some(identity);
7691    })
7692}
7693
7694fn apply_deliberate_severance_marker(
7695    snapshot: &SharedSnapshot,
7696    exited_identity: Option<ProcessIdentity>,
7697    mut exit_report: ExitReport,
7698) -> ExitReport {
7699    let marker = lock_snapshot(snapshot)
7700        .ok()
7701        .and_then(|mut state| state.deliberate_severance.take());
7702    if marker.is_some() && marker == exited_identity {
7703        exit_report.kind = ExitKind::DeliberateSeverance;
7704    }
7705    exit_report
7706}
7707
7708fn classify_reaped_child_exit(
7709    snapshot: &SharedSnapshot,
7710    child: &SupervisedChild,
7711    status: &ExitStatus,
7712) -> ExitReport {
7713    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
7714}
7715
7716fn fail_snapshot(
7717    snapshot: &SharedSnapshot,
7718    module_id: Option<&str>,
7719    last_exit: Option<ExitReport>,
7720) {
7721    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
7722        state.state = ModuleState::Failed;
7723        clear_current_process_facts(state);
7724        if let Some(last_exit) = last_exit {
7725            state.last_exit = Some(last_exit);
7726        }
7727    }) {
7728        error!(error = %err, "failed to mark supervisor state failed");
7729    }
7730}
7731
7732fn update_snapshot(
7733    snapshot: &SharedSnapshot,
7734    module_id: Option<&str>,
7735    update: impl FnOnce(&mut SupervisorSnapshot),
7736) -> Result<(), SuperviseError> {
7737    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7738        module_id: module_id.map(ToOwned::to_owned),
7739    })?;
7740    update(&mut state);
7741    Ok(())
7742}
7743
7744const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
7745
7746fn lock_snapshot_for_control<'a>(
7747    snapshot: &'a SharedSnapshot,
7748    module_id: &str,
7749    caller: &'static str,
7750) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
7751    let started_at = Instant::now();
7752    let guard = lock_snapshot(snapshot)?;
7753    let waited = started_at.elapsed();
7754    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
7755        warn!(
7756            module_id = %module_id,
7757            waited_ms = waited.as_millis() as u64,
7758            caller = %caller,
7759            "slow snapshot lock"
7760        );
7761    }
7762    Ok(guard)
7763}
7764
7765fn lock_snapshot(
7766    snapshot: &SharedSnapshot,
7767) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
7768    snapshot
7769        .lock()
7770        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
7771}
7772
7773#[cfg(test)]
7774mod terminal_history_tests {
7775    use std::{
7776        path::PathBuf,
7777        sync::Arc,
7778        time::{Duration, Instant},
7779    };
7780
7781    use tokio::time::sleep;
7782
7783    use super::{
7784        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
7785        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
7786        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
7787        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
7788        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
7789        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
7790        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
7791    };
7792    // The supervisor's clock, distinct from the `std::time::Instant` these tests
7793    // use for their own wall-clock deadlines: crash-restart instants must be on
7794    // the same clock the production code stamps them with, which is tokio's (and
7795    // is what `start_paused` tests can move).
7796    use super::Instant as ClockInstant;
7797    use crate::{
7798        registry::Registry,
7799        terminal_ring::{TerminalRing, TerminalRingConfig},
7800    };
7801    use std::sync::Mutex;
7802    use subc_control::TerminalDisposition;
7803
7804    /// See the twin in `control.rs` for why this derives the path from
7805    /// `current_exe()` and why the existence check is here: `--lib` alone does
7806    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
7807    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
7808    fn fake_aft_stub_path() -> PathBuf {
7809        let mut path = std::env::current_exe().expect("current_exe available in tests");
7810        path.pop();
7811        path.pop();
7812        path.push(if cfg!(windows) {
7813            "fake-aft-stub.exe"
7814        } else {
7815            "fake-aft-stub"
7816        });
7817        assert!(
7818            path.exists(),
7819            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
7820             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
7821            path.display()
7822        );
7823        path
7824    }
7825
7826    #[test]
7827    fn reserved_never_spawned_refuses_every_hello() {
7828        // The canary hole: a reserved id whose module has never spawned had NO
7829        // gate entry and admitted anyone -- the reservation protected the nonce
7830        // holder, not the NAME. Now the entry is present with no legitimate
7831        // holder and refuses all comers.
7832        let supervisor = SupervisorHandle::default();
7833        supervisor.apply_identity_configuration(&ModuleSpec {
7834            module_id: "never-spawned".to_string(),
7835            program: PathBuf::from("/usr/bin/false"),
7836            args: Vec::new(),
7837            env: Vec::new(),
7838            reserved: true,
7839            reserved_prefixes: Vec::new(),
7840            protocol: ModuleProtocol::Subc,
7841            overlap: Default::default(),
7842        });
7843        assert!(
7844            supervisor
7845                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
7846                .is_some(),
7847            "forged nonce must refuse on a reserved never-spawned id"
7848        );
7849        assert!(
7850            supervisor
7851                .reserved_hello_rejection("never-spawned", None)
7852                .is_some(),
7853            "absent nonce must refuse on a reserved never-spawned id"
7854        );
7855        // And a real spawn nonce minted later admits exactly that nonce.
7856        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
7857        supervisor.apply_identity_configuration(&ModuleSpec {
7858            module_id: "never-spawned".to_string(),
7859            program: PathBuf::from("/usr/bin/false"),
7860            args: Vec::new(),
7861            env: Vec::new(),
7862            reserved: true,
7863            reserved_prefixes: Vec::new(),
7864            protocol: ModuleProtocol::Subc,
7865            overlap: Default::default(),
7866        });
7867        assert!(supervisor
7868            .reserved_hello_rejection("never-spawned", Some("minted"))
7869            .is_none());
7870        assert!(supervisor
7871            .reserved_hello_rejection("never-spawned", Some("forged"))
7872            .is_some());
7873    }
7874
7875    /// Put `count` crash restarts on a snapshot's ring as if they had all just
7876    /// happened, which is what "spent budget" looks like to every reader.
7877    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
7878        let now = ClockInstant::now();
7879        for _ in 0..count {
7880            state.crash_restarts.push_back(now);
7881        }
7882    }
7883
7884    /// Age the oldest recorded restart out of `window`, standing in for the hours
7885    /// that would otherwise have to pass. Injecting the instant is the point: a
7886    /// test that slept a real window would take ten minutes and still prove less.
7887    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
7888        let aged = state
7889            .crash_restarts
7890            .front()
7891            .expect("a crash restart must be recorded before it can be aged")
7892            .checked_sub(window + Duration::from_secs(1))
7893            .expect("the test clock is far enough from its origin to age an instant");
7894        state.crash_restarts[0] = aged;
7895    }
7896
7897    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
7898        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
7899        seed_crash_restarts(&mut state, count);
7900        state
7901    }
7902
7903    #[test]
7904    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
7905        let policy = RestartPolicy::new(3, Duration::ZERO);
7906        let now = ClockInstant::now();
7907        assert!(daemon_will_restart(
7908            &mut snapshot_with_restarts(true, 2),
7909            &policy,
7910            now
7911        ));
7912        assert!(!daemon_will_restart(
7913            &mut snapshot_with_restarts(true, 3),
7914            &policy,
7915            now
7916        ));
7917        assert!(!daemon_will_restart(
7918            &mut snapshot_with_restarts(false, 0),
7919            &policy,
7920            now
7921        ));
7922    }
7923
7924    #[test]
7925    fn crash_restart_backoff_escalates_with_in_window_count() {
7926        let policy = RestartPolicy::new(4, Duration::from_millis(100))
7927            .with_max_backoff(Duration::from_secs(30));
7928        let now = ClockInstant::now();
7929        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7930        let schedules = (0..4)
7931            .map(|_| {
7932                state
7933                    .next_crash_restart(&policy, now)
7934                    .expect("the test policy allows four crash restarts")
7935            })
7936            .collect::<Vec<_>>();
7937
7938        assert_eq!(
7939            schedules
7940                .iter()
7941                .map(|schedule| schedule.restart_in_window)
7942                .collect::<Vec<_>>(),
7943            vec![0, 1, 2, 3]
7944        );
7945        assert_eq!(
7946            schedules
7947                .iter()
7948                .map(|schedule| schedule.delay)
7949                .collect::<Vec<_>>(),
7950            vec![
7951                Duration::from_millis(100),
7952                Duration::from_secs(1),
7953                Duration::from_secs(10),
7954                Duration::from_secs(30),
7955            ]
7956        );
7957    }
7958
7959    #[test]
7960    fn crash_restart_backoff_resets_after_ring_clear() {
7961        let policy = RestartPolicy::new(3, Duration::from_millis(100));
7962        let now = ClockInstant::now();
7963        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7964        assert_eq!(
7965            state.next_crash_restart(&policy, now).unwrap().delay,
7966            Duration::from_millis(100)
7967        );
7968        assert_eq!(
7969            state.next_crash_restart(&policy, now).unwrap().delay,
7970            Duration::from_secs(1)
7971        );
7972
7973        state.clear_crash_restarts();
7974        let schedule = state
7975            .next_crash_restart(&policy, now)
7976            .expect("a cleared ring must allow another restart");
7977        assert_eq!(schedule.restart_in_window, 0);
7978        assert_eq!(schedule.delay, Duration::from_millis(100));
7979    }
7980
7981    #[test]
7982    fn crash_restart_backoff_ignores_aged_restarts() {
7983        let policy = RestartPolicy::new(3, Duration::from_millis(100));
7984        let now = ClockInstant::now();
7985        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7986        state
7987            .next_crash_restart(&policy, now)
7988            .expect("the first restart is allowed");
7989        state
7990            .next_crash_restart(&policy, now)
7991            .expect("the second restart is allowed");
7992        state.crash_restarts[0] = now
7993            .checked_sub(policy.window + Duration::from_secs(1))
7994            .expect("the fake clock can age a restart past the window");
7995
7996        let schedule = state
7997            .next_crash_restart(&policy, now)
7998            .expect("an aged restart must release its slot");
7999        assert_eq!(schedule.restart_in_window, 1);
8000        assert_eq!(schedule.delay, Duration::from_secs(1));
8001        assert_eq!(state.crash_restarts.len(), 2);
8002    }
8003
8004    /// The budget is a rate: the same three spent restarts refuse a respawn
8005    /// while they are recent and allow one once they have aged past the window.
8006    /// Nothing about the module changed in between, which is the whole point.
8007    #[test]
8008    fn a_budget_spent_before_the_window_no_longer_refuses() {
8009        let policy = RestartPolicy::new(3, Duration::ZERO);
8010        let mut state = snapshot_with_restarts(true, 3);
8011        let now = ClockInstant::now();
8012        assert!(!daemon_will_restart(&mut state, &policy, now));
8013
8014        assert!(daemon_will_restart(
8015            &mut state,
8016            &policy,
8017            now + policy.window + Duration::from_secs(1)
8018        ));
8019        assert!(
8020            state.crash_restarts.is_empty(),
8021            "reading the budget must drop the instants that left the window"
8022        );
8023    }
8024
8025    fn module_with_recovery_snapshot(
8026        state: ModuleState,
8027        enabled: bool,
8028        restart_count: u32,
8029    ) -> SupervisedModule {
8030        let registry = Arc::new(Registry::default());
8031        let supervisor =
8032            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
8033        let module = supervisor
8034            .spawn(ModuleSpec {
8035                module_id: "recovery-snapshot".to_string(),
8036                program: fake_aft_stub_path(),
8037                args: Vec::new(),
8038                env: Vec::new(),
8039                reserved: false,
8040                reserved_prefixes: Vec::new(),
8041                protocol: ModuleProtocol::Subc,
8042                overlap: Default::default(),
8043            })
8044            .unwrap();
8045        update_snapshot(
8046            &module.inner.snapshot,
8047            Some("recovery-snapshot"),
8048            |snapshot| {
8049                snapshot.state = state;
8050                snapshot.enabled = enabled;
8051                seed_crash_restarts(snapshot, restart_count);
8052            },
8053        )
8054        .unwrap();
8055        module
8056    }
8057
8058    #[cfg(target_os = "linux")]
8059    #[tokio::test]
8060    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
8061        let supervisor = Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
8062            .with_cgroup_placement(None);
8063        let result = supervisor.spawn(ModuleSpec {
8064            module_id: "no-cgroup-placement".to_string(),
8065            program: fake_aft_stub_path(),
8066            args: Vec::new(),
8067            env: Vec::new(),
8068            reserved: false,
8069            reserved_prefixes: Vec::new(),
8070            protocol: ModuleProtocol::Subc,
8071            overlap: Default::default(),
8072        });
8073
8074        assert!(
8075            result.is_ok(),
8076            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
8077        );
8078    }
8079
8080    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8081    async fn undecided_snapshot_uses_shared_restart_predicate() {
8082        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
8083            .will_recover_after_connection_loss()
8084            .unwrap());
8085        assert!(
8086            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
8087                .will_recover_after_connection_loss()
8088                .unwrap()
8089        );
8090    }
8091
8092    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8093    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
8094        assert!(
8095            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
8096                .will_recover_after_connection_loss()
8097                .unwrap()
8098        );
8099    }
8100
8101    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8102    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
8103        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
8104            .will_recover_after_connection_loss()
8105            .unwrap());
8106        assert!(
8107            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
8108                .will_recover_after_connection_loss()
8109                .unwrap()
8110        );
8111    }
8112
8113    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8114    async fn warming_snapshot_is_limited_to_startup_phases() {
8115        for state in [
8116            ModuleState::Starting,
8117            ModuleState::Running,
8118            ModuleState::Restarting,
8119        ] {
8120            assert!(
8121                module_with_recovery_snapshot(state, true, 0)
8122                    .is_warming()
8123                    .unwrap(),
8124                "{state:?} should be warming"
8125            );
8126        }
8127        for state in [
8128            ModuleState::Unresponsive,
8129            ModuleState::Draining,
8130            ModuleState::Stopped,
8131            ModuleState::Failed,
8132            ModuleState::Disabled,
8133        ] {
8134            assert!(
8135                !module_with_recovery_snapshot(state, true, 0)
8136                    .is_warming()
8137                    .unwrap(),
8138                "{state:?} should not be warming"
8139            );
8140        }
8141    }
8142
8143    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8144    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
8145        let registry = Arc::new(Registry::default());
8146        let supervisor =
8147            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
8148        let module = supervisor
8149            .spawn(ModuleSpec {
8150                module_id: "terminal-history".to_string(),
8151                program: fake_aft_stub_path(),
8152                args: Vec::new(),
8153                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8154                reserved: false,
8155                reserved_prefixes: Vec::new(),
8156                protocol: ModuleProtocol::Subc,
8157                overlap: Default::default(),
8158            })
8159            .unwrap();
8160
8161        let deadline = Instant::now() + Duration::from_secs(5);
8162        loop {
8163            let history = module.terminal_history();
8164            if history.entries.len() == 2 {
8165                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
8166                assert_eq!(history.dropped, 0);
8167                assert_eq!(
8168                    history
8169                        .entries
8170                        .iter()
8171                        .map(|entry| entry.exit_code)
8172                        .collect::<Vec<_>>(),
8173                    vec![Some(23), Some(23)]
8174                );
8175                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
8176                return;
8177            }
8178            assert!(
8179                Instant::now() < deadline,
8180                "module did not retain two terminal exits: {history:?}"
8181            );
8182            sleep(Duration::from_millis(10)).await;
8183        }
8184    }
8185
8186    /// A disable issued while a crash respawn is still backing off must preempt
8187    /// that respawn: the operator's stop wins, the disable must not queue behind
8188    /// the backoff, and the module must never come back up afterwards.
8189    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8190    async fn disable_during_crash_backoff_cancels_pending_respawn() {
8191        let backoff = Duration::from_secs(2);
8192        let supervisor = Supervisor::new(
8193            Arc::new(Registry::default()),
8194            RestartPolicy::new(10, backoff),
8195        );
8196        let module = supervisor
8197            .spawn(ModuleSpec {
8198                module_id: "disable-during-backoff".to_string(),
8199                program: fake_aft_stub_path(),
8200                args: Vec::new(),
8201                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8202                reserved: false,
8203                reserved_prefixes: Vec::new(),
8204                protocol: ModuleProtocol::Subc,
8205                overlap: Default::default(),
8206            })
8207            .unwrap();
8208
8209        // Wait for the first crash to put the module into its backoff window.
8210        let deadline = Instant::now() + Duration::from_secs(5);
8211        loop {
8212            if module.status().unwrap().state == ModuleState::Restarting {
8213                break;
8214            }
8215            assert!(
8216                Instant::now() < deadline,
8217                "module never entered the crash backoff"
8218            );
8219            sleep(Duration::from_millis(10)).await;
8220        }
8221
8222        let started = Instant::now();
8223        module.set_enabled(false).await.unwrap();
8224        let waited = started.elapsed();
8225
8226        assert!(
8227            waited < backoff / 2,
8228            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
8229        );
8230        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
8231
8232        // Outlast the backoff: the respawn it was counting down to must never run.
8233        sleep(backoff + Duration::from_millis(500)).await;
8234        let status = module.status().unwrap();
8235        assert_eq!(status.state, ModuleState::Disabled);
8236        assert_eq!(
8237            status.spawn_generation, 1,
8238            "module respawned after the operator disabled it"
8239        );
8240    }
8241
8242    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
8243    /// the shape of nats-server, the program this rule exists for.
8244    #[cfg(unix)]
8245    fn protocol_none_sigterm_exits_clean_spec(
8246        module_id: &str,
8247        dir: &std::path::Path,
8248    ) -> (ModuleSpec, PathBuf, PathBuf) {
8249        let ready = dir.join("ready");
8250        let marker = dir.join("sigterm");
8251        let spec = ModuleSpec {
8252            module_id: module_id.to_string(),
8253            program: fake_aft_stub_path(),
8254            args: Vec::new(),
8255            env: vec![
8256                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
8257                (
8258                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
8259                    marker.display().to_string(),
8260                ),
8261                (
8262                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
8263                    ready.display().to_string(),
8264                ),
8265            ],
8266            reserved: false,
8267            reserved_prefixes: Vec::new(),
8268            protocol: ModuleProtocol::None,
8269            overlap: Default::default(),
8270        };
8271        (spec, ready, marker)
8272    }
8273
8274    /// Wait for a file the child writes, so a signal is never sent before the
8275    /// child's SIGTERM handler is installed (the default disposition would
8276    /// kill it by signal and the exit would not be clean).
8277    #[cfg(unix)]
8278    async fn wait_for_file(path: &std::path::Path) {
8279        let deadline = Instant::now() + Duration::from_secs(10);
8280        while !path.exists() {
8281            assert!(
8282                Instant::now() < deadline,
8283                "{} never appeared",
8284                path.display()
8285            );
8286            sleep(Duration::from_millis(10)).await;
8287        }
8288    }
8289
8290    /// A protocol-none module that exits 0 because something OUTSIDE the
8291    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
8292    /// the crash-path disposition rather than `stopped`.
8293    #[cfg(unix)]
8294    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8295    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
8296        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
8297        let (spec, ready, marker) =
8298            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
8299        let supervisor = Supervisor::new(
8300            Arc::new(Registry::default()),
8301            RestartPolicy::new(3, Duration::ZERO),
8302        );
8303        let module = supervisor.spawn(spec).unwrap();
8304        wait_for_file(&ready).await;
8305        let first_pid = module
8306            .status()
8307            .unwrap()
8308            .pid
8309            .expect("a running module reports its pid");
8310
8311        rustix::process::kill_process(
8312            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
8313            rustix::process::Signal::TERM,
8314        )
8315        .unwrap();
8316
8317        let deadline = Instant::now() + Duration::from_secs(10);
8318        let respawned = loop {
8319            let status = module.status().unwrap();
8320            if status.state == ModuleState::Running
8321                && status.pid.is_some_and(|pid| pid != first_pid)
8322            {
8323                break status;
8324            }
8325            assert!(
8326                Instant::now() < deadline,
8327                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
8328            );
8329            sleep(Duration::from_millis(10)).await;
8330        };
8331        assert_eq!(respawned.spawn_generation, 2);
8332        assert!(
8333            marker.exists(),
8334            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
8335        );
8336
8337        let history = module.terminal_history();
8338        assert_eq!(history.entries.len(), 1, "{history:?}");
8339        let entry = &history.entries[0];
8340        assert_eq!(entry.exit_code, Some(0));
8341        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
8342        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
8343
8344        module.stop().await.unwrap();
8345    }
8346
8347    /// Repeated unrequested clean exits of a protocol-none module spend the
8348    /// restart budget exactly as crashes do, and the module ends `failed` with
8349    /// the budget named.
8350    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8351    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
8352        let supervisor = Supervisor::new(
8353            Arc::new(Registry::default()),
8354            RestartPolicy::new(1, Duration::ZERO),
8355        );
8356        let module = supervisor
8357            .spawn(ModuleSpec {
8358                module_id: "none-clean-exit-budget".to_string(),
8359                program: fake_aft_stub_path(),
8360                args: Vec::new(),
8361                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8362                reserved: false,
8363                reserved_prefixes: Vec::new(),
8364                protocol: ModuleProtocol::None,
8365                overlap: Default::default(),
8366            })
8367            .unwrap();
8368
8369        let deadline = Instant::now() + Duration::from_secs(10);
8370        loop {
8371            let status = module.status().unwrap();
8372            if status.state == ModuleState::Failed {
8373                break;
8374            }
8375            assert!(
8376                Instant::now() < deadline,
8377                "module never exhausted its budget: {status:?} {:?}",
8378                module.terminal_history()
8379            );
8380            sleep(Duration::from_millis(10)).await;
8381        }
8382        let history = module.terminal_history();
8383        assert_eq!(
8384            history
8385                .entries
8386                .iter()
8387                .map(|entry| (entry.exit_code, entry.disposition.clone()))
8388                .collect::<Vec<_>>(),
8389            vec![
8390                (Some(0), TerminalDisposition::Restarting),
8391                (Some(0), TerminalDisposition::Failed),
8392            ]
8393        );
8394        let detail = history.entries[1]
8395            .disposition_detail
8396            .as_deref()
8397            .expect("a budget failure names the budget");
8398        assert!(detail.contains("max_restarts=1"), "{detail}");
8399        assert_eq!(module.status().unwrap().spawn_generation, 2);
8400    }
8401
8402    /// A stop the supervisor itself requests still stops a protocol-none
8403    /// module, even though the child answers the SIGTERM with exit 0.
8404    #[cfg(unix)]
8405    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8406    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
8407        for disable in [false, true] {
8408            let label = if disable {
8409                "none-requested-disable"
8410            } else {
8411                "none-requested-stop"
8412            };
8413            let dir = subc_test_support::TestTempDir::new(label);
8414            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
8415            let supervisor = Supervisor::new(
8416                Arc::new(Registry::default()),
8417                RestartPolicy::new(3, Duration::ZERO),
8418            );
8419            let module = supervisor.spawn(spec).unwrap();
8420            wait_for_file(&ready).await;
8421
8422            if disable {
8423                module.set_enabled(false).await.unwrap();
8424            } else {
8425                module.stop().await.unwrap();
8426            }
8427            assert!(
8428                marker.exists(),
8429                "{label}: the child must have left through its SIGTERM handler with exit 0"
8430            );
8431
8432            // Long enough for a zero-backoff respawn to have happened if the
8433            // exit had been treated as a crash.
8434            sleep(Duration::from_millis(500)).await;
8435            let status = module.status().unwrap();
8436            let expected = if disable {
8437                ModuleState::Disabled
8438            } else {
8439                ModuleState::Stopped
8440            };
8441            assert_eq!(status.state, expected, "{label}");
8442            assert_eq!(
8443                status.spawn_generation, 1,
8444                "{label}: respawned after a requested stop"
8445            );
8446            let history = module.terminal_history();
8447            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
8448            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
8449            assert_ne!(
8450                history.entries[0].disposition,
8451                TerminalDisposition::Restarting,
8452                "{label}"
8453            );
8454        }
8455    }
8456
8457    /// A subc-wire module that exits 0 on its own is still a stop: the
8458    /// protocol-none rule must not reach it.
8459    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8460    async fn subc_wire_clean_exit_is_still_a_stop() {
8461        let supervisor = Supervisor::new(
8462            Arc::new(Registry::default()),
8463            RestartPolicy::new(3, Duration::ZERO),
8464        );
8465        let module = supervisor
8466            .spawn(ModuleSpec {
8467                module_id: "wire-clean-exit".to_string(),
8468                program: fake_aft_stub_path(),
8469                args: Vec::new(),
8470                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8471                reserved: false,
8472                reserved_prefixes: Vec::new(),
8473                protocol: ModuleProtocol::Subc,
8474                overlap: Default::default(),
8475            })
8476            .unwrap();
8477
8478        let deadline = Instant::now() + Duration::from_secs(10);
8479        while module.terminal_history().entries.is_empty() {
8480            assert!(Instant::now() < deadline, "module never exited");
8481            sleep(Duration::from_millis(10)).await;
8482        }
8483        // Long enough for a zero-backoff respawn to have happened.
8484        sleep(Duration::from_millis(500)).await;
8485        let status = module.status().unwrap();
8486        assert_eq!(status.state, ModuleState::Stopped);
8487        assert_eq!(status.spawn_generation, 1);
8488        let history = module.terminal_history();
8489        assert_eq!(history.entries.len(), 1, "{history:?}");
8490        assert_eq!(history.entries[0].exit_code, Some(0));
8491        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
8492    }
8493
8494    /// Each restart-producing arm has its own state transition. Keeping their
8495    /// lifetime count assertions adjacent prevents a later new arm from silently
8496    /// spending budget without recording the historical restart.
8497    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8498    async fn every_restart_increment_path_advances_lifetime_count() {
8499        let supervisor = Supervisor::new(
8500            Arc::new(Registry::default()),
8501            RestartPolicy::new(1, Duration::ZERO),
8502        );
8503        let runtime = supervisor.runtime_config();
8504        let spec = ModuleSpec {
8505            module_id: "lifetime-increment-path".to_string(),
8506            program: PathBuf::from("/unused/lifetime-increment-path"),
8507            args: Vec::new(),
8508            env: Vec::new(),
8509            reserved: false,
8510            reserved_prefixes: Vec::new(),
8511            protocol: ModuleProtocol::Subc,
8512            overlap: Default::default(),
8513        };
8514
8515        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8516        assert!(matches!(
8517            on_child_exit(
8518                &spec,
8519                runtime.restart_policy,
8520                &supervisor.registry,
8521                &crash_snapshot,
8522                &runtime.terminal_ring,
8523                &runtime.spawn_events,
8524                &runtime.child_roster,
8525                ExitReport {
8526                    kind: ExitKind::Crash,
8527                    code: Some(1),
8528                    signal: None,
8529                    at_ms: 1,
8530                },
8531            )
8532            .await,
8533            NextAction::Restart { schedule: _ }
8534        ));
8535        let (crash_restarts, crash_lifetime) = {
8536            let state = lock_snapshot(&crash_snapshot).unwrap();
8537            (state.crash_restarts.len(), state.lifetime_restarts)
8538        };
8539        assert_eq!(crash_restarts, 1);
8540        assert_eq!(crash_lifetime, 1);
8541
8542        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8543        let mut health_child = None;
8544        assert!(matches!(
8545            health_restart_child(
8546                &spec,
8547                &runtime,
8548                &supervisor.registry,
8549                &supervisor.process_liveness,
8550                &health_snapshot,
8551                &mut health_child,
8552                SupervisorHealthStatus::Failing,
8553                None,
8554                2,
8555            )
8556            .await,
8557            Err(SuperviseError::Spawn { .. })
8558        ));
8559        let (health_restarts, health_lifetime) = {
8560            let state = lock_snapshot(&health_snapshot).unwrap();
8561            (state.crash_restarts.len(), state.lifetime_restarts)
8562        };
8563        assert_eq!(health_restarts, 1);
8564        assert_eq!(health_lifetime, 1);
8565
8566        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8567        let mut reload_child = None;
8568        assert!(matches!(
8569            handle_reload_spawn_failure(
8570                &spec,
8571                &runtime,
8572                &supervisor.process_liveness,
8573                &reload_snapshot,
8574                &mut reload_child,
8575                "forced reload spawn failure".to_string(),
8576            )
8577            .await,
8578            Err(SuperviseError::ReloadFailed { .. })
8579        ));
8580        let (reload_restarts, reload_lifetime) = {
8581            let state = lock_snapshot(&reload_snapshot).unwrap();
8582            (state.crash_restarts.len(), state.lifetime_restarts)
8583        };
8584        assert_eq!(reload_restarts, 1);
8585        assert_eq!(reload_lifetime, 1);
8586    }
8587
8588    #[tokio::test]
8589    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
8590        let supervisor = Supervisor::new(
8591            Arc::new(Registry::default()),
8592            RestartPolicy::new(3, Duration::ZERO),
8593        );
8594        let runtime = supervisor.runtime_config();
8595        let spec = ModuleSpec {
8596            module_id: "deliberately-severed".to_string(),
8597            program: PathBuf::from("/unused/deliberately-severed"),
8598            args: Vec::new(),
8599            env: Vec::new(),
8600            reserved: false,
8601            reserved_prefixes: Vec::new(),
8602            protocol: ModuleProtocol::Subc,
8603            overlap: Default::default(),
8604        };
8605        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8606        let process = ProcessIdentity {
8607            pid: 41,
8608            start_time: 101,
8609        };
8610        record_deliberate_severance(&snapshot, process).unwrap();
8611        let exit_report = apply_deliberate_severance_marker(
8612            &snapshot,
8613            Some(process),
8614            ExitReport {
8615                kind: ExitKind::Crash,
8616                code: Some(1),
8617                signal: None,
8618                at_ms: 1,
8619            },
8620        );
8621        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
8622
8623        assert!(matches!(
8624            on_child_exit(
8625                &spec,
8626                runtime.restart_policy,
8627                &supervisor.registry,
8628                &snapshot,
8629                &runtime.terminal_ring,
8630                &runtime.spawn_events,
8631                &runtime.child_roster,
8632                exit_report,
8633            )
8634            .await,
8635            NextAction::Restart { schedule: _ }
8636        ));
8637        let state = lock_snapshot(&snapshot).unwrap();
8638        assert_eq!(state.lifetime_restarts, 1);
8639        assert_eq!(state.crash_restarts.len(), 0);
8640    }
8641
8642    #[tokio::test]
8643    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
8644        let supervisor = Supervisor::new(
8645            Arc::new(Registry::default()),
8646            RestartPolicy::new(3, Duration::ZERO),
8647        );
8648        let runtime = supervisor.runtime_config();
8649        let spec = ModuleSpec {
8650            module_id: "genuine-crash".to_string(),
8651            program: PathBuf::from("/unused/genuine-crash"),
8652            args: Vec::new(),
8653            env: Vec::new(),
8654            reserved: false,
8655            reserved_prefixes: Vec::new(),
8656            protocol: ModuleProtocol::Subc,
8657            overlap: Default::default(),
8658        };
8659        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8660
8661        assert!(matches!(
8662            on_child_exit(
8663                &spec,
8664                runtime.restart_policy,
8665                &supervisor.registry,
8666                &snapshot,
8667                &runtime.terminal_ring,
8668                &runtime.spawn_events,
8669                &runtime.child_roster,
8670                ExitReport {
8671                    kind: ExitKind::Crash,
8672                    code: Some(1),
8673                    signal: None,
8674                    at_ms: 1,
8675                },
8676            )
8677            .await,
8678            NextAction::Restart { schedule: _ }
8679        ));
8680        let state = lock_snapshot(&snapshot).unwrap();
8681        assert_eq!(state.lifetime_restarts, 1);
8682        assert_eq!(state.crash_restarts.len(), 1);
8683    }
8684
8685    fn crash_exit_report(at_ms: u64) -> ExitReport {
8686        ExitReport {
8687            kind: ExitKind::Crash,
8688            code: Some(1),
8689            signal: None,
8690            at_ms,
8691        }
8692    }
8693
8694    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
8695        ModuleSpec {
8696            module_id: module_id.to_string(),
8697            program: PathBuf::from("/unused").join(module_id),
8698            args: Vec::new(),
8699            env: Vec::new(),
8700            reserved: false,
8701            reserved_prefixes: Vec::new(),
8702            protocol: ModuleProtocol::Subc,
8703            overlap: Default::default(),
8704        }
8705    }
8706
8707    /// A real crash loop still stops. Three crashes with nothing aging out spend
8708    /// a budget of two and the third respawn is refused, and both surfaces an
8709    /// operator has -- the log line and the retained terminal record -- name the
8710    /// window rather than only the cap, because `max_restarts=2` alone is what
8711    /// this budget used to mean.
8712    #[tokio::test]
8713    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
8714        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
8715        let supervisor = Supervisor::new(
8716            Arc::new(Registry::default()),
8717            RestartPolicy::new(2, Duration::ZERO),
8718        );
8719        let runtime = supervisor.runtime_config();
8720        let spec = windowed_crash_spec("crash-loop-in-window");
8721        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8722
8723        for attempt in 1..=2 {
8724            assert!(
8725                matches!(
8726                    on_child_exit(
8727                        &spec,
8728                        runtime.restart_policy,
8729                        &supervisor.registry,
8730                        &snapshot,
8731                        &runtime.terminal_ring,
8732                        &runtime.spawn_events,
8733                        &runtime.child_roster,
8734                        crash_exit_report(attempt),
8735                    )
8736                    .await,
8737                    NextAction::Restart { schedule: _ }
8738                ),
8739                "crash {attempt} is inside the budget and must respawn"
8740            );
8741        }
8742
8743        assert!(matches!(
8744            on_child_exit(
8745                &spec,
8746                runtime.restart_policy,
8747                &supervisor.registry,
8748                &snapshot,
8749                &runtime.terminal_ring,
8750                &runtime.spawn_events,
8751                &runtime.child_roster,
8752                crash_exit_report(3),
8753            )
8754            .await,
8755            NextAction::Stop { .. }
8756        ));
8757
8758        {
8759            let state = lock_snapshot(&snapshot).unwrap();
8760            assert_eq!(state.state, ModuleState::Failed);
8761            assert_eq!(state.crash_restarts.len(), 2);
8762            assert_eq!(state.lifetime_restarts, 2);
8763        }
8764
8765        let history = runtime
8766            .terminal_ring
8767            .lock()
8768            .expect("terminal ring is not poisoned")
8769            .snapshot();
8770        let last = history
8771            .entries
8772            .last()
8773            .expect("the refused crash is retained");
8774        assert_eq!(last.disposition, TerminalDisposition::Failed);
8775        assert_eq!(
8776            last.disposition_detail.as_deref(),
8777            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
8778        );
8779
8780        let captured = crate::router::test_log::captured_logs(&logs);
8781        assert!(
8782            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
8783            "the stop must be logged with its window: {captured}"
8784        );
8785    }
8786
8787    /// The rate, stated as a test: three crashes where the first has aged past
8788    /// the window are two crashes as far as the budget is concerned, so the
8789    /// third respawn is allowed and the ring holds only the two recent ones.
8790    ///
8791    /// This is the case a lifetime counter got wrong -- and the case the daemon
8792    /// now hits routinely, since a module exits non-zero every time its
8793    /// connection to the daemon drops.
8794    #[tokio::test]
8795    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
8796        let supervisor = Supervisor::new(
8797            Arc::new(Registry::default()),
8798            RestartPolicy::new(2, Duration::ZERO),
8799        );
8800        let runtime = supervisor.runtime_config();
8801        let spec = windowed_crash_spec("crash-across-windows");
8802        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8803
8804        for attempt in 1..=2 {
8805            assert!(matches!(
8806                on_child_exit(
8807                    &spec,
8808                    runtime.restart_policy,
8809                    &supervisor.registry,
8810                    &snapshot,
8811                    &runtime.terminal_ring,
8812                    &runtime.spawn_events,
8813                    &runtime.child_roster,
8814                    crash_exit_report(attempt),
8815                )
8816                .await,
8817                NextAction::Restart { schedule: _ }
8818            ));
8819        }
8820
8821        // The oldest crash moves out of the window; nothing else about the
8822        // module changes.
8823        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
8824            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
8825        })
8826        .unwrap();
8827
8828        assert!(
8829            matches!(
8830                on_child_exit(
8831                    &spec,
8832                    runtime.restart_policy,
8833                    &supervisor.registry,
8834                    &snapshot,
8835                    &runtime.terminal_ring,
8836                    &runtime.spawn_events,
8837                    &runtime.child_roster,
8838                    crash_exit_report(3),
8839                )
8840                .await,
8841                NextAction::Restart { schedule: _ }
8842            ),
8843            "a crash older than the window must not hold a budget slot"
8844        );
8845
8846        let state = lock_snapshot(&snapshot).unwrap();
8847        assert_eq!(state.state, ModuleState::Restarting);
8848        assert_eq!(
8849            state.crash_restarts.len(),
8850            2,
8851            "the aged instant is dropped and the new one takes its place"
8852        );
8853        assert_eq!(
8854            state.lifetime_restarts, 3,
8855            "the ledger counts every restart, including the ones the window forgot"
8856        );
8857    }
8858
8859    /// An operator restart hands the budget back whole, and the ledger keeps
8860    /// counting. Those are different questions -- "how close is this module to
8861    /// being stopped" and "how many times has it been replaced" -- and the
8862    /// operator action answers only the first.
8863    #[tokio::test]
8864    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
8865        let supervisor = Supervisor::new(
8866            Arc::new(Registry::default()),
8867            RestartPolicy::new(2, Duration::ZERO),
8868        );
8869        let runtime = supervisor.runtime_config();
8870        let spec = windowed_crash_spec("operator-cleared-budget");
8871        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8872
8873        for attempt in 1..=2 {
8874            assert!(matches!(
8875                on_child_exit(
8876                    &spec,
8877                    runtime.restart_policy,
8878                    &supervisor.registry,
8879                    &snapshot,
8880                    &runtime.terminal_ring,
8881                    &runtime.spawn_events,
8882                    &runtime.child_roster,
8883                    crash_exit_report(attempt),
8884                )
8885                .await,
8886                NextAction::Restart { schedule: _ }
8887            ));
8888        }
8889
8890        reset_restart_count(&snapshot, &spec.module_id).unwrap();
8891        {
8892            let state = lock_snapshot(&snapshot).unwrap();
8893            assert!(
8894                state.crash_restarts.is_empty(),
8895                "an operator restart returns the full budget"
8896            );
8897            assert_eq!(
8898                state.lifetime_restarts, 2,
8899                "clearing the budget must not unmake the crashes"
8900            );
8901        }
8902
8903        assert!(
8904            matches!(
8905                on_child_exit(
8906                    &spec,
8907                    runtime.restart_policy,
8908                    &supervisor.registry,
8909                    &snapshot,
8910                    &runtime.terminal_ring,
8911                    &runtime.spawn_events,
8912                    &runtime.child_roster,
8913                    crash_exit_report(3),
8914                )
8915                .await,
8916                NextAction::Restart { schedule: _ }
8917            ),
8918            "the cleared budget must be spendable again"
8919        );
8920        let state = lock_snapshot(&snapshot).unwrap();
8921        assert_eq!(state.crash_restarts.len(), 1);
8922        assert_eq!(state.lifetime_restarts, 3);
8923    }
8924
8925    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8926    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
8927        let severed = ProcessIdentity {
8928            pid: 41,
8929            start_time: 101,
8930        };
8931        let successor = ProcessIdentity {
8932            pid: 41,
8933            start_time: 202,
8934        };
8935        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
8936        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
8937            state.pid = Some(successor.pid);
8938            state.process_start_time = Some(successor.start_time);
8939        })
8940        .unwrap();
8941        assert!(!module.record_deliberate_severance(severed).unwrap());
8942
8943        let exit_report = apply_deliberate_severance_marker(
8944            &module.inner.snapshot,
8945            Some(successor),
8946            ExitReport {
8947                kind: ExitKind::Crash,
8948                code: Some(1),
8949                signal: None,
8950                at_ms: 1,
8951            },
8952        );
8953
8954        assert_eq!(exit_report.kind, ExitKind::Crash);
8955    }
8956
8957    #[tokio::test]
8958    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
8959        let registry = Registry::default();
8960        let supervisor = Supervisor::new(
8961            Arc::new(Registry::default()),
8962            RestartPolicy::new(3, Duration::ZERO),
8963        );
8964        let runtime = supervisor.runtime_config();
8965        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8966        let spec = ModuleSpec {
8967            module_id: "drain-deliberate-severance".to_string(),
8968            program: fake_aft_stub_path(),
8969            args: Vec::new(),
8970            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8971            reserved: false,
8972            reserved_prefixes: Vec::new(),
8973            protocol: ModuleProtocol::Subc,
8974            overlap: Default::default(),
8975        };
8976        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
8977        let process = ProcessIdentity {
8978            pid: 41,
8979            start_time: 101,
8980        };
8981        child.process_identity = Some(process);
8982        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
8983            state.pid = Some(process.pid);
8984            state.process_start_time = Some(process.start_time);
8985        })
8986        .unwrap();
8987        record_deliberate_severance(&snapshot, process).unwrap();
8988
8989        drain_child_to_state(
8990            &spec.module_id,
8991            spec.protocol,
8992            // The child exits on its own; no signal may change the exit this
8993            // test classifies.
8994            StopNotice::SentOverConnection,
8995            &registry,
8996            &snapshot,
8997            &runtime.terminal_ring,
8998            &runtime.spawn_events,
8999            child,
9000            Duration::from_secs(1),
9001            ModuleState::Stopped,
9002            Some(false),
9003        )
9004        .await
9005        .unwrap();
9006
9007        let state = lock_snapshot(&snapshot).unwrap();
9008        assert_eq!(
9009            state.last_exit.as_ref().map(|exit| exit.kind),
9010            Some(ExitKind::DeliberateSeverance)
9011        );
9012        assert_eq!(state.lifetime_restarts, 1);
9013        assert_eq!(state.crash_restarts.len(), 0);
9014        drop(state);
9015        let history = runtime.terminal_ring.lock().unwrap().snapshot();
9016        assert_eq!(
9017            history.entries[0].exit_kind,
9018            subc_control::TerminalExitKind::DeliberateSeverance
9019        );
9020    }
9021
9022    #[tokio::test]
9023    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
9024        let registry = Registry::default();
9025        let supervisor = Supervisor::new(
9026            Arc::new(Registry::default()),
9027            RestartPolicy::new(3, Duration::ZERO),
9028        );
9029        let runtime = supervisor.runtime_config();
9030        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9031        let spec = ModuleSpec {
9032            module_id: "ordinary-drain".to_string(),
9033            program: fake_aft_stub_path(),
9034            args: Vec::new(),
9035            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9036            reserved: false,
9037            reserved_prefixes: Vec::new(),
9038            protocol: ModuleProtocol::Subc,
9039            overlap: Default::default(),
9040        };
9041        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9042
9043        drain_child_to_state(
9044            &spec.module_id,
9045            spec.protocol,
9046            // The child exits on its own; no signal may change the exit this
9047            // test classifies.
9048            StopNotice::SentOverConnection,
9049            &registry,
9050            &snapshot,
9051            &runtime.terminal_ring,
9052            &runtime.spawn_events,
9053            child,
9054            Duration::from_secs(1),
9055            ModuleState::Stopped,
9056            Some(false),
9057        )
9058        .await
9059        .unwrap();
9060
9061        let state = lock_snapshot(&snapshot).unwrap();
9062        assert_eq!(
9063            state.last_exit.as_ref().map(|exit| exit.kind),
9064            Some(ExitKind::Crash)
9065        );
9066        assert_eq!(state.lifetime_restarts, 0);
9067        assert_eq!(state.crash_restarts.len(), 0);
9068    }
9069
9070    #[test]
9071    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
9072        // The server's generic fatal-routing branch only knows that the
9073        // connection failed; it does not know that the daemon deliberately
9074        // initiated a process-killing severance. Keep this seam explicit so a
9075        // future connection error path cannot silently reintroduce the stale
9076        // exemption that mislabels a later genuine crash.
9077        assert!(!include_str!("server.rs")
9078            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
9079    }
9080
9081    /// The `route.closed` `drained` value must be the quiescence wait's own
9082    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
9083    /// measurement at all and `false` is the one honest constant. This is the exact
9084    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
9085    /// on every return path, including the one that used to return early via `?`
9086    /// with `route.closing` already sent and no `route.closed` ever following.
9087    #[test]
9088    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
9089        assert!(drained_after_quiescence_wait(&Ok(true)));
9090        assert!(!drained_after_quiescence_wait(&Ok(false)));
9091        assert!(!drained_after_quiescence_wait(&Err(
9092            SuperviseError::StatePoisoned { module_id: None }
9093        )));
9094    }
9095
9096    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
9097    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
9098    /// already reaped out-of-band) still leaves a terminal record rather than none
9099    /// at all. Triggering the real `wait()` I/O error from an integration test would
9100    /// need a genuine already-reaped-child race, which is OS-specific and not
9101    /// something this suite attempts elsewhere; this test instead verifies the
9102    /// record produced for that arm end-to-end through the real `TerminalRing`, and
9103    /// the call site itself is verified by inspection to sit in that exact arm.
9104    #[test]
9105    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
9106        let ring = Arc::new(Mutex::new(TerminalRing::new(
9107            TerminalRingConfig::default(),
9108            0,
9109        )));
9110        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
9111
9112        let snapshot = ring.lock().unwrap().snapshot();
9113        assert_eq!(snapshot.entries.len(), 1);
9114        let entry = &snapshot.entries[0];
9115        assert_eq!(entry.exit_code, None);
9116        assert_eq!(entry.exit_signal, None);
9117        assert_eq!(entry.disposition, TerminalDisposition::Failed);
9118    }
9119
9120    #[test]
9121    fn wait_error_exit_path_preserves_spawn_event_density() {
9122        let feed = super::SpawnEventFeed::default();
9123        feed.configure_incarnation("wait-error-density".to_string());
9124        feed.emit_spawned("wait-error", 41, 1);
9125        let ring = Arc::new(Mutex::new(TerminalRing::new(
9126            TerminalRingConfig::default(),
9127            0,
9128        )));
9129
9130        record_wait_error_terminal("wait-error", &ring, &feed);
9131        feed.emit_spawned("after-wait-error", 42, 2);
9132
9133        let state = feed.0.lock().unwrap();
9134        let sequences = state
9135            .events
9136            .iter()
9137            .map(|event| event.cursor.seq)
9138            .collect::<Vec<_>>();
9139        assert_eq!(sequences, vec![1, 2, 3]);
9140        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
9141        assert_eq!(state.events[1].exit_code, None);
9142        assert_eq!(state.events[1].exit_signal, None);
9143    }
9144
9145    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
9146    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
9147    /// not a clean exit it never actually observed.
9148    #[test]
9149    fn wait_error_exit_report_is_classified_as_a_crash() {
9150        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
9151    }
9152}
9153
9154#[cfg(test)]
9155mod health_evidence_tests {
9156    use super::{HealthProbeError, HealthProbeEvidence};
9157    use std::collections::HashSet;
9158
9159    /// The evidential asymmetry, asserted rather than described.
9160    ///
9161    /// Exactly ONE observation is proof a module cannot serve, and the one that
9162    /// fires under CPU starvation is not it. Before the split, all fifteen
9163    /// construction sites collapsed into a single String, so a timeout carried the
9164    /// same weight as a dead lane -- which is how a healthy module was restarted
9165    /// three times in one day.
9166    #[test]
9167    fn only_a_dead_lane_is_proof_of_death() {
9168        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
9169        // Three non-proof classes, each for a different reason: silence is
9170        // consistent with health, a bad answer proves the module ALIVE, and a
9171        // daemon-side fault never reached the module at all.
9172        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
9173        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
9174        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
9175    }
9176
9177    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
9178    ///
9179    /// A shared label renders two different observations identically in the line an
9180    /// operator reads after an unexplained restart -- the exact confusion this
9181    /// change removes.
9182    #[test]
9183    fn every_evidence_class_has_a_distinct_label() {
9184        let labels = [
9185            HealthProbeError::lane_dead("").label(),
9186            HealthProbeError::no_answer("").label(),
9187            HealthProbeError::bad_answer("").label(),
9188            HealthProbeError::misconfigured("").label(),
9189        ];
9190        let unique: HashSet<_> = labels.iter().collect();
9191        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
9192    }
9193
9194    /// The class is additional information, not a replacement.
9195    ///
9196    /// An operator needs both "this was silence" and the specific text saying how
9197    /// long we waited; a classification that swallowed the message would trade one
9198    /// missing distinction for another.
9199    #[test]
9200    fn classification_preserves_the_original_message() {
9201        let err = HealthProbeError::no_answer("module did not answer within 5s");
9202        assert_eq!(err.to_string(), "module did not answer within 5s");
9203        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9204    }
9205}
9206
9207#[cfg(test)]
9208mod health_tombstone_tests {
9209    use std::{path::PathBuf, sync::Arc, time::Duration};
9210
9211    use subc_protocol::{
9212        manifest::Concurrency,
9213        session::{HealthStatus, ModuleControlResponse},
9214    };
9215    use tokio::sync::mpsc;
9216
9217    use super::{
9218        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
9219        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
9220    };
9221    use crate::{
9222        control::ControlHandler,
9223        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
9224        registry::{ConnectionId, Registry},
9225        router::FrameSink,
9226    };
9227
9228    struct ProbeHarness {
9229        spec: ModuleSpec,
9230        runtime: SupervisorRuntimeConfig,
9231        forwarding: Arc<ForwardingTable>,
9232        module_connection: ConnectionId,
9233        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
9234        handler: ControlHandler,
9235        module: super::SupervisedModule,
9236    }
9237
9238    fn probe_harness() -> ProbeHarness {
9239        let registry = Arc::new(Registry::default());
9240        let forwarding = Arc::new(ForwardingTable::default());
9241        let supervisor_handle = super::SupervisorHandle::new();
9242        let health = HealthConfig {
9243            cadence: Duration::from_secs(30),
9244            deadline: Duration::from_secs(5),
9245            failure_threshold: 3,
9246            on_degraded: HealthAction::Report,
9247            on_failing: HealthAction::Report,
9248            critical: false,
9249        };
9250        let supervisor = Supervisor::new(Arc::clone(&registry), RestartPolicy::default())
9251            .with_forwarding(Arc::clone(&forwarding))
9252            .with_handle(supervisor_handle.clone())
9253            .with_health_config(health);
9254        let spec = ModuleSpec {
9255            module_id: "late-health-module".to_string(),
9256            program: PathBuf::from("disabled-module"),
9257            args: Vec::new(),
9258            env: Vec::new(),
9259            reserved: false,
9260            reserved_prefixes: Vec::new(),
9261            protocol: ModuleProtocol::Subc,
9262            overlap: Default::default(),
9263        };
9264        let module = supervisor
9265            .supervise_configured(spec.clone(), false)
9266            .unwrap();
9267        let runtime = supervisor.runtime_config();
9268        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
9269            .with_supervisor(supervisor_handle);
9270        let module_connection = ConnectionId::new(700);
9271        let (module_tx, module_rx) = mpsc::channel(8);
9272        forwarding
9273            .register_module_connection(
9274                module_connection,
9275                spec.module_id.clone(),
9276                subc_protocol::PROTOCOL_VERSION,
9277                Concurrency::ModuleManaged,
9278                FrameSink::new(module_tx),
9279            )
9280            .unwrap();
9281
9282        ProbeHarness {
9283            spec,
9284            runtime,
9285            forwarding,
9286            module_connection,
9287            module_rx,
9288            handler,
9289            module,
9290        }
9291    }
9292
9293    async fn finish_after(
9294        harness: &mut ProbeHarness,
9295        stall: Duration,
9296    ) -> ModuleControlRpcCompletion {
9297        assert!(stall > harness.runtime.health.deadline);
9298        let deadline = harness.runtime.health.deadline;
9299        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9300        let answer = async {
9301            let frame = harness.module_rx.recv().await.expect("health.check frame");
9302            tokio::time::advance(deadline).await;
9303            tokio::task::yield_now().await;
9304            tokio::time::advance(stall - deadline).await;
9305            harness
9306                .forwarding
9307                .complete_module_control_rpc(
9308                    harness.module_connection,
9309                    frame.header.corr,
9310                    Some("health.check"),
9311                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
9312                        status: HealthStatus::Ok,
9313                        detail: None,
9314                        metrics: None,
9315                    }),
9316                )
9317                .unwrap()
9318        };
9319        let (probe_result, completion) = tokio::join!(probe, answer);
9320        let err = probe_result.expect_err("probe must miss its deadline");
9321        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9322        completion
9323    }
9324
9325    async fn time_out_without_answer(harness: &mut ProbeHarness) {
9326        let deadline = harness.runtime.health.deadline;
9327        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9328        let exhaust_deadline = async {
9329            let _frame = harness.module_rx.recv().await.expect("health.check frame");
9330            tokio::time::advance(deadline).await;
9331            tokio::task::yield_now().await;
9332        };
9333        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
9334        let err = probe_result.expect_err("probe must miss its deadline");
9335        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9336    }
9337
9338    #[tokio::test(start_paused = true)]
9339    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
9340        let mut harness = probe_harness();
9341
9342        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
9343        let first_latency = match &first {
9344            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9345            other => panic!("late answer was not retained: {other:?}"),
9346        };
9347        assert!(harness.handler.observe_module_control_completion(first));
9348
9349        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
9350        let second_latency = match &second {
9351            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9352            other => panic!("late answer was not retained: {other:?}"),
9353        };
9354        assert!(harness.handler.observe_module_control_completion(second));
9355
9356        assert_eq!(first_latency, Duration::from_secs(8));
9357        assert_eq!(
9358            second_latency - first_latency,
9359            Duration::from_secs(3),
9360            "latency must grow linearly with the additional stall"
9361        );
9362        let health = harness.module.status().unwrap().health;
9363        assert_eq!(health.late_answer_count, 2);
9364        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
9365    }
9366
9367    /// A module that answers every probe late must never march to the kill
9368    /// threshold: the late answer proves it is alive, so it must clear the miss
9369    /// streak the timeout recorded. Without the reset, a CPU-starved module
9370    /// that serves every probe seconds past the deadline accumulates
9371    /// `consecutive_failures` to the threshold and is killed — the exact
9372    /// sequence from the 2026-08-14 aft disable, where the daemon logged
9373    /// "proves the module is alive" five times while counting five misses.
9374    #[tokio::test(start_paused = true)]
9375    async fn late_answer_clears_the_consecutive_failure_streak() {
9376        let mut harness = probe_harness();
9377
9378        // Timeout recorded first: the probe path saw no answer in time.
9379        time_out_without_answer(&mut harness).await;
9380        harness
9381            .module
9382            .record_health_probe_failure_for_test("[no-answer] test miss")
9383            .unwrap();
9384        assert_eq!(
9385            harness.module.status().unwrap().health.consecutive_failures,
9386            1,
9387            "precondition: the miss must be on the streak before the late answer"
9388        );
9389
9390        // The stalled reply then lands: proof of life.
9391        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
9392        assert!(matches!(
9393            late,
9394            ModuleControlRpcCompletion::LateHealthAnswer { .. }
9395        ));
9396        assert!(harness.handler.observe_module_control_completion(late));
9397
9398        let health = harness.module.status().unwrap().health;
9399        assert_eq!(
9400            health.consecutive_failures, 0,
9401            "a late answer is an answer: the streak must reset"
9402        );
9403        assert_eq!(health.late_answer_count, 1);
9404    }
9405
9406    #[tokio::test(start_paused = true)]
9407    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
9408        let mut harness = probe_harness();
9409
9410        for _ in 0..20 {
9411            time_out_without_answer(&mut harness).await;
9412            assert_eq!(
9413                harness.forwarding.health_probe_tombstone_count().unwrap(),
9414                1
9415            );
9416        }
9417    }
9418}
9419
9420#[cfg(test)]
9421mod child_env_tests {
9422    use super::{
9423        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
9424        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
9425        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
9426    };
9427    use std::{ffi::OsStr, path::PathBuf};
9428    use tokio::process::Command;
9429
9430    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
9431        ModuleSpec {
9432            module_id: "env-plan".to_string(),
9433            program: PathBuf::from("/nonexistent"),
9434            args: Vec::new(),
9435            env,
9436            reserved: false,
9437            reserved_prefixes: Vec::new(),
9438            protocol: ModuleProtocol::Subc,
9439            overlap: Default::default(),
9440        }
9441    }
9442
9443    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
9444    /// one still gets its own.
9445    ///
9446    /// This is the narrow goal `env_clear()` was reached for, and the reason the
9447    /// fix is `env_remove` rather than deleting the line: an operator's ambient
9448    /// filter silently becoming an unconfigured module's log level is a real
9449    /// defect, just a much smaller one than clearing the environment.
9450    ///
9451    /// Asserted on the command plan rather than a spawned child because proving
9452    /// the ABSENCE of an inherited variable needs the parent's environment
9453    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
9454    /// removal as `(key, None)`, which is exactly the distinction wanted: not
9455    /// "absent because nobody set it" but "explicitly unset for the child".
9456    #[test]
9457    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
9458        let mut command = Command::new("/nonexistent");
9459        apply_child_env(&mut command, &spec(Vec::new()));
9460        let removed = command
9461            .as_std()
9462            .get_envs()
9463            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
9464        assert!(
9465            removed,
9466            "ambient CK_LOG must be explicitly removed for an unconfigured module"
9467        );
9468
9469        let mut configured = Command::new("/nonexistent");
9470        apply_child_env(
9471            &mut configured,
9472            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
9473        );
9474        let effective = configured
9475            .as_std()
9476            .get_envs()
9477            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
9478            .last()
9479            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9480        assert_eq!(
9481            effective,
9482            Some(Some("debug".to_string())),
9483            "a module's configured CK_LOG must survive the ambient removal"
9484        );
9485    }
9486
9487    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
9488    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
9489    /// the same reason as the CK_LOG test above.
9490    ///
9491    /// The argument is the load-bearing half: a stock binary exits on an
9492    /// unknown flag before it listens, so with `--subc` appended the mode
9493    /// could not supervise the one process it exists for. Found by the first
9494    /// conformance run (nats-server: `flag provided but not defined: -subc`).
9495    #[test]
9496    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
9497        let connection_file = std::path::Path::new("/run/subc-connection.json");
9498        let handle = SupervisorHandle::new();
9499
9500        let mut none_spec = spec(Vec::new());
9501        none_spec.protocol = ModuleProtocol::None;
9502        let mut none = Command::new("/nonexistent");
9503        let none_handoff =
9504            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
9505                .expect("protocol-none spawn args apply");
9506        assert!(
9507            none_handoff.is_none(),
9508            "protocol:none spawn must not receive a nonce descriptor"
9509        );
9510        assert!(
9511            !none.as_std().get_envs().any(|(key, value)| key
9512                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
9513                && value.is_some()),
9514            "protocol:none spawn must not name a nonce descriptor"
9515        );
9516        let none_args: Vec<String> = none
9517            .as_std()
9518            .get_args()
9519            .map(|a| a.to_string_lossy().into_owned())
9520            .collect();
9521        assert!(
9522            !none_args.iter().any(|a| a == SUBC_ARG),
9523            "protocol:none argv must not carry --subc; got {none_args:?}"
9524        );
9525        let none_has_nonce = none
9526            .as_std()
9527            .get_envs()
9528            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
9529        assert!(
9530            !none_has_nonce,
9531            "protocol:none spawn must not receive a launch nonce"
9532        );
9533        let none_has_module_id = none
9534            .as_std()
9535            .get_envs()
9536            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
9537        assert!(
9538            none_has_module_id,
9539            "SUBC_MODULE_ID is inert and stays on every path"
9540        );
9541        assert!(
9542            handle.spawn_nonce(&none_spec.module_id).is_none(),
9543            "no nonce record for a process that will never present one"
9544        );
9545
9546        // Control: the subc-wire path is unchanged by the branch above.
9547        let wire_spec = spec(Vec::new());
9548        let mut wire = Command::new("/nonexistent");
9549        let wire_handoff =
9550            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
9551                .expect("subc-wire spawn args apply");
9552        let wire_fd_env = wire
9553            .as_std()
9554            .get_envs()
9555            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
9556            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9557        #[cfg(unix)]
9558        assert_eq!(
9559            wire_fd_env,
9560            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
9561            "a subc-wire spawn names the pipe it will receive at descriptor 3"
9562        );
9563        #[cfg(not(unix))]
9564        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
9565        let wire_args: Vec<String> = wire
9566            .as_std()
9567            .get_args()
9568            .map(|a| a.to_string_lossy().into_owned())
9569            .collect();
9570        assert_eq!(
9571            wire_args,
9572            vec![
9573                SUBC_ARG.to_string(),
9574                connection_file.to_string_lossy().into_owned()
9575            ],
9576            "a subc-wire spawn still carries --subc <path>"
9577        );
9578        assert_eq!(
9579            wire.as_std()
9580                .get_envs()
9581                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
9582            !cfg!(unix),
9583            "only Windows supplies the environment nonce"
9584        );
9585        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
9586    }
9587
9588    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
9589    /// spec tries to set it; only a swap candidate carries it.
9590    ///
9591    /// "Set it only on candidates" is not enough, because spawn applies the
9592    /// spec's env verbatim and the daemon's own environment is inherited: either
9593    /// could hand a plain restart the swap role, and a module reading it would
9594    /// warm on its long swap budget while callers wait. Asserted as an explicit
9595    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
9596    /// test above gives.
9597    #[test]
9598    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
9599        let role = |command: &Command| {
9600            command
9601                .as_std()
9602                .get_envs()
9603                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
9604                .last()
9605                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
9606        };
9607        let forged = spec(vec![(
9608            SUBC_SPAWN_ROLE_ENV.to_string(),
9609            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
9610        )]);
9611
9612        let mut plain = Command::new("/nonexistent");
9613        apply_child_env(&mut plain, &forged);
9614        apply_spawn_role(&mut plain, SpawnRole::Plain);
9615        assert_eq!(
9616            role(&plain),
9617            Some(None),
9618            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
9619        );
9620
9621        let mut candidate = Command::new("/nonexistent");
9622        apply_child_env(&mut candidate, &spec(Vec::new()));
9623        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
9624        assert_eq!(
9625            role(&candidate),
9626            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
9627        );
9628    }
9629
9630    /// Daemon-private capture retention keys never reach the child.
9631    ///
9632    /// cortexkit-log exposes retention as a Rust struct with no environment
9633    /// names, so these entries are supervisor metadata. Passing them through
9634    /// would invent a public child-process contract by accident.
9635    #[test]
9636    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
9637        let mut command = Command::new("/nonexistent");
9638        apply_child_env(
9639            &mut command,
9640            &spec(vec![
9641                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
9642                ("KEPT".to_string(), "yes".to_string()),
9643            ]),
9644        );
9645        let keys: Vec<String> = command
9646            .as_std()
9647            .get_envs()
9648            .filter(|(_, value)| value.is_some())
9649            .map(|(key, _)| key.to_string_lossy().into_owned())
9650            .collect();
9651        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
9652        assert!(
9653            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
9654            "daemon-private capture key leaked to the child: {keys:?}"
9655        );
9656    }
9657}
9658
9659#[cfg(test)]
9660mod jitter_tests {
9661    use super::jittered_health_delay;
9662    use std::{collections::HashSet, time::Duration};
9663
9664    /// Module ids drawn from a real fleet, so the dispersal claim is about names
9665    /// that actually occur rather than invented ones.
9666    ///
9667    /// This is a SAMPLE, not a registry: the property under test is that distinct
9668    /// ids disperse, which holds for any set of distinct strings. Several entries
9669    /// are already historical (modules get renamed), and that costs nothing here --
9670    /// but it means a reader must not mistake this for the live module set, and a
9671    /// rename sweep will match it without there being anything to change.
9672    const FLEET: [&str; 14] = [
9673        "aft",
9674        "alfonso-core",
9675        "magic-context",
9676        "broca",
9677        "thalamus",
9678        "quota",
9679        "engram",
9680        "plexus",
9681        "cerebellum",
9682        "astrocyte",
9683        "synapse",
9684        "subc-mcp",
9685        "cortexkit-credentials",
9686        "subc-federation",
9687    ];
9688
9689    /// Probes must not converge after a fleet-wide restart.
9690    ///
9691    /// This is the property the jitter exists for: every module reconnects at
9692    /// once, and without dispersal all fourteen would then probe on the same
9693    /// tick forever. Nothing failed visibly when this went untested -- a
9694    /// convergent fleet still probes correctly, just in a burst, so the symptom
9695    /// is a periodic load spike that looks like whatever else is running.
9696    #[test]
9697    fn probe_delays_disperse_across_the_fleet() {
9698        let cadence = Duration::from_secs(30);
9699        let delays: HashSet<Duration> = FLEET
9700            .iter()
9701            .map(|id| jittered_health_delay(id, 0, cadence))
9702            .collect();
9703        assert_eq!(
9704            delays.len(),
9705            FLEET.len(),
9706            "every supervised module must land on its own probe offset"
9707        );
9708    }
9709
9710    /// The offset may only ever DELAY a probe, never bring it forward.
9711    ///
9712    /// A delay below the cadence would probe a module more often than
9713    /// configured, which is the opposite of what an operator asked for and
9714    /// would tighten the failure budget without anyone changing it.
9715    #[test]
9716    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
9717        let cadence = Duration::from_secs(30);
9718        let span = cadence / 10;
9719        for id in FLEET {
9720            for probe_index in 0..8 {
9721                let delay = jittered_health_delay(id, probe_index, cadence);
9722                assert!(
9723                    delay >= cadence,
9724                    "{id}#{probe_index}: jitter must not shorten the cadence"
9725                );
9726                assert!(
9727                    delay < cadence + span,
9728                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
9729                );
9730            }
9731        }
9732    }
9733
9734    /// A module keeps its offset across daemon restarts.
9735    ///
9736    /// The delay is derived rather than randomised precisely so a restart does
9737    /// not re-roll every module into a fresh chance of collision. A random
9738    /// source would satisfy the dispersal test above and quietly lose this.
9739    #[test]
9740    fn a_module_offset_is_stable_across_restarts() {
9741        let cadence = Duration::from_secs(30);
9742        for id in FLEET {
9743            assert_eq!(
9744                jittered_health_delay(id, 0, cadence),
9745                jittered_health_delay(id, 0, cadence),
9746                "{id}: the same module and probe index must produce the same offset"
9747            );
9748        }
9749    }
9750
9751    /// A zero cadence disables probing rather than producing a busy loop.
9752    #[test]
9753    fn zero_cadence_yields_zero_delay() {
9754        assert_eq!(
9755            jittered_health_delay("aft", 0, Duration::ZERO),
9756            Duration::ZERO
9757        );
9758    }
9759}
9760
9761#[cfg(all(test, target_os = "linux"))]
9762mod cgroup_placement_tests {
9763    use super::{
9764        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
9765        SupervisedChild,
9766    };
9767    use crate::stderr_tail::{StderrRing, StderrTailConfig};
9768    use std::{
9769        fs, io,
9770        path::{Path, PathBuf},
9771        sync::{Arc, Mutex},
9772    };
9773    use subc_test_support::TestTempDir;
9774    use tokio::process::Command;
9775
9776    #[test]
9777    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
9778        let path = Path::new("/definitely-missing-subc-cgroup");
9779        let mut command = Command::new("true");
9780        let error = apply_cgroup_placement(
9781            &mut command,
9782            &ModuleSpec {
9783                module_id: "broken-cgroup".to_string(),
9784                program: PathBuf::from("true"),
9785                args: Vec::new(),
9786                env: Vec::new(),
9787                reserved: false,
9788                reserved_prefixes: Vec::new(),
9789                protocol: ModuleProtocol::Subc,
9790                overlap: Default::default(),
9791            },
9792            path,
9793        )
9794        .expect_err("a parent cgroup open failure must reject the supervised spawn");
9795        let reason = error.to_string();
9796
9797        assert!(
9798            matches!(error, SuperviseError::Cgroup { .. }),
9799            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
9800        );
9801        assert!(
9802            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
9803            "parent cgroup open failure must name cgroup.procs: {reason}"
9804        );
9805    }
9806
9807    #[tokio::test]
9808    async fn reaping_a_child_removes_its_empty_module_cgroup() {
9809        let root = TestTempDir::new("supervisor-reap-cgroup");
9810        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9811        let placement = subc_cgroup::prepare_at(&root)
9812            .expect("prepare scratch cgroup root")
9813            .expect("scratch root has a cgroup.procs marker");
9814        let module_id = "reaped-module";
9815        let module = placement
9816            .module_path(module_id)
9817            .expect("create scratch module cgroup");
9818        let child = Command::new("true")
9819            .spawn()
9820            .expect("spawn short-lived child");
9821        let pid = child.id().expect("spawned child has pid");
9822        let mut child = SupervisedChild {
9823            child,
9824            module_id: module_id.to_string(),
9825            cgroup_placement: Some(placement),
9826            stdout_pump: None,
9827            stderr_pump: None,
9828            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
9829            spawned_at_ms: 0,
9830            spawned_from: PathBuf::from("true"),
9831            spawned_file_identity: None,
9832            process_start_time: None,
9833            process_identity: None,
9834            pid,
9835            roster_guard: None,
9836        };
9837
9838        child.wait().await.expect("reap short-lived child");
9839
9840        assert!(
9841            !module.exists(),
9842            "reaping the supervised child must remove its empty cgroup"
9843        );
9844    }
9845
9846    #[test]
9847    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
9848        let root = TestTempDir::new("supervisor-non-empty-cgroup");
9849        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9850        let placement = subc_cgroup::prepare_at(&root)
9851            .expect("prepare scratch cgroup root")
9852            .expect("scratch root has a cgroup.procs marker");
9853        let module = placement
9854            .module_path("surviving-module")
9855            .expect("create scratch module cgroup");
9856        fs::write(module.join("surviving-process"), b"still present")
9857            .expect("make scratch cgroup non-empty");
9858        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
9859
9860        remove_module_cgroup(&placement, "surviving-module");
9861
9862        let logs = crate::router::test_log::captured_logs(&logs);
9863        assert!(
9864            module.exists(),
9865            "failed removal must leave the cgroup intact"
9866        );
9867        assert!(
9868            logs.contains("could not remove module cgroup after process exit; continuing teardown")
9869                && logs.contains("surviving-module"),
9870            "best-effort removal must report the failure without returning it: {logs}"
9871        );
9872    }
9873
9874    #[test]
9875    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
9876        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
9877        let reason = SuperviseError::Spawn {
9878            program: PathBuf::from("/bin/true"),
9879            source: io::Error::from_raw_os_error(13),
9880            cgroup_path: Some(cgroup_path.clone()),
9881        }
9882        .to_string();
9883
9884        assert!(
9885            reason.contains(&cgroup_path.display().to_string()),
9886            "a pre_exec spawn failure must name the cgroup path: {reason}"
9887        );
9888    }
9889}
9890
9891#[cfg(test)]
9892mod spawn_subscriber_lag_tests {
9893    use super::*;
9894
9895    /// A subscriber whose connection stops draining is dropped once its frame
9896    /// channel fills. The client must learn that from a terminal Error frame
9897    /// after the frames already queued for it, not from a stream that simply
9898    /// goes quiet.
9899    #[tokio::test]
9900    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
9901        let feed = SpawnEventFeed::default();
9902        feed.configure_incarnation("lag-incarnation".to_string());
9903        // A one-slot connection queue that nobody reads until the emits are
9904        // done: the forwarder parks on it and the subscriber channel fills.
9905        let (tx, mut rx) = mpsc::channel(1);
9906        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
9907            .expect("subscribe");
9908        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
9909        for index in 0..emitted {
9910            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
9911            // Let the forwarder take what it can so the fill point is the
9912            // subscriber channel, not a scheduling accident.
9913            tokio::task::yield_now().await;
9914        }
9915        assert_eq!(
9916            feed.subscriber_count(),
9917            0,
9918            "the lagged subscriber must be removed"
9919        );
9920
9921        let mut data = Vec::new();
9922        let mut last = None;
9923        loop {
9924            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
9925                .await
9926                .expect("the forwarder must finish once the subscriber is dropped");
9927            let Some(outbound) = next else { break };
9928            let frame = outbound.frame;
9929            if frame.header.ty == FrameType::StreamData {
9930                assert!(last.is_none(), "no data may follow the terminal frame");
9931                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
9932                data.push(event.cursor.seq);
9933            } else {
9934                assert!(last.is_none(), "exactly one terminal frame");
9935                last = Some(frame);
9936            }
9937        }
9938        assert!(!data.is_empty(), "queued frames drain before the terminal");
9939        for pair in data.windows(2) {
9940            assert_eq!(
9941                pair[1],
9942                pair[0] + 1,
9943                "queued frames arrive dense and in order"
9944            );
9945        }
9946        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
9947        assert_eq!(terminal.header.ty, FrameType::Error);
9948        assert_eq!(terminal.header.corr, 7);
9949        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
9950        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
9951        let detail = body.detail.expect("lagged error carries detail");
9952        assert_eq!(
9953            detail["first_undelivered_cursor"]["seq"],
9954            data.last().unwrap() + 1,
9955            "the named cursor is the first event the subscriber did not receive"
9956        );
9957        assert_eq!(
9958            detail["first_undelivered_cursor"]["daemon_incarnation"],
9959            "lag-incarnation"
9960        );
9961    }
9962}
9963
9964#[cfg(test)]
9965mod terminal_history_read_concurrency_tests {
9966    use super::*;
9967    use crate::terminal_journal::read_pause;
9968    use std::sync::mpsc as std_mpsc;
9969    use subc_test_support::TestTempDir;
9970
9971    fn journaled_ring(
9972        journal: &Arc<crate::terminal_journal::TerminalJournal>,
9973    ) -> Arc<Mutex<TerminalRing>> {
9974        Arc::new(Mutex::new(
9975            TerminalRing::new(TerminalRingConfig::default(), 1)
9976                .with_journal(Some(Arc::clone(journal))),
9977        ))
9978    }
9979
9980    fn crash(at_ms: u64) -> ExitReport {
9981        ExitReport {
9982            kind: ExitKind::Crash,
9983            code: Some(1),
9984            signal: None,
9985            at_ms,
9986        }
9987    }
9988
9989    /// Record an exit on another thread and report whether it finished within
9990    /// `bound`. The recorder thread is left running if it did not.
9991    fn record_within(
9992        module_id: &'static str,
9993        ring: &Arc<Mutex<TerminalRing>>,
9994        at_ms: u64,
9995        bound: Duration,
9996    ) -> bool {
9997        let ring = Arc::clone(ring);
9998        let (done, done_rx) = std_mpsc::channel();
9999        std::thread::spawn(move || {
10000            record_terminal(
10001                module_id,
10002                &ring,
10003                &SpawnEventFeed::default(),
10004                &crash(at_ms),
10005                TerminalDisposition::Restarting,
10006            );
10007            let _ = done.send(());
10008        });
10009        done_rx.recv_timeout(bound).is_ok()
10010    }
10011
10012    /// A history read in progress must not hold the journal writer (which every
10013    /// module's exit recording needs) or the module's own ring. Exits recorded
10014    /// while the read is paused complete promptly; the paused read answers as of
10015    /// the moment it started, and the next read has each exit exactly once.
10016    #[test]
10017    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
10018        let dir = TestTempDir::new("terminal-history-concurrent-read");
10019        let path = dir.join("terminals.jsonl");
10020        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
10021            path.clone(),
10022            "daemon".into(),
10023        ));
10024        let reader_ring = journaled_ring(&journal);
10025        let other_ring = journaled_ring(&journal);
10026        assert!(record_within(
10027            "reader-module",
10028            &reader_ring,
10029            10,
10030            Duration::from_secs(5)
10031        ));
10032
10033        let (started, release) = read_pause::install(&path);
10034        let reading = {
10035            let ring = Arc::clone(&reader_ring);
10036            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
10037        };
10038        started
10039            .recv_timeout(Duration::from_secs(5))
10040            .expect("the history read reached its pause");
10041
10042        let bound = Duration::from_secs(1);
10043        assert!(
10044            record_within("other-module", &other_ring, 20, bound),
10045            "another module's exit waited on a history read (journal writer held)"
10046        );
10047        assert!(
10048            record_within("reader-module", &reader_ring, 30, bound),
10049            "the read module's own exit waited on its history read (ring held)"
10050        );
10051
10052        drop(release);
10053        let paused = reading.join().unwrap();
10054        assert_eq!(
10055            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10056            vec![10],
10057            "an exit recorded after the read began lands in neither half of it"
10058        );
10059        assert_eq!(paused.journal_skipped_lines, 0);
10060        assert_eq!(paused.journal_read_errors, 0);
10061
10062        let after = durable_terminal_history_of(&reader_ring, "reader-module");
10063        assert_eq!(
10064            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10065            vec![10, 30],
10066            "the next read merges ring and journal with no duplicate"
10067        );
10068        assert_eq!(after.journal_skipped_lines, 0);
10069    }
10070}
10071
10072/// What a restart does with the exited process's stderr reader. These drive
10073/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
10074/// holds, so a reader that has not been scheduled by the bound is a controlled
10075/// input rather than something only a loaded machine produces.
10076#[cfg(test)]
10077mod stderr_settle_tests {
10078    use std::{
10079        future::Future,
10080        io,
10081        pin::Pin,
10082        sync::{Arc, Mutex},
10083        task::{Context, Poll},
10084        time::Duration,
10085    };
10086
10087    use tokio::{
10088        io::{AsyncRead, ReadBuf},
10089        sync::oneshot,
10090        time::Instant,
10091    };
10092
10093    use super::{settle_stderr_pump, StderrPump};
10094    use crate::stderr_tail::{
10095        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
10096    };
10097
10098    const BOUND: Duration = Duration::from_millis(250);
10099
10100    /// Yields `before`, then stays pending until the gate is released, then
10101    /// yields `after` and reaches EOF. The bytes after the gate were written
10102    /// by a process that has already exited; only the reader is behind.
10103    struct HeldReader {
10104        before: Option<Vec<u8>>,
10105        gate: Option<oneshot::Receiver<()>>,
10106        after: io::Cursor<Vec<u8>>,
10107    }
10108
10109    impl AsyncRead for HeldReader {
10110        fn poll_read(
10111            mut self: Pin<&mut Self>,
10112            cx: &mut Context<'_>,
10113            buf: &mut ReadBuf<'_>,
10114        ) -> Poll<io::Result<()>> {
10115            if let Some(bytes) = self.before.take() {
10116                buf.put_slice(&bytes);
10117                return Poll::Ready(Ok(()));
10118            }
10119            if let Some(gate) = self.gate.as_mut() {
10120                match Pin::new(gate).poll(cx) {
10121                    Poll::Pending => return Poll::Pending,
10122                    Poll::Ready(_) => self.gate = None,
10123                }
10124            }
10125            Pin::new(&mut self.after).poll_read(cx, buf)
10126        }
10127    }
10128
10129    struct DiscardSink;
10130
10131    impl OutputSink for DiscardSink {
10132        fn write_line(&mut self, _line: &[u8]) {}
10133    }
10134
10135    fn line(text: &str) -> TailEntry {
10136        TailEntry::Line {
10137            text: text.to_string(),
10138            truncated: false,
10139            at_ms: None,
10140        }
10141    }
10142
10143    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
10144        ring.lock().unwrap()
10145    }
10146
10147    /// Start a reader for a new process generation that delivers `before`
10148    /// immediately and `after` only once the returned sender fires (or is
10149    /// dropped).
10150    fn held_pump(
10151        ring: &Arc<Mutex<StderrRing>>,
10152        before: &str,
10153        after: &str,
10154    ) -> (StderrPump, oneshot::Sender<()>) {
10155        let generation = lock(ring).begin_process();
10156        let (release, gate) = oneshot::channel();
10157        let reader = HeldReader {
10158            before: Some(before.as_bytes().to_vec()),
10159            gate: Some(gate),
10160            after: io::Cursor::new(after.as_bytes().to_vec()),
10161        };
10162        let task = tokio::spawn(pump_stderr_to(
10163            reader,
10164            Arc::clone(ring),
10165            generation,
10166            DiscardSink,
10167        ));
10168        (StderrPump { task, generation }, release)
10169    }
10170
10171    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
10172        for _ in 0..1000 {
10173            if done(&lock(ring)) {
10174                return;
10175            }
10176            tokio::time::sleep(Duration::from_millis(1)).await;
10177        }
10178        panic!(
10179            "ring never reached the expected state: {:?}",
10180            lock(ring).snapshot(None, None)
10181        );
10182    }
10183
10184    #[tokio::test(start_paused = true)]
10185    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
10186        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10187        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
10188
10189        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
10190        let before_release = lock(&ring).snapshot(None, None);
10191        assert!(
10192            matches!(before_release.capture, CaptureState::Incomplete { .. }),
10193            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
10194        );
10195
10196        // The restart: the next process starts and writes before the old
10197        // reader catches up.
10198        let next = lock(&ring).begin_process();
10199        lock(&ring).push_line_from(next, "next process booting");
10200        release.send(()).unwrap();
10201        wait_until(&ring, |ring| {
10202            ring.snapshot(None, None).capture == CaptureState::Captured
10203        })
10204        .await;
10205
10206        assert_eq!(
10207            untimed(lock(&ring).snapshot(None, None).entries),
10208            vec![
10209                line("booting"),
10210                line("config error: missing storage"),
10211                TailEntry::ProcessStart,
10212                line("next process booting"),
10213            ],
10214            "the crash's last line must survive a slow reader and stay in the crashed process's section"
10215        );
10216    }
10217
10218    #[tokio::test(start_paused = true)]
10219    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
10220    ) {
10221        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10222        // `_held` is never fired: a descendant keeps the pipe open for the
10223        // whole test.
10224        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
10225
10226        let started = Instant::now();
10227        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
10228        assert_eq!(
10229            started.elapsed(),
10230            BOUND,
10231            "the restart must wait exactly the bound for a pipe that stays open, no longer"
10232        );
10233
10234        let next = lock(&ring).begin_process();
10235        lock(&ring).push_line_from(next, "next process booting");
10236        tokio::time::sleep(Duration::from_secs(60)).await;
10237
10238        let snapshot = lock(&ring).snapshot(None, None);
10239        match &snapshot.capture {
10240            CaptureState::Incomplete { reason } => assert!(
10241                reason.contains("had not reached EOF") && reason.contains("250ms"),
10242                "the reason must say what is missing and after how long: {reason}"
10243            ),
10244            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
10245        }
10246        assert_eq!(
10247            untimed(snapshot.entries),
10248            vec![
10249                line("parent exiting"),
10250                TailEntry::ProcessStart,
10251                line("next process booting"),
10252            ]
10253        );
10254    }
10255
10256    #[tokio::test(start_paused = true)]
10257    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
10258        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10259        let (pump, release) = held_pump(&ring, "one\n", "two\n");
10260        release.send(()).unwrap();
10261
10262        settle_stderr_pump("clean", &ring, pump, BOUND).await;
10263
10264        let snapshot = lock(&ring).snapshot(None, None);
10265        assert_eq!(snapshot.capture, CaptureState::Captured);
10266        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
10267    }
10268}
10269
10270/// Containment of a module's process tree (issue #109).
10271///
10272/// The behaviour these defend against is a module helper surviving its module:
10273/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
10274/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
10275/// compounds it.
10276///
10277/// They run against the SUPERVISOR rather than the job-object crate because the
10278/// claim is about teardown: a crate-level test proves a job can reap a tree, not
10279/// that the daemon's drain path reaches it.
10280///
10281/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
10282/// lane there is a separate containment path with its own tests.
10283#[cfg(all(test, windows))]
10284mod job_containment_tests {
10285    use super::*;
10286    use std::{
10287        path::{Path, PathBuf},
10288        sync::{Arc, Mutex},
10289        time::{Duration, Instant},
10290    };
10291    use subc_test_support::TestTempDir;
10292
10293    /// The stub, expected beside this test executable.
10294    ///
10295    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
10296    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
10297    /// failure then reads as a broken test rather than an unbuilt dependency.
10298    fn stub_path() -> PathBuf {
10299        let mut path = std::env::current_exe().expect("current_exe available in tests");
10300        path.pop();
10301        path.pop();
10302        path.push("fake-aft-stub.exe");
10303        assert!(
10304            path.exists(),
10305            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
10306             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
10307            path.display()
10308        );
10309        path
10310    }
10311
10312    /// Poll for the grandchild pid the stub records, and parse it.
10313    fn read_grandchild_pid(path: &Path) -> u32 {
10314        let deadline = Instant::now() + Duration::from_secs(10);
10315        loop {
10316            if let Ok(contents) = std::fs::read_to_string(path) {
10317                if let Ok(pid) = contents.trim().parse() {
10318                    return pid;
10319                }
10320            }
10321            assert!(
10322                Instant::now() < deadline,
10323                "the stub never recorded a grandchild pid at {}",
10324                path.display()
10325            );
10326            std::thread::sleep(Duration::from_millis(10));
10327        }
10328    }
10329
10330    /// Everything one fixture run needs, so the two tests below differ in exactly
10331    /// one place: whether the child is contained.
10332    struct Fixture {
10333        _dir: TestTempDir,
10334        module_id: String,
10335        grandchild: u32,
10336        child: Option<SupervisedChild>,
10337        registry: Arc<Registry>,
10338        snapshot: Arc<Mutex<SupervisorSnapshot>>,
10339        terminal_ring: Arc<Mutex<TerminalRing>>,
10340        spawn_events: SpawnEventFeed,
10341    }
10342
10343    fn fixture(label: &str, module_id: &str) -> Fixture {
10344        let dir = TestTempDir::new(label);
10345        let pid_file = dir.join("grandchild.pid");
10346        let supervisor = Supervisor::new(
10347            Arc::new(Registry::default()),
10348            RestartPolicy::new(3, Duration::ZERO),
10349        );
10350        let runtime = supervisor.runtime_config();
10351        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10352        let spec = ModuleSpec {
10353            module_id: module_id.to_string(),
10354            program: stub_path(),
10355            // Zero args deliberately: a `--subc` argument would make the stub dial
10356            // a daemon that is not there, and the failure would land in the same
10357            // stderr ring this fixture exists to keep quiet.
10358            args: Vec::new(),
10359            env: vec![
10360                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10361                (
10362                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
10363                    pid_file.display().to_string(),
10364                ),
10365            ],
10366            reserved: false,
10367            reserved_prefixes: Vec::new(),
10368            protocol: ModuleProtocol::Subc,
10369            overlap: Default::default(),
10370        };
10371        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
10372            .expect("spawn the supervised fixture");
10373        let grandchild = read_grandchild_pid(&pid_file);
10374        Fixture {
10375            _dir: dir,
10376            module_id: module_id.to_string(),
10377            grandchild,
10378            child: Some(child),
10379            registry: Arc::new(Registry::default()),
10380            snapshot,
10381            terminal_ring: Arc::clone(&runtime.terminal_ring),
10382            spawn_events: SpawnEventFeed::default(),
10383        }
10384    }
10385
10386    impl Fixture {
10387        /// Drain through the supervisor's own teardown path.
10388        async fn drain(&mut self) {
10389            let child = self
10390                .child
10391                .take()
10392                .expect("the fixture child is still present");
10393            drain_child_to_state(
10394                &self.module_id,
10395                ModuleProtocol::Subc,
10396                // No forwarding table in this fixture, so nothing reaches the
10397                // child over a connection.
10398                StopNotice::NotSent,
10399                &self.registry,
10400                &self.snapshot,
10401                &self.terminal_ring,
10402                &self.spawn_events,
10403                child,
10404                Duration::from_millis(500),
10405                ModuleState::Stopped,
10406                Some(false),
10407            )
10408            .await
10409            .expect("drain the supervised fixture");
10410        }
10411    }
10412
10413    /// Teardown reaps the grandchild, not merely the direct child.
10414    ///
10415    /// This is the assertion the change exists for. Before containment the
10416    /// grandchild survived: it is a separate process, and `start_kill` is
10417    /// `TerminateProcess` scoped to one pid.
10418    #[tokio::test]
10419    async fn teardown_reaps_the_grandchild() {
10420        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
10421        let grandchild = fixture.grandchild;
10422
10423        assert!(
10424            subc_jobobject::process_exists(grandchild),
10425            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
10426        );
10427
10428        fixture.drain().await;
10429
10430        assert!(
10431            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10432            "grandchild {grandchild} outlived module teardown: the tree was not contained"
10433        );
10434    }
10435
10436    /// The mutation control: with containment withheld, the grandchild survives
10437    /// the same kill.
10438    ///
10439    /// This is the defect reproduction from #109 — a direct-child kill reaches
10440    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
10441    /// supervisor because `spawn_and_mark_running` now always contains on
10442    /// Windows, which is the point: there is no longer a path that spawns
10443    /// uncontained, so the control has to construct one.
10444    ///
10445    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
10446    /// grandchild ever dies here, that test is passing for a reason unrelated to
10447    /// the job object and the containment claim is unproven.
10448    #[test]
10449    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
10450        let dir = TestTempDir::new("teardown-uncontained");
10451        let pid_file = dir.join("grandchild.pid");
10452        let mut child = std::process::Command::new(stub_path())
10453            .env("FAKE_AFT_NEVER_CONNECT", "1")
10454            .env(
10455                "FAKE_AFT_GRANDCHILD_PID_FILE",
10456                pid_file.display().to_string(),
10457            )
10458            .stdin(std::process::Stdio::null())
10459            .stdout(std::process::Stdio::null())
10460            .stderr(std::process::Stdio::null())
10461            .spawn()
10462            .expect("spawn the uncontained fixture");
10463        let grandchild = read_grandchild_pid(&pid_file);
10464
10465        // Exactly what the pre-fix teardown did: kill the direct child.
10466        child.kill().expect("kill the direct child");
10467        let _ = child.wait();
10468
10469        assert!(
10470            subc_jobobject::process_exists(grandchild),
10471            "grandchild {grandchild} died with the direct child, so this control no longer \
10472             distinguishes contained from uncontained teardown and the regression test is \
10473             passing vacuously"
10474        );
10475
10476        // The orphan this control demonstrates is the leak the fix prevents, so
10477        // the control must not leave one behind.
10478        kill_tree(grandchild);
10479    }
10480
10481    /// Crash durability: closing the containment handle reaps the tree with no
10482    /// teardown code running at all.
10483    ///
10484    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
10485    /// call anything — and it is why containment is a kernel property of the
10486    /// handle rather than a step in the drain. Discovered by getting the
10487    /// mutation control wrong: clearing `job` to "disable" containment instead
10488    /// killed the tree, which is the guarantee, not a mistake.
10489    #[tokio::test]
10490    async fn dropping_containment_reaps_the_grandchild() {
10491        let mut fixture = fixture("drop-containment", "tree-drop");
10492        let grandchild = fixture.grandchild;
10493
10494        assert!(subc_jobobject::process_exists(grandchild));
10495
10496        // No `drain` call, no kill: dropping the handle is the entire mechanism.
10497        fixture.child.as_mut().expect("child present").job = None;
10498
10499        assert!(
10500            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10501            "grandchild {grandchild} survived the containment handle closing, so a daemon \
10502             crash would leave the tree behind"
10503        );
10504    }
10505
10506    /// Kill a pid and its tree, then confirm it is gone.
10507    fn kill_tree(pid: u32) {
10508        let _ = std::process::Command::new("taskkill.exe")
10509            .args(["/PID", &pid.to_string(), "/T", "/F"])
10510            .stdin(std::process::Stdio::null())
10511            .stdout(std::process::Stdio::null())
10512            .stderr(std::process::Stdio::null())
10513            .status();
10514        assert!(
10515            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
10516            "could not clean up grandchild {pid}"
10517        );
10518    }
10519}
10520
10521/// The daemon's real spawn path hands a subc-wire child its launch nonce on
10522/// descriptor 3, without an environment copy. The shell records the nonce
10523/// and its environment after exec so these tests observe the real handover.
10524#[cfg(all(test, unix))]
10525mod launch_nonce_descriptor_tests {
10526    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
10527    use crate::stderr_tail::{StderrRing, StderrTailConfig};
10528    use std::{
10529        path::PathBuf,
10530        sync::{Arc, Mutex},
10531        time::{Duration, Instant},
10532    };
10533    use subc_test_support::TestTempDir;
10534
10535    async fn probe(role: super::SpawnRole) {
10536        let scratch = TestTempDir::new("launch-nonce-descriptor");
10537        let fd_copy = scratch.join("from-descriptor");
10538        let env_copy = scratch.join("environment");
10539        let script = format!(
10540            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
10541            fd = fd_copy.display(), env = env_copy.display(),
10542        );
10543        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
10544        let spec = ModuleSpec {
10545            module_id: "nonce-descriptor-probe".to_string(),
10546            program: PathBuf::from("/bin/sh"),
10547            args: vec!["-c".to_string(), script],
10548            env: vec![
10549                xdg("XDG_DATA_HOME"),
10550                xdg("XDG_RUNTIME_DIR"),
10551                xdg("XDG_CONFIG_HOME"),
10552                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
10553            ],
10554            reserved: true,
10555            reserved_prefixes: Vec::new(),
10556            protocol: ModuleProtocol::Subc,
10557            overlap: Default::default(),
10558        };
10559        let handle = SupervisorHandle::new();
10560        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10561        let roster = ChildRoster::default();
10562        let child = super::spawn_child_in_slot(
10563            &spec,
10564            None,
10565            Some(&handle),
10566            &ring,
10567            None,
10568            &roster,
10569            #[cfg(target_os = "linux")]
10570            None,
10571            role,
10572            matches!(role, super::SpawnRole::SwapCandidate),
10573        )
10574        .expect("spawn probe");
10575        let deadline = Instant::now() + Duration::from_secs(10);
10576        while !(fd_copy.exists() && env_copy.exists()) {
10577            assert!(Instant::now() < deadline, "probe never wrote its copies");
10578            tokio::time::sleep(Duration::from_millis(20)).await;
10579        }
10580        let nonce = std::fs::read_to_string(fd_copy).unwrap();
10581        assert!(!nonce.is_empty());
10582        let environment = std::fs::read_to_string(env_copy).unwrap();
10583        assert!(environment
10584            .lines()
10585            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
10586        let copy = environment
10587            .lines()
10588            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
10589        assert_eq!(
10590            copy, None,
10591            "Unix children must never receive the environment nonce"
10592        );
10593        if matches!(role, super::SpawnRole::Plain) {
10594            assert_eq!(
10595                handle.spawn_nonce(&spec.module_id).as_deref(),
10596                Some(nonce.as_str())
10597            );
10598        }
10599        drop(child);
10600    }
10601
10602    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10603    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
10604        probe(super::SpawnRole::Plain).await;
10605    }
10606
10607    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10608    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
10609        probe(super::SpawnRole::SwapCandidate).await;
10610    }
10611}