Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113struct SupervisedChild {
114    child: Child,
115    /// The name of this process's cgroup: the module id, or for a swap
116    /// candidate the alternate name (see `swap::cgroup_name`).
117    #[cfg(target_os = "linux")]
118    module_id: String,
119    #[cfg(target_os = "linux")]
120    cgroup_placement: Option<subc_cgroup::Placement>,
121    /// The job that contains this child and every process it spawns (issue #109).
122    ///
123    /// Dropping this handle is what reaps a surviving tree when no supervisor
124    /// code runs — a daemon crash — because the job carries
125    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
126    ///
127    /// That limit is not crash-only, and the difference is worth knowing: a
128    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
129    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
130    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
131    /// module at once. Before this change they survived that, saw EOF on the
132    /// control socket, and ran their own teardown; Unix keeps that path
133    /// deliberately, so a module can seal a WAL or close a capture rather than
134    /// be killed mid-write. So this trades graceful teardown on every Windows
135    /// daemon stop for containment on a crash, which is the right way round
136    /// today: orphaned GPU workers are a reported, recurring problem, and the
137    /// modules that write most heavily do not run on Windows.
138    ///
139    /// The fix is a real Windows stop path — the daemon draining before it
140    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
141    /// reaches only what the drain left behind, which is what it should reach.
142    #[cfg(windows)]
143    job: Option<subc_jobobject::JobObject>,
144    stdout_pump: Option<JoinHandle<()>>,
145    stderr_pump: Option<StderrPump>,
146    stderr_ring: Arc<Mutex<StderrRing>>,
147    spawned_at_ms: u64,
148    spawned_from: PathBuf,
149    spawned_file_identity: Option<SpawnedFileIdentity>,
150    process_start_time: Option<u64>,
151    process_identity: Option<ProcessIdentity>,
152    pid: u32,
153    /// This process's entry in the daemon's child roster, released when the
154    /// process is reaped or this handle is dropped.
155    roster_guard: Option<crate::child_roster::RosterGuard>,
156}
157
158impl SupervisedChild {
159    fn id(&self) -> Option<u32> {
160        Some(self.pid)
161    }
162
163    fn process_identity(&self) -> Option<ProcessIdentity> {
164        self.process_identity
165    }
166
167    async fn wait(&mut self) -> io::Result<ExitStatus> {
168        // The roster entry is NOT released here. A daemon shutdown waits for the
169        // roster to empty and then exits the process, so releasing at the reap
170        // let it exit before the exit handler wrote this child's terminal record
171        // (the stderr drain and snapshot update sit in between), and the
172        // shutdown's own `daemon_shutdown` record was intermittently lost. The
173        // caller releases it after recording the exit (`release_roster`), and
174        // dropping the handle releases it too.
175        let result = self.child.wait().await;
176        #[cfg(target_os = "linux")]
177        if result.is_ok() {
178            if let Some(placement) = self.cgroup_placement.take() {
179                remove_module_cgroup(&placement, &self.module_id);
180            }
181        }
182        result
183    }
184
185    /// Releases this child's daemon-shutdown roster entry once its exit has
186    /// been recorded. The pid is already reaped and free for reuse, so the
187    /// entry must not outlive the record any longer than that.
188    fn release_roster(&mut self) {
189        self.roster_guard = None;
190    }
191
192    /// Kill the child, and on Windows the whole tree it spawned (issue #109).
193    ///
194    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
195    /// helper process leaked the helper — the Synapse embedding module's CUDA
196    /// worker holds the GPU allocation, so the leak cost VRAM until the next
197    /// restart of something else. Terminating the job reaches grandchildren that
198    /// a tree walk cannot, including one whose parent has already exited and
199    /// been reparented away.
200    ///
201    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
202    /// direct-child kill still decides the outcome, so containment can never
203    /// change whether a module is reported as stopped.
204    fn start_kill(&mut self) -> io::Result<()> {
205        #[cfg(windows)]
206        if let Some(job) = &self.job {
207            if let Err(error) = job.terminate() {
208                debug!(
209                    error = %error,
210                    "job termination failed; the direct-child kill still owns the outcome"
211                );
212            }
213        }
214        self.child.start_kill()
215    }
216
217    async fn drain_stderr(&mut self, module_id: &str) {
218        if let Some(mut pump) = self.stdout_pump.take() {
219            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
220                Ok(Ok(())) => {}
221                Ok(Err(error)) => {
222                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
223                }
224                Err(_) => {
225                    pump.abort();
226                    warn!(
227                        module_id,
228                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
229                        "stdout pump did not drain before restart; stopped it before the next process"
230                    );
231                }
232            }
233        }
234
235        let Some(pump) = self.stderr_pump.take() else {
236            return;
237        };
238        settle_stderr_pump(
239            module_id,
240            &self.stderr_ring,
241            pump,
242            STDERR_PUMP_DRAIN_TIMEOUT,
243        )
244        .await;
245    }
246}
247
248/// The reader task for one process's stderr, with the ring generation its
249/// lines are attributed to.
250struct StderrPump {
251    task: JoinHandle<()>,
252    generation: u64,
253}
254
255/// Retire an exited process's stderr reader and wait up to `bound` for it to
256/// reach EOF. A reader still running at the bound is detached, not stopped: it
257/// keeps filling the exited process's section of the ring until its pipe
258/// closes, and the tail reads `Incomplete` until then. See
259/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
260async fn settle_stderr_pump(
261    module_id: &str,
262    ring: &Arc<Mutex<StderrRing>>,
263    pump: StderrPump,
264    bound: Duration,
265) {
266    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
267    let StderrPump {
268        mut task,
269        generation,
270    } = pump;
271    lock().retire_pump(generation);
272    match timeout(bound, &mut task).await {
273        Ok(Ok(())) => {}
274        Ok(Err(err)) => {
275            let mut ring = lock();
276            ring.mark_incomplete(format!("stderr pump ended unexpectedly: {err}"));
277            ring.finish_pump(generation);
278            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
279        }
280        Err(_) => {
281            // Dropping the handle detaches the task; it ends at EOF on its pipe.
282            drop(task);
283            lock().mark_pump_late(
284                generation,
285                format!(
286                    "stderr of the exited process had not reached EOF {bound:?} after it was \
287                     retired (a descendant may still hold the pipe open); lines it still \
288                     writes are kept in that process's section"
289                ),
290            );
291            warn!(
292                module_id,
293                waited = ?bound,
294                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
295            );
296        }
297    }
298}
299
300fn registration_release_events() -> &'static watch::Sender<u64> {
301    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
302    EVENTS.get_or_init(|| {
303        let (sender, _receiver) = watch::channel(0);
304        sender
305    })
306}
307
308pub(crate) fn notify_registration_release() {
309    let events = registration_release_events();
310    let next_generation = (*events.borrow()).wrapping_add(1);
311    events.send_replace(next_generation);
312}
313
314/// How to launch one singleton module process.
315#[derive(Debug, Clone, PartialEq, Eq)]
316pub struct ModuleSpec {
317    pub module_id: String,
318    pub program: PathBuf,
319    pub args: Vec<String>,
320    pub env: Vec<(String, String)>,
321    /// When true this is a reserved module: each spawn gets a fresh one-time launch
322    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
323    /// process can register this module_id (a security-boundary module like the
324    /// credential vault must not be impersonable while it is down/restarting).
325    pub reserved: bool,
326    /// Include SUBC_LAUNCH_NONCE for older readers; Unix can pass the nonce only by pipe.
327    pub launch_nonce_env: bool,
328    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
329    /// Prefixes come from daemon config and must end in `:` before they reach the
330    /// supervisor; the owner module's current spawn nonce authorizes claims under
331    /// each prefix.
332    pub reserved_prefixes: Vec<String>,
333    /// The wire protocol this module speaks, as DECLARED in daemon config.
334    ///
335    /// [`ModuleProtocol::None`] changes five things and nothing else: health
336    /// probing is suppressed, teardown sends SIGTERM before waiting,
337    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
338    /// and NO launch nonce, and a clean exit the daemon did not request is
339    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
340    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
341    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
342    /// because a process ignores an environment variable it does not read.
343    ///
344    /// The argument is the part that cannot be "harmless to a process that
345    /// ignores it": a stock binary exits on an unknown flag before it listens
346    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
347    /// first conformance run against this mode found it. The nonce is withheld
348    /// because a process that will never present it gains nothing from holding
349    /// it, and a secret in the environment of a process that does not need it is
350    /// a leak surface for no benefit.
351    pub protocol: ModuleProtocol,
352    /// Whether two processes of this module may run at once, which is what a
353    /// blue/green swap does for the length of its overlap. Declared in daemon
354    /// config because the daemon must be able to answer it while the module is
355    /// down, and so a module cannot talk itself into it after registering.
356    pub overlap: ModuleOverlap,
357}
358
359/// Whether a module tolerates a second process of itself running alongside.
360///
361/// Most modules are single-writer on their store (a WAL, a capture log, a
362/// resident index behind a writer barrier), and two processes on one store
363/// corrupt it. So a swap, which overlaps the old and new process by design,
364/// is refused unless the module's config opts in with `overlap: "safe"`.
365#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
366pub enum ModuleOverlap {
367    /// Never run two processes of this module at once. The default.
368    #[default]
369    Exclusive,
370    /// The module has said a second process of itself is harmless for the
371    /// length of a swap.
372    ///
373    /// Declare it only if a second instance can run for a few seconds without
374    /// touching ANY single-writer store: every database, WAL, index, projector
375    /// and scheduled job the module owns. A lease on part of that state is not
376    /// enough. broca's session lease guards WAL appends while its run index, its
377    /// store projector and its archive fold timer (which unlinks live WAL files)
378    /// stay single-writer, so broca is exclusive despite holding a lease. The
379    /// refusal only fires after this has been decided, so the decision is the
380    /// check.
381    Safe,
382}
383
384impl ModuleOverlap {
385    pub fn as_str(self) -> &'static str {
386        match self {
387            Self::Exclusive => "exclusive",
388            Self::Safe => "safe",
389        }
390    }
391}
392
393/// Environment variable telling a spawned module which case it was started
394/// for, before it sends HELLO. Only a swap candidate carries it, as
395/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
396///
397/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
398/// longer because nobody waits on it, while a plain restart must flip ready
399/// quickly because callers see `module_warming` until it does. Absence means
400/// plain restart, the safe reading. The daemon trusts nothing about it; the
401/// candidate is proven by its launch nonce at HELLO.
402pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
403/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
404pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
405/// How long a swap waits for its candidate to register and declare itself
406/// ready when the operator does not say. A module warming as a swap candidate
407/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
408/// daemon allows that plus time to start the process and send HELLO.
409pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
410
411/// Bounded restart policy for crash exits.
412///
413/// `max_restarts` is the number of replacement processes allowed after the
414/// initial spawn WITHIN `window`. After that many crash restarts inside one
415/// window the module enters [`ModuleState::Failed`] and the supervisor stops
416/// the crash loop.
417///
418/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
419/// and that only survived because crashes were rare: a module that crashed
420/// three times across a week was disabled forever by crashes that had nothing
421/// to do with each other. That stopped being survivable once modules began
422/// exiting non-zero whenever the daemon's connection to them drops, because
423/// then every daemon-side connection drop spends a unit of the same budget and
424/// one flappy hour permanently stops a healthy module. Restarts older than
425/// `window` release their slot, so a module that crashed twice yesterday has a
426/// full budget today, while a genuine crash loop -- which is fast by
427/// definition -- still reaches the cap and stops.
428#[derive(Debug, Clone, Copy, PartialEq, Eq)]
429pub struct RestartPolicy {
430    pub max_restarts: u32,
431    /// Base delay before a crash replacement. The actual delay escalates with
432    /// the number of recent crash replacements and is capped by `max_backoff`.
433    pub backoff: Duration,
434    /// Maximum delay before a crash replacement.
435    pub max_backoff: Duration,
436    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
437    /// budget effectively infinite (nothing is ever in-window), which is why
438    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
439    pub window: Duration,
440}
441
442impl RestartPolicy {
443    /// A policy with the default crash window. Callers that care about the
444    /// window say so with [`Self::with_window`]; the ones that do not are
445    /// asking for the standard rate limit, not for no limit.
446    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
447        Self {
448            max_restarts,
449            backoff,
450            max_backoff: DEFAULT_MAX_BACKOFF,
451            window: DEFAULT_RESTART_WINDOW,
452        }
453    }
454
455    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
456        self.max_backoff = max_backoff;
457        self
458    }
459
460    pub fn with_window(mut self, window: Duration) -> Self {
461        self.window = window;
462        self
463    }
464
465    /// Calculate the capped exponential delay for the next crash replacement.
466    /// `restart_in_window` is zero for the first replacement after an operator
467    /// action (restart, reload, re-enable) cleared the crash ring, or after all
468    /// older crash replacements have aged out of the window.
469    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
470        if self.backoff.is_zero() || self.max_backoff.is_zero() {
471            return Duration::ZERO;
472        }
473
474        let mut delay = self.backoff;
475        for _ in 0..restart_in_window {
476            if delay >= self.max_backoff {
477                return self.max_backoff;
478            }
479            delay = delay
480                .checked_mul(10)
481                .unwrap_or(self.max_backoff)
482                .min(self.max_backoff);
483        }
484        delay.min(self.max_backoff)
485    }
486
487    /// The one sentence that explains a budget-exhausted stop, used for both the
488    /// log line and the terminal record so the two cannot drift. It names the
489    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
490    /// exactly what this budget is not.
491    fn budget_exhausted_detail(&self) -> String {
492        format!(
493            "crash budget exhausted: max_restarts={} within window_secs={}",
494            self.max_restarts,
495            self.window.as_secs()
496        )
497    }
498}
499
500impl Default for RestartPolicy {
501    fn default() -> Self {
502        Self {
503            max_restarts: DEFAULT_MAX_RESTARTS,
504            backoff: DEFAULT_BACKOFF,
505            max_backoff: DEFAULT_MAX_BACKOFF,
506            window: DEFAULT_RESTART_WINDOW,
507        }
508    }
509}
510
511#[derive(Debug, Clone, Copy, PartialEq, Eq)]
512struct CrashRestartSchedule {
513    restart_in_window: u32,
514    delay: Duration,
515}
516
517/// Whether the daemon itself will bring this module back after the exit being
518/// handled: it is enabled AND its in-window crash restarts are below the cap.
519///
520/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
521/// the window are dropped here rather than by a timer, so the count is right
522/// the moment somebody asks and no bookkeeping runs for idle modules.
523fn daemon_will_restart(
524    state: &mut SupervisorSnapshot,
525    policy: &RestartPolicy,
526    now: Instant,
527) -> bool {
528    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
529}
530
531const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
532const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
533const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
534const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
535
536#[derive(Debug, Clone, Copy, PartialEq, Eq)]
537pub enum HealthAction {
538    Report,
539    Restart,
540    Alert,
541}
542
543impl fmt::Display for HealthAction {
544    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
545        f.write_str(match self {
546            Self::Report => "report",
547            Self::Restart => "restart",
548            Self::Alert => "alert",
549        })
550    }
551}
552
553#[derive(Debug, Clone, Copy, PartialEq, Eq)]
554pub struct HealthConfig {
555    pub cadence: Duration,
556    pub deadline: Duration,
557    pub failure_threshold: u32,
558    pub on_degraded: HealthAction,
559    pub on_failing: HealthAction,
560    pub critical: bool,
561}
562
563impl Default for HealthConfig {
564    fn default() -> Self {
565        Self {
566            cadence: DEFAULT_HEALTH_CADENCE,
567            deadline: DEFAULT_HEALTH_DEADLINE,
568            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
569            on_degraded: HealthAction::Report,
570            on_failing: HealthAction::Report,
571            critical: false,
572        }
573    }
574}
575
576/// The supervisor's view of one module's health, relayed to clients over
577/// channel-0 and rendered by `ck health`.
578///
579/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
580/// stated here rather than only at the wire type a consumer reads. A reader can
581/// look up what `None` means; only a writer can silently change it, and the
582/// writer has no reason to go looking at a downstream contract before editing.
583///
584/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
585/// back to `None` on re-registration precisely so a respawned module does not
586/// carry its predecessor's timestamp — so an old value and an absent one call for
587/// opposite readings, and anything that defaulted this to a number would make a
588/// never-probed module indistinguishable from one probed at the epoch.
589///
590/// `detail` and `metrics` are `None` when the module published none on this
591/// probe, which does not mean it reported nothing wrong — it is also the shape
592/// when the probe never reached it. `last_probe_ms` is what separates those.
593#[derive(Debug, Clone, PartialEq)]
594pub struct ModuleHealthStatus {
595    pub status: SupervisorHealthStatus,
596    pub last_probe_ms: Option<u64>,
597    pub detail: Option<String>,
598    pub metrics: Option<Value>,
599    pub consecutive_failures: u32,
600    /// Number of replies received after a recurring health probe's deadline.
601    /// Unlike a timeout, every increment proves the module was alive.
602    pub late_answer_count: u64,
603    /// End-to-end latency of the newest late reply, measured from probe start.
604    pub last_late_answer_latency_ms: Option<u64>,
605    pub last_action: Option<String>,
606    /// Set together with `last_action`; the pair moves as one, and both being
607    /// absent means no escalation has ever been taken rather than that the last
608    /// one succeeded.
609    pub last_action_ms: Option<u64>,
610}
611
612impl Default for ModuleHealthStatus {
613    fn default() -> Self {
614        Self {
615            status: SupervisorHealthStatus::Unknown,
616            last_probe_ms: None,
617            detail: None,
618            metrics: None,
619            consecutive_failures: 0,
620            late_answer_count: 0,
621            last_late_answer_latency_ms: None,
622            last_action: None,
623            last_action_ms: None,
624        }
625    }
626}
627
628/// Typed lifecycle state for a supervised module.
629#[derive(Debug, Clone, Copy, PartialEq, Eq)]
630pub enum ModuleState {
631    Starting,
632    Running,
633    Unresponsive,
634    Restarting,
635    Draining,
636    Stopped,
637    Failed,
638    Disabled,
639}
640
641impl fmt::Display for ModuleState {
642    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
643        f.write_str(match self {
644            Self::Starting => "starting",
645            Self::Running => "running",
646            Self::Unresponsive => "unresponsive",
647            Self::Restarting => "restarting",
648            Self::Draining => "draining",
649            Self::Stopped => "stopped",
650            Self::Failed => "failed",
651            Self::Disabled => "disabled",
652        })
653    }
654}
655
656/// Supervisor classification of a child-process exit.
657#[derive(Debug, Clone, Copy, PartialEq, Eq)]
658pub enum ExitKind {
659    Clean,
660    Crash,
661    DeliberateSeverance,
662}
663
664impl From<ExitKind> for TerminalExitKind {
665    fn from(kind: ExitKind) -> Self {
666        match kind {
667            ExitKind::Clean => Self::Clean,
668            ExitKind::Crash => Self::Crash,
669            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
670        }
671    }
672}
673
674/// Exact process identity retained when a supervised module registers its
675/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
676#[derive(Debug, Clone, Copy, PartialEq, Eq)]
677pub(crate) struct ProcessIdentity {
678    pub(crate) pid: u32,
679    pub(crate) start_time: u64,
680}
681
682/// Last observed child exit, if any.
683#[derive(Debug, Clone, PartialEq, Eq)]
684pub struct ExitReport {
685    pub kind: ExitKind,
686    pub code: Option<i32>,
687    pub signal: Option<i32>,
688    pub at_ms: u64,
689}
690
691/// Point-in-time module status answerable by subc without forwarding to the
692/// module process.
693#[derive(Debug, Clone, PartialEq)]
694pub struct ModuleStatus {
695    pub module_id: String,
696    pub state: ModuleState,
697    pub enabled: bool,
698    pub process_alive: bool,
699    pub registration_active: bool,
700    /// The module's declared wire protocol, carried beside `live` because it is
701    /// what makes `live` readable: the two fields answer one question together.
702    pub protocol: ModuleProtocol,
703    /// Whether the module is serving, under the strongest definition the daemon
704    /// can assert for its protocol.
705    ///
706    /// A subc module must also be REGISTERED: its process being alive says
707    /// nothing about whether it can take a request. A `protocol: "none"` module
708    /// never registers, so that term is dropped and this falls back to "enabled,
709    /// running, and the process the daemon launched is alive" -- which is all
710    /// the daemon observes about a process that speaks no subc wire. It stays a
711    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
712    /// rather than printing it bare.
713    pub live: bool,
714    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
715    /// restarts have already released their slot, so this count can go down
716    /// without anybody touching the module.
717    pub restart_count: u32,
718    /// Replacement processes spawned over this module's entire supervisor lifetime;
719    /// unlike `restart_count`, this value is never reset by an operator action
720    /// and never falls out of a window.
721    pub lifetime_restarts: u32,
722    pub spawn_generation: u64,
723    /// The budget `restart_count` is spent against. Carried alongside the count
724    /// because the count alone does not say how close the module is to being
725    /// disabled, and reporting one without the other is what makes an
726    /// about-to-be-retired module look ordinary.
727    pub max_restarts: u32,
728    /// The span `restart_count` is counted over. Carried with the pair above for
729    /// the same reason they are carried together: "2 of 3" means one thing for a
730    /// ten-minute window and something else entirely for a lifetime.
731    pub restart_window: Duration,
732    /// Effective drain and restart timing policy used by this running module.
733    /// These values are carried together with the restart budget so status
734    /// readers can compare configured intent with what the supervisor applied.
735    pub drain_timeout: Duration,
736    pub restart_backoff: Duration,
737    pub restart_max_backoff: Duration,
738    pub pid: Option<u32>,
739    pub spawned_at_ms: Option<u64>,
740    pub spawned_from: Option<PathBuf>,
741    pub process_start_time: Option<u64>,
742    pub last_exit: Option<ExitReport>,
743    pub health: ModuleHealthStatus,
744}
745
746#[derive(Debug, Clone, PartialEq)]
747struct SupervisorSnapshot {
748    state: ModuleState,
749    enabled: bool,
750    process_alive: bool,
751    /// When each crash restart was spent, oldest first. This IS the crash
752    /// budget: its in-window length is the count an operator sees and the count
753    /// the restart decision is made against, so there is no second counter that
754    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
755    /// operator actions that used to zero the old lifetime counter.
756    crash_restarts: VecDeque<Instant>,
757    lifetime_restarts: u32,
758    /// Successful child spawns in this daemon incarnation.
759    ///
760    /// `lifetime_restarts` was considered and rejected: it starts at zero
761    /// (line 640), successful initial/operator spawns in `set_running` do not
762    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
763    /// increments before a successful replacement exists (lines 604, 3846,
764    /// and 3921), so a failed spawn can consume it. This counter moves only
765    /// when a live PID is accepted below.
766    spawn_generation: u64,
767    pid: Option<u32>,
768    spawned_at_ms: Option<u64>,
769    spawned_from: Option<PathBuf>,
770    spawned_file_identity: Option<SpawnedFileIdentity>,
771    process_start_time: Option<u64>,
772    deliberate_severance: Option<ProcessIdentity>,
773    last_exit: Option<ExitReport>,
774    health: ModuleHealthStatus,
775    /// Whether the current process was started as a swap candidate and so
776    /// lives in the module's alternate cgroup. The next swap's candidate takes
777    /// the other one, so the two processes of a swap never share a cgroup. A
778    /// plain spawn always uses the primary cgroup.
779    in_alternate_slot: bool,
780    /// Whether the current `Draining` state ends in a replacement process
781    /// (restart, reload, health restart) rather than a stop. Only meaningful
782    /// while `state` is `Draining`; every entry into that state rewrites it.
783    /// It is what lets route.open answer the retryable `module_reloading` to a
784    /// consumer that reaches a still-registered process mid-restart, instead of
785    /// the `supervisor_not_live` a stop or disable deserves.
786    draining_to_replace: bool,
787    /// Whether a configuration update has been applied since the current
788    /// process was spawned, so that process runs an older spec than the one
789    /// the supervisor now holds. A queued restart is only coalesced into a
790    /// fresher process when this is false: a restart requested to pick up a
791    /// new configuration must not be satisfied by a process that predates it.
792    configuration_updated_since_spawn: bool,
793}
794
795impl SupervisorSnapshot {
796    fn starting() -> Self {
797        Self::new(ModuleState::Starting, true)
798    }
799
800    fn disabled() -> Self {
801        Self::new(ModuleState::Disabled, false)
802    }
803
804    fn failed() -> Self {
805        Self::new(ModuleState::Failed, true)
806    }
807
808    /// Crash restarts still inside `window`, having dropped the ones that are
809    /// not. Pruning on read is what makes the budget a rate: an instant older
810    /// than the window stops holding a slot the moment anybody counts.
811    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
812        while let Some(oldest) = self.crash_restarts.front() {
813            if now.duration_since(*oldest) > window {
814                self.crash_restarts.pop_front();
815            } else {
816                break;
817            }
818        }
819        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
820    }
821
822    /// Spend one unit of the crash budget and record the restart in the ledger.
823    ///
824    /// The ring is bounded by the cap because more than `max_restarts` in-window
825    /// instants can never be reached (the caller refuses the restart first), so
826    /// anything beyond that is an unbounded queue waiting to happen.
827    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
828        self.crash_restarts.push_back(now);
829        while self.crash_restarts.len() > policy.max_restarts as usize {
830            self.crash_restarts.pop_front();
831        }
832        self.lifetime_restarts += 1;
833    }
834
835    /// Reserve one crash-restart slot and calculate the delay before respawning.
836    /// The count is captured before recording this restart, so the first retry
837    /// uses the base delay and each later in-window retry escalates once.
838    fn next_crash_restart(
839        &mut self,
840        policy: &RestartPolicy,
841        now: Instant,
842    ) -> Option<CrashRestartSchedule> {
843        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
844        if restart_in_window >= policy.max_restarts {
845            return None;
846        }
847        self.record_crash_restart(policy, now);
848        Some(CrashRestartSchedule {
849            restart_in_window,
850            delay: policy.delay_for_restart(restart_in_window),
851        })
852    }
853
854    /// Give the module its full budget back, as an operator restart, reload, or
855    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
856    /// ledger of what actually happened, and an operator action does not unmake
857    /// the crashes.
858    fn clear_crash_restarts(&mut self) {
859        self.crash_restarts.clear();
860    }
861
862    fn new(state: ModuleState, enabled: bool) -> Self {
863        Self {
864            state,
865            enabled,
866            process_alive: false,
867            crash_restarts: VecDeque::new(),
868            lifetime_restarts: 0,
869            spawn_generation: 0,
870            pid: None,
871            spawned_at_ms: None,
872            spawned_from: None,
873            spawned_file_identity: None,
874            process_start_time: None,
875            deliberate_severance: None,
876            last_exit: None,
877            health: ModuleHealthStatus::default(),
878            in_alternate_slot: false,
879            draining_to_replace: false,
880            configuration_updated_since_spawn: false,
881        }
882    }
883}
884
885type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
886
887type SpawnSubscriberKey = (ConnectionId, u64);
888
889#[derive(Debug)]
890struct SpawnSubscriber {
891    version: u8,
892    frames: mpsc::Sender<Frame>,
893    /// Tells this subscriber's forwarder that it was dropped for lagging, and
894    /// from which event. The full frame channel cannot carry that news, so it
895    /// travels beside it; see `SpawnEventFeed::subscribe`.
896    lagged: Option<oneshot::Sender<SpawnCursor>>,
897}
898
899#[derive(Debug)]
900struct SpawnEventState {
901    daemon_incarnation: String,
902    seq: u64,
903    capacity: usize,
904    live: HashMap<String, LiveSpawn>,
905    generations: HashMap<String, u64>,
906    events: VecDeque<SpawnEvent>,
907    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
908}
909
910impl Default for SpawnEventState {
911    fn default() -> Self {
912        Self {
913            daemon_incarnation: "unconfigured".to_string(),
914            seq: 0,
915            capacity: SPAWN_EVENT_RING_CAPACITY,
916            live: HashMap::new(),
917            generations: HashMap::new(),
918            events: VecDeque::new(),
919            subscribers: HashMap::new(),
920        }
921    }
922}
923
924#[derive(Debug, Clone, Default)]
925struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
926
927#[derive(Debug, Clone, PartialEq, Eq)]
928pub(crate) enum SpawnSubscribeRefusal {
929    ForeignIncarnation { current: String },
930    TooOld { oldest: SpawnCursor },
931    Frame(String),
932}
933
934impl SpawnEventFeed {
935    fn configure_incarnation(&self, daemon_incarnation: String) {
936        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
937        state.daemon_incarnation = daemon_incarnation;
938        state.seq = 0;
939        state.live.clear();
940        state.generations.clear();
941        state.events.clear();
942        state.subscribers.clear();
943    }
944
945    fn cursor(state: &SpawnEventState) -> SpawnCursor {
946        SpawnCursor {
947            daemon_incarnation: state.daemon_incarnation.clone(),
948            seq: state.seq,
949        }
950    }
951
952    fn snapshot(&self) -> SpawnSnapshot {
953        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
954        let mut live = state.live.values().cloned().collect::<Vec<_>>();
955        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
956        SpawnSnapshot {
957            cursor: Self::cursor(&state),
958            ring_bound: state.capacity as u64,
959            live,
960        }
961    }
962
963    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
964        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
965        let generation = state
966            .generations
967            .get(module_id)
968            .copied()
969            .unwrap_or(0)
970            .checked_add(1)
971            .expect("spawn generation exhausted");
972        state.generations.insert(module_id.to_string(), generation);
973        let live = LiveSpawn {
974            module_id: module_id.to_string(),
975            spawn_generation: generation,
976            pid,
977            spawned_at_ms,
978        };
979        state.live.insert(module_id.to_string(), live);
980        Self::emit_locked(
981            &mut state,
982            SpawnEventKind::Spawned,
983            module_id.to_string(),
984            generation,
985            pid,
986            None,
987            None,
988        );
989        generation
990    }
991
992    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
993        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
994        let Some(live) = state.live.remove(module_id) else {
995            warn!(
996                module_id,
997                "terminal record had no live spawn event identity"
998            );
999            return;
1000        };
1001        Self::emit_locked(
1002            &mut state,
1003            SpawnEventKind::Exited,
1004            module_id.to_string(),
1005            live.spawn_generation,
1006            live.pid,
1007            exit_code,
1008            exit_signal,
1009        );
1010    }
1011
1012    /// Report the exit of a process that a swap has already replaced.
1013    ///
1014    /// `emit_exited` removes the module's live entry, which after a swap's
1015    /// cutover describes the promoted candidate, not the old process now
1016    /// exiting. This emits the old generation's exit and leaves the live entry
1017    /// alone unless it still names that generation.
1018    fn emit_superseded_exited(
1019        &self,
1020        module_id: &str,
1021        spawn_generation: u64,
1022        pid: u32,
1023        exit_code: Option<i32>,
1024        exit_signal: Option<i32>,
1025    ) {
1026        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1027        if state
1028            .live
1029            .get(module_id)
1030            .is_some_and(|live| live.spawn_generation == spawn_generation)
1031        {
1032            state.live.remove(module_id);
1033        }
1034        Self::emit_locked(
1035            &mut state,
1036            SpawnEventKind::Exited,
1037            module_id.to_string(),
1038            spawn_generation,
1039            pid,
1040            exit_code,
1041            exit_signal,
1042        );
1043    }
1044
1045    #[allow(clippy::too_many_arguments)]
1046    fn emit_locked(
1047        state: &mut SpawnEventState,
1048        kind: SpawnEventKind,
1049        module_id: String,
1050        spawn_generation: u64,
1051        pid: u32,
1052        exit_code: Option<i32>,
1053        exit_signal: Option<i32>,
1054    ) {
1055        state.seq = state
1056            .seq
1057            .checked_add(1)
1058            .expect("spawn event sequence exhausted");
1059        let event = SpawnEvent {
1060            cursor: Self::cursor(state),
1061            kind,
1062            module_id,
1063            spawn_generation,
1064            pid,
1065            exit_code,
1066            exit_signal,
1067        };
1068        state.events.push_back(event.clone());
1069        while state.events.len() > state.capacity {
1070            state.events.pop_front();
1071        }
1072        let body = match serde_json::to_vec(&event) {
1073            Ok(body) => body,
1074            Err(error) => {
1075                error!(%error, "failed to serialize supervisor spawn event");
1076                return;
1077            }
1078        };
1079        state.subscribers.retain(|(connection_id, corr), subscriber| {
1080            let frame = Frame::build_with_version(
1081                subscriber.version,
1082                FrameType::StreamData,
1083                control_flags(),
1084                0,
1085                0,
1086                *corr,
1087                body.clone(),
1088            );
1089            match frame {
1090                Ok(frame) => {
1091                    if subscriber.frames.try_send(frame).is_ok() {
1092                        true
1093                    } else {
1094                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1095                        if let Some(lagged) = subscriber.lagged.take() {
1096                            let _ = lagged.send(event.cursor.clone());
1097                        }
1098                        false
1099                    }
1100                }
1101                Err(error) => {
1102                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1103                    false
1104                }
1105            }
1106        });
1107    }
1108
1109    fn subscribe(
1110        &self,
1111        connection_id: ConnectionId,
1112        corr: u64,
1113        version: u8,
1114        since: Option<SpawnCursor>,
1115        sink: FrameSink,
1116    ) -> Result<(), SpawnSubscribeRefusal> {
1117        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1118        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1119        {
1120            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1121            let replay = if let Some(since) = since {
1122                if since.daemon_incarnation != state.daemon_incarnation {
1123                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1124                        current: state.daemon_incarnation.clone(),
1125                    });
1126                }
1127                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1128                    if since.seq < oldest.seq.saturating_sub(1) {
1129                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1130                    }
1131                }
1132                state
1133                    .events
1134                    .iter()
1135                    .filter(|event| event.cursor.seq > since.seq)
1136                    .cloned()
1137                    .collect::<Vec<_>>()
1138            } else {
1139                Vec::new()
1140            };
1141            for event in replay {
1142                let body = serde_json::to_vec(&event)
1143                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1144                let frame = Frame::build_with_version(
1145                    version,
1146                    FrameType::StreamData,
1147                    control_flags(),
1148                    0,
1149                    0,
1150                    corr,
1151                    body,
1152                )
1153                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1154                frames
1155                    .try_send(frame)
1156                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1157            }
1158            state.subscribers.insert(
1159                (connection_id, corr),
1160                SpawnSubscriber {
1161                    version,
1162                    frames,
1163                    lagged: Some(lagged),
1164                },
1165            );
1166        }
1167        // The lagged terminal is sent here, by the forwarder, rather than by
1168        // the emitter: at the moment of the drop the subscriber's own channel
1169        // is full, and writing to the connection sink directly from the emitter
1170        // would put the Error AHEAD of the events still queued in that channel
1171        // (and the emitter holds the feed lock, so it cannot await the sink).
1172        // Dropping the subscriber drops the only sender, so `recv` drains every
1173        // queued event and then returns `None`; only then is the Error sent, so
1174        // the client sees each event it can keep, then the reason it was cut.
1175        // Cancel and connection removal drop the oneshot unsent, so they end
1176        // the stream with no Error.
1177        tokio::spawn(async move {
1178            while let Some(frame) = receiver.recv().await {
1179                if sink.send(frame).await.is_err() {
1180                    return;
1181                }
1182            }
1183            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1184                return;
1185            };
1186            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1187                Ok(frame) => {
1188                    let _ = sink.send(frame).await;
1189                }
1190                Err(error) => {
1191                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1192                }
1193            }
1194        });
1195        Ok(())
1196    }
1197
1198    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1199        let Some(subscriber) = self
1200            .0
1201            .lock()
1202            .unwrap_or_else(|p| p.into_inner())
1203            .subscribers
1204            .remove(&(connection_id, corr))
1205        else {
1206            return false;
1207        };
1208        if let Ok(frame) = Frame::build_with_version(
1209            subscriber.version,
1210            FrameType::StreamEnd,
1211            control_flags(),
1212            0,
1213            0,
1214            corr,
1215            Vec::new(),
1216        ) {
1217            tokio::spawn(async move {
1218                let _ = subscriber.frames.send(frame).await;
1219            });
1220        }
1221        true
1222    }
1223
1224    fn remove_connection(&self, connection_id: ConnectionId) {
1225        self.0
1226            .lock()
1227            .unwrap_or_else(|p| p.into_inner())
1228            .subscribers
1229            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1230    }
1231
1232    #[cfg(any(test, feature = "test-support"))]
1233    fn set_capacity(&self, capacity: usize) {
1234        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1235    }
1236
1237    #[cfg(any(test, feature = "test-support"))]
1238    fn subscriber_count(&self) -> usize {
1239        self.0
1240            .lock()
1241            .unwrap_or_else(|p| p.into_inner())
1242            .subscribers
1243            .len()
1244    }
1245}
1246
1247/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1248/// The terminal Error a lagged spawn subscriber receives after its queued events.
1249fn spawn_subscriber_lagged_frame(
1250    version: u8,
1251    corr: u64,
1252    first_undelivered: SpawnCursor,
1253) -> Result<Frame, String> {
1254    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1255        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1256        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1257            .to_string(),
1258        detail: Some(serde_json::json!({
1259            "first_undelivered_cursor": first_undelivered
1260        })),
1261    })
1262    .map_err(|error| error.to_string())?;
1263    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1264        .map_err(|error| error.to_string())
1265}
1266
1267pub trait ModuleProcessLiveness: Send + Sync {
1268    fn process_live(&self, module_id: &str) -> Option<bool>;
1269
1270    /// Whether the supervisor is replacing this module's process right now: an
1271    /// operator restart or reload, a health restart, or a crash respawn whose
1272    /// backoff is running. A module in that state is not live, but a consumer
1273    /// refused now should retry shortly rather than treat the target as gone.
1274    /// Stopped, failed, and disabled modules are not replacing.
1275    fn process_replacing(&self, _module_id: &str) -> bool {
1276        false
1277    }
1278}
1279
1280/// Shared process-liveness registry keyed by supervised `module_id`.
1281#[derive(Debug, Clone, Default)]
1282pub struct SupervisorProcessLiveness {
1283    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1284}
1285
1286impl SupervisorProcessLiveness {
1287    pub fn new() -> Self {
1288        Self::default()
1289    }
1290
1291    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1292        let mut snapshots = self
1293            .snapshots
1294            .lock()
1295            .unwrap_or_else(|poisoned| poisoned.into_inner());
1296        snapshots.insert(module_id, snapshot);
1297    }
1298
1299    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1300        let mut snapshots = self
1301            .snapshots
1302            .lock()
1303            .unwrap_or_else(|poisoned| poisoned.into_inner());
1304        let is_current = snapshots
1305            .get(module_id)
1306            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1307            .unwrap_or(false);
1308        if is_current {
1309            snapshots.remove(module_id);
1310        }
1311    }
1312}
1313
1314impl ModuleProcessLiveness for SupervisorProcessLiveness {
1315    fn process_live(&self, module_id: &str) -> Option<bool> {
1316        let snapshot = {
1317            let snapshots = self
1318                .snapshots
1319                .lock()
1320                .unwrap_or_else(|poisoned| poisoned.into_inner());
1321            snapshots.get(module_id).cloned()
1322        }?;
1323        let snapshot = snapshot
1324            .lock()
1325            .unwrap_or_else(|poisoned| poisoned.into_inner());
1326        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1327    }
1328
1329    fn process_replacing(&self, module_id: &str) -> bool {
1330        let Some(snapshot) = self
1331            .snapshots
1332            .lock()
1333            .unwrap_or_else(|poisoned| poisoned.into_inner())
1334            .get(module_id)
1335            .cloned()
1336        else {
1337            return false;
1338        };
1339        let snapshot = snapshot
1340            .lock()
1341            .unwrap_or_else(|poisoned| poisoned.into_inner());
1342        snapshot.enabled
1343            && match snapshot.state {
1344                ModuleState::Restarting => true,
1345                ModuleState::Draining => snapshot.draining_to_replace,
1346                ModuleState::Starting
1347                | ModuleState::Running
1348                | ModuleState::Unresponsive
1349                | ModuleState::Stopped
1350                | ModuleState::Failed
1351                | ModuleState::Disabled => false,
1352            }
1353    }
1354}
1355
1356#[derive(Debug, Clone)]
1357struct SupervisorRuntimeConfig {
1358    restart_policy: RestartPolicy,
1359    /// This module's RESOLVED drain budget: per-module config when present,
1360    /// else `default_drain_timeout`.
1361    drain_timeout: Duration,
1362    /// Shared with the status handle so the attested value changes atomically
1363    /// when a rescan updates the running drain policy.
1364    effective_drain_timeout: Arc<Mutex<Duration>>,
1365    /// The supervisor-wide fallback, kept so a configuration update that
1366    /// REMOVES the per-module override can re-resolve to it.
1367    default_drain_timeout: Duration,
1368    health: HealthConfig,
1369    connection_file_path: Option<PathBuf>,
1370    capture_logs_dir: Option<PathBuf>,
1371    forwarding: Option<Arc<ForwardingTable>>,
1372    /// The shared handle, so every spawn path (initial, restart, reload) records the
1373    /// reserved-module launch nonce the HELLO verifier checks against.
1374    supervisor_handle: Option<SupervisorHandle>,
1375    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1376    /// status queries.
1377    ///
1378    /// One ring per module, held across every respawn. The lines explaining an exit
1379    /// are written BEFORE that exit, so a ring recreated per process would be empty
1380    /// exactly when it is asked for.
1381    stderr_ring: Arc<Mutex<StderrRing>>,
1382    terminal_ring: Arc<Mutex<TerminalRing>>,
1383    spawn_events: SpawnEventFeed,
1384    child_roster: ChildRoster,
1385    #[cfg(target_os = "linux")]
1386    cgroup_placement: Option<subc_cgroup::Placement>,
1387    #[cfg(test)]
1388    test_seed_stale_facts_before_enable_spawn: bool,
1389}
1390
1391#[derive(Debug, Clone, PartialEq, Eq)]
1392struct SupervisedConfiguration {
1393    spec: ModuleSpec,
1394    health: HealthConfig,
1395}
1396
1397/// Shared daemon lookup table for supervised module handles.
1398///
1399/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1400/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1401/// launch nonces recorded at spawn are checked by the same daemon instance.
1402#[derive(Debug, Clone, Default)]
1403pub struct SupervisorHandle {
1404    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1405    spawn_events: SpawnEventFeed,
1406    /// The current expected launch nonce for each reserved module_id. Set when the
1407    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1408    /// non-reserved module never has an entry here and is never nonce-checked.
1409    /// Reserved module ids and the nonce that authorizes their next HELLO.
1410    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1411    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1412    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1413    /// had NO entry and admitted anyone: the reservation protected the nonce
1414    /// holder, not the NAME (found live by CKCRED's canary probe registering
1415    /// against a reserved scratch id).
1416    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1417    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1418    ///
1419    /// This is deliberately in-memory only: subc is state-free across daemon
1420    /// restarts, and the tombstone only explains the hours-after-removal window
1421    /// while this executing daemon is still alive. Do not persist it in a store.
1422    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1423    /// The current launch nonce for every supervised spawn. This is separate from
1424    /// reserved_nonces because consumer route.open attestation applies to all spawned
1425    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1426    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1427    /// Reserved namespace prefixes mapped to the supervised owner module whose
1428    /// current spawn nonce authorizes HELLO claims below the prefix.
1429    ///
1430    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1431    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1432    /// accidental collisions and lower-trust processes from squatting protected
1433    /// namespaces.
1434    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1435    /// Blue/green swaps in progress, by module id. An entry exists from just
1436    /// before the candidate process is spawned until the swap has failed, or
1437    /// has cut over and the old process is gone. While it exists, HELLO for the
1438    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1439    /// consumer attestation accepts both processes' nonces.
1440    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1441    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1442    promotion_observer: PromotionObserverSlot,
1443    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1444    /// this daemon-wide ordering, a rescan could retire or update a module while a
1445    /// concurrent reload still held its old handle and launch specification.
1446    operation_lock: Arc<AsyncMutex<()>>,
1447}
1448
1449/// Told when a swap has promoted its candidate to be the module's active
1450/// registration.
1451///
1452/// An ordinary HELLO runs the control plane's registration side effects (the
1453/// capability cache, the deny census, the requirement recompute) as it
1454/// registers. A swap candidate's HELLO does not, because it is not routable;
1455/// promotion is when those must run instead, and promotion happens in the
1456/// supervisor, which has no other way into the control handler.
1457pub(crate) trait SwapPromotionObserver: Send + Sync {
1458    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1459}
1460
1461/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1462/// control handler) owns this handle, so a strong reference back would be a
1463/// cycle that keeps both alive.
1464#[derive(Clone, Default)]
1465struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1466
1467impl fmt::Debug for PromotionObserverSlot {
1468    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1469        f.write_str("PromotionObserverSlot")
1470    }
1471}
1472
1473/// The nonces of one open swap.
1474#[derive(Debug, Clone)]
1475struct OpenSwap {
1476    /// The launch nonce minted for the candidate process. It is the swap
1477    /// token: the only thing that admits a HELLO into the candidate slot.
1478    candidate_nonce: String,
1479    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1480    /// here because cutover moves the module's recorded spawn nonce to the
1481    /// candidate while the incumbent is still draining and its consumers are
1482    /// still attesting with this one.
1483    incumbent_nonce: Option<String>,
1484    /// Set once a HELLO has been admitted with the swap token, so the token
1485    /// admits one registration and cannot be replayed after cutover empties
1486    /// the candidate slot.
1487    candidate_admitted: bool,
1488}
1489
1490/// What the swap gate says about a HELLO. See
1491/// [`SupervisorHandle::swap_hello_admission`].
1492#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1493pub(crate) enum SwapHelloAdmission {
1494    /// No swap is open for the id (or the HELLO carries the incumbent's own
1495    /// nonce); the ordinary gates decide.
1496    NotSwapping,
1497    /// The HELLO carries the swap token: register it into the candidate slot.
1498    Candidate,
1499    /// A swap is open and the HELLO carries a nonce the supervisor did not
1500    /// mint for this id, no nonce, or a token already used.
1501    Refused,
1502}
1503
1504#[derive(Debug, Clone, PartialEq, Eq)]
1505pub(crate) enum ReservedHelloRejection {
1506    Exact {
1507        module_id: String,
1508    },
1509    Prefix {
1510        prefix: String,
1511        owner_module_id: String,
1512    },
1513}
1514
1515impl SupervisorHandle {
1516    pub fn new() -> Self {
1517        Self::default()
1518    }
1519
1520    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1521        self.spawn_events.snapshot()
1522    }
1523
1524    pub(crate) fn subscribe_spawns(
1525        &self,
1526        connection_id: ConnectionId,
1527        corr: u64,
1528        version: u8,
1529        since: Option<SpawnCursor>,
1530        sink: FrameSink,
1531    ) -> Result<(), SpawnSubscribeRefusal> {
1532        self.spawn_events
1533            .subscribe(connection_id, corr, version, since, sink)
1534    }
1535
1536    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1537        self.spawn_events.cancel(connection_id, corr)
1538    }
1539
1540    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1541        self.spawn_events.remove_connection(connection_id);
1542    }
1543
1544    #[cfg(any(test, feature = "test-support"))]
1545    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1546        assert!(capacity > 0, "spawn event capacity must be non-zero");
1547        self.spawn_events.set_capacity(capacity);
1548    }
1549
1550    #[cfg(any(test, feature = "test-support"))]
1551    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1552        self.spawn_events.subscriber_count()
1553    }
1554
1555    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1556    /// a respawn invalidates stale consumer identities.
1557    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1558        self.spawn_nonces
1559            .lock()
1560            .unwrap_or_else(|poisoned| poisoned.into_inner())
1561            .insert(module_id.to_string(), nonce);
1562    }
1563
1564    /// Record the launch nonce expected from the next HELLO for a reserved module,
1565    /// replacing any prior nonce (a respawn invalidates the previous one).
1566    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1567        self.reserved_nonces
1568            .lock()
1569            .unwrap_or_else(|poisoned| poisoned.into_inner())
1570            .insert(module_id.to_string(), Some(nonce));
1571    }
1572
1573    /// Record namespace prefixes owned by a supervised module.
1574    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1575        let mut owners = self
1576            .reserved_prefix_owners
1577            .lock()
1578            .unwrap_or_else(|poisoned| poisoned.into_inner());
1579        owners.retain(|_, owner| owner != owner_module_id);
1580        for prefix in prefixes {
1581            owners.insert(prefix.clone(), owner_module_id.to_string());
1582        }
1583    }
1584
1585    /// The launch nonce most recently minted for a module's spawn, if any.
1586    #[cfg(test)]
1587    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1588        self.spawn_nonces
1589            .lock()
1590            .unwrap_or_else(|poisoned| poisoned.into_inner())
1591            .get(module_id)
1592            .cloned()
1593    }
1594
1595    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1596        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1597        let spawn_nonce = self
1598            .spawn_nonces
1599            .lock()
1600            .unwrap_or_else(|poisoned| poisoned.into_inner())
1601            .get(&spec.module_id)
1602            .cloned();
1603        let mut reserved_nonces = self
1604            .reserved_nonces
1605            .lock()
1606            .unwrap_or_else(|poisoned| poisoned.into_inner());
1607        if spec.reserved {
1608            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1609            // reserved name whose module has never spawned has no legitimate
1610            // holder, and the entry's absence is what used to leave the name
1611            // open to the first claimant.
1612            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1613        }
1614        drop(reserved_nonces);
1615        // A later unreserved declaration must not silently unreserve an id that
1616        // was retained after its reserved configuration was removed. The explicit
1617        // release ceremony is the only operation that retires that gate.
1618        self.removal_tombstones
1619            .lock()
1620            .unwrap_or_else(|poisoned| poisoned.into_inner())
1621            .remove(&spec.module_id);
1622    }
1623
1624    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
1625    /// authorized only by its expected nonce; otherwise a matching reserved prefix
1626    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
1627    /// with no matching prefix are always authorized.
1628    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1629        self.reserved_hello_rejection(module_id, presented)
1630            .is_none()
1631    }
1632
1633    pub(crate) fn reserved_hello_rejection(
1634        &self,
1635        module_id: &str,
1636        presented: Option<&str>,
1637    ) -> Option<ReservedHelloRejection> {
1638        let nonces = self
1639            .reserved_nonces
1640            .lock()
1641            .unwrap_or_else(|poisoned| poisoned.into_inner());
1642        if let Some(expected) = nonces.get(module_id) {
1643            // `None` = reserved with no legitimate holder: refuse every
1644            // presentation, because no process can hold a nonce that was never
1645            // minted. Only a real minted nonce admits, in constant time.
1646            let authorized = match expected {
1647                Some(expected) => {
1648                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1649                }
1650                None => false,
1651            };
1652            if authorized {
1653                return None;
1654            }
1655            return Some(ReservedHelloRejection::Exact {
1656                module_id: module_id.to_string(),
1657            });
1658        }
1659        drop(nonces);
1660
1661        let matched_prefix = self
1662            .reserved_prefix_owners
1663            .lock()
1664            .unwrap_or_else(|poisoned| poisoned.into_inner())
1665            .iter()
1666            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
1667            .max_by_key(|(prefix, _)| prefix.len())
1668            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
1669        let (prefix, owner_module_id) = matched_prefix?;
1670
1671        let authorized = presented.is_some_and(|presented| {
1672            self.spawn_nonces
1673                .lock()
1674                .unwrap_or_else(|poisoned| poisoned.into_inner())
1675                .get(&owner_module_id)
1676                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
1677                // While the owner is being swapped, children started by
1678                // either of its two processes hold that process's nonce.
1679                || self.swap_nonce_matches(&owner_module_id, presented)
1680        });
1681        if authorized {
1682            None
1683        } else {
1684            Some(ReservedHelloRejection::Prefix {
1685                prefix,
1686                owner_module_id,
1687            })
1688        }
1689    }
1690
1691    /// Whether a consumer connection proved it came from a daemon-spawned module.
1692    ///
1693    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
1694    /// accepted only for module ids the supervisor has spawned.
1695    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
1696        if presented.is_empty() {
1697            return false;
1698        }
1699        let nonces = self
1700            .spawn_nonces
1701            .lock()
1702            .unwrap_or_else(|poisoned| poisoned.into_inner());
1703        let current = nonces
1704            .get(module_id)
1705            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
1706        drop(nonces);
1707        // During a swap two processes of the module are alive, and a consumer
1708        // started by either one presents that process's nonce. Accepting only
1709        // the recorded one would fail the incumbent's consumers for the whole
1710        // overlap once cutover moves the record to the candidate.
1711        current || self.swap_nonce_matches(module_id, presented)
1712    }
1713
1714    /// Whether `presented` is either nonce of an open swap for `module_id`.
1715    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
1716        let swaps = self
1717            .swaps
1718            .lock()
1719            .unwrap_or_else(|poisoned| poisoned.into_inner());
1720        swaps.get(module_id).is_some_and(|swap| {
1721            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
1722                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
1723                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
1724                })
1725        })
1726    }
1727
1728    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
1729    /// Called before the candidate process exists.
1730    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
1731        let incumbent_nonce = self
1732            .spawn_nonces
1733            .lock()
1734            .unwrap_or_else(|poisoned| poisoned.into_inner())
1735            .get(module_id)
1736            .cloned();
1737        self.swaps
1738            .lock()
1739            .unwrap_or_else(|poisoned| poisoned.into_inner())
1740            .insert(
1741                module_id.to_string(),
1742                OpenSwap {
1743                    candidate_nonce,
1744                    incumbent_nonce,
1745                    candidate_admitted: false,
1746                },
1747            );
1748    }
1749
1750    /// Close the swap for `module_id`, releasing whichever nonce is no longer
1751    /// the module's recorded one.
1752    pub(crate) fn close_swap(&self, module_id: &str) {
1753        self.swaps
1754            .lock()
1755            .unwrap_or_else(|poisoned| poisoned.into_inner())
1756            .remove(module_id);
1757    }
1758
1759    /// Install the observer told about swap promotions, replacing any earlier
1760    /// one.
1761    pub(crate) fn set_swap_promotion_observer(
1762        &self,
1763        observer: std::sync::Weak<dyn SwapPromotionObserver>,
1764    ) {
1765        *self
1766            .promotion_observer
1767            .0
1768            .lock()
1769            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
1770    }
1771
1772    /// Tell the installed observer, if it is still alive, that a swap promoted
1773    /// `registration`.
1774    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
1775        let observer = self
1776            .promotion_observer
1777            .0
1778            .lock()
1779            .unwrap_or_else(|poisoned| poisoned.into_inner())
1780            .as_ref()
1781            .and_then(std::sync::Weak::upgrade);
1782        if let Some(observer) = observer {
1783            observer.swap_promoted(registration);
1784        }
1785    }
1786
1787    /// Whether a swap is open for `module_id`.
1788    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
1789        self.swaps
1790            .lock()
1791            .unwrap_or_else(|poisoned| poisoned.into_inner())
1792            .contains_key(module_id)
1793    }
1794
1795    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
1796    /// respawn would, once cutover has made the candidate the module's process.
1797    /// The swap stays open so the incumbent's nonce keeps attesting until the
1798    /// incumbent has drained and exited.
1799    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
1800        let candidate_nonce = self
1801            .swaps
1802            .lock()
1803            .unwrap_or_else(|poisoned| poisoned.into_inner())
1804            .get(module_id)
1805            .map(|swap| swap.candidate_nonce.clone());
1806        let Some(nonce) = candidate_nonce else {
1807            return;
1808        };
1809        self.set_spawn_nonce(module_id, nonce.clone());
1810        if reserved {
1811            self.set_reserved_nonce(module_id, nonce);
1812        }
1813    }
1814
1815    /// The swap gate for a HELLO claiming `module_id`.
1816    ///
1817    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
1818    /// presents the candidate nonce, which the reserved gate (holding the
1819    /// incumbent's nonce) would refuse as `reserved_module` before swap
1820    /// admission was ever reached. And it applies to unreserved ids too: for an
1821    /// unreserved id the only thing that ever stopped a second process claiming
1822    /// a live id was the `duplicate_module_id` refusal, which is exactly the
1823    /// refusal a swap lifts for its candidate.
1824    ///
1825    /// The incumbent's own nonce falls through to the ordinary gates, which
1826    /// treat it as they always have (a live incumbent is refused as a
1827    /// duplicate). Anything else while a swap is open is refused, including an
1828    /// absent nonce.
1829    pub(crate) fn swap_hello_admission(
1830        &self,
1831        module_id: &str,
1832        presented: Option<&str>,
1833    ) -> SwapHelloAdmission {
1834        let swaps = self
1835            .swaps
1836            .lock()
1837            .unwrap_or_else(|poisoned| poisoned.into_inner());
1838        let Some(swap) = swaps.get(module_id) else {
1839            return SwapHelloAdmission::NotSwapping;
1840        };
1841        let Some(presented) = presented else {
1842            return SwapHelloAdmission::Refused;
1843        };
1844        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
1845            return if swap.candidate_admitted {
1846                SwapHelloAdmission::Refused
1847            } else {
1848                SwapHelloAdmission::Candidate
1849            };
1850        }
1851        if swap
1852            .incumbent_nonce
1853            .as_deref()
1854            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
1855        {
1856            return SwapHelloAdmission::NotSwapping;
1857        }
1858        SwapHelloAdmission::Refused
1859    }
1860
1861    /// Record that the swap token has registered a candidate, so it admits no
1862    /// second HELLO.
1863    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
1864        if let Some(swap) = self
1865            .swaps
1866            .lock()
1867            .unwrap_or_else(|poisoned| poisoned.into_inner())
1868            .get_mut(module_id)
1869        {
1870            swap.candidate_admitted = true;
1871        }
1872    }
1873
1874    /// Test/support lookup for the current launch nonce of a supervised spawn.
1875    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1876        self.spawn_nonces
1877            .lock()
1878            .unwrap_or_else(|poisoned| poisoned.into_inner())
1879            .get(module_id)
1880            .cloned()
1881    }
1882
1883    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
1884    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1885        self.reserved_nonces
1886            .lock()
1887            .unwrap_or_else(|poisoned| poisoned.into_inner())
1888            .get(module_id)
1889            .cloned()
1890            .flatten()
1891    }
1892
1893    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
1894        let mut modules = self
1895            .modules
1896            .lock()
1897            .unwrap_or_else(|poisoned| poisoned.into_inner());
1898        modules.insert(module.module_id().to_string(), module)
1899    }
1900
1901    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
1902        let modules = self
1903            .modules
1904            .lock()
1905            .unwrap_or_else(|poisoned| poisoned.into_inner());
1906        modules.get(module_id).cloned()
1907    }
1908
1909    pub(crate) fn record_late_health_answer(
1910        &self,
1911        module_id: &str,
1912        latency_ms: u64,
1913    ) -> Result<bool, SuperviseError> {
1914        let Some(module) = self.get(module_id) else {
1915            return Ok(false);
1916        };
1917        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
1918            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
1919            state.health.last_late_answer_latency_ms = Some(latency_ms);
1920            // A late answer is an answer: the module served the probe, just past
1921            // the deadline. Leaving the miss streak in place while logging
1922            // "proves the module is alive" is how a CPU-starved module that
1923            // answers every probe a few seconds late still marches to the
1924            // threshold and gets killed — the exact kill class `NoAnswer` is
1925            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
1926            // is degradation, and degradation reports; it does not restart.
1927            state.health.consecutive_failures = 0;
1928        })?;
1929        Ok(true)
1930    }
1931
1932    /// Arm the one-shot marker for the module process that this caller
1933    /// deliberately initiated severance against. Generic connection teardown
1934    /// must not call this:
1935    /// a surviving process would otherwise retain an exemption for a later
1936    /// genuine crash.
1937    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
1938        let Some(module) = self.get(module_id) else {
1939            return Ok(false);
1940        };
1941        let status = module.status()?;
1942        let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
1943            return Ok(false);
1944        };
1945        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
1946    }
1947
1948    pub fn list(&self) -> Vec<SupervisedModule> {
1949        let modules = self
1950            .modules
1951            .lock()
1952            .unwrap_or_else(|poisoned| poisoned.into_inner());
1953        let mut modules = modules.values().cloned().collect::<Vec<_>>();
1954        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
1955        modules
1956    }
1957
1958    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
1959        self.spawn_nonces
1960            .lock()
1961            .unwrap_or_else(|poisoned| poisoned.into_inner())
1962            .remove(module_id);
1963        self.close_swap(module_id);
1964        let mut reserved_nonces = self
1965            .reserved_nonces
1966            .lock()
1967            .unwrap_or_else(|poisoned| poisoned.into_inner());
1968        if reserved_nonces.contains_key(module_id) {
1969            // The old nonce must die with the removed process, but the exact-id
1970            // gate remains until an operator explicitly releases it.
1971            reserved_nonces.insert(module_id.to_string(), None);
1972        }
1973        drop(reserved_nonces);
1974        self.reserved_prefix_owners
1975            .lock()
1976            .unwrap_or_else(|poisoned| poisoned.into_inner())
1977            .retain(|_, owner| owner != module_id);
1978        self.modules
1979            .lock()
1980            .unwrap_or_else(|poisoned| poisoned.into_inner())
1981            .remove(module_id)
1982    }
1983
1984    /// Remember a module removed by a non-preview rescan so route.open can
1985    /// distinguish that intentional removal from an unknown id.
1986    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
1987        self.removal_tombstones
1988            .lock()
1989            .unwrap_or_else(|poisoned| poisoned.into_inner())
1990            .insert(module_id.to_string(), unix_ms_now());
1991    }
1992
1993    /// Return how long ago a rescan removed this module in milliseconds.
1994    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
1995        self.removal_tombstones
1996            .lock()
1997            .unwrap_or_else(|poisoned| poisoned.into_inner())
1998            .get(module_id)
1999            .copied()
2000            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2001    }
2002
2003    /// Retire a reserved-id gate only after its module has left supervision.
2004    ///
2005    /// A retained gate has no live nonce (`None`), so releasing any other entry
2006    /// would weaken a currently configured or otherwise active reservation.
2007    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2008        if self.get(module_id).is_some() {
2009            return false;
2010        }
2011        let mut reserved_nonces = self
2012            .reserved_nonces
2013            .lock()
2014            .unwrap_or_else(|poisoned| poisoned.into_inner());
2015        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2016            return false;
2017        }
2018        reserved_nonces.remove(module_id);
2019        true
2020    }
2021
2022    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2023        Arc::clone(&self.operation_lock)
2024    }
2025}
2026
2027/// Process supervisor for subc-owned singleton modules.
2028#[derive(Debug, Clone)]
2029pub struct Supervisor {
2030    registry: Arc<Registry>,
2031    restart_policy: RestartPolicy,
2032    drain_timeout: Duration,
2033    connection_file_path: Option<PathBuf>,
2034    capture_logs_dir: Option<PathBuf>,
2035    forwarding: Option<Arc<ForwardingTable>>,
2036    process_liveness: Arc<SupervisorProcessLiveness>,
2037    supervisor_handle: Option<SupervisorHandle>,
2038    health: HealthConfig,
2039    daemon_start_clock: crate::clock::StartClock,
2040    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2041    spawn_events: SpawnEventFeed,
2042    provenance_probe: ExecutableIdentityProbe,
2043    /// Every process spawned through this supervisor (and its clones) and not
2044    /// yet reaped, so daemon shutdown can end them.
2045    child_roster: ChildRoster,
2046    #[cfg(target_os = "linux")]
2047    cgroup_placement: Option<subc_cgroup::Placement>,
2048}
2049
2050impl Supervisor {
2051    /// The first step of an announced daemon shutdown, before the notice and
2052    /// before any connection is closed.
2053    ///
2054    /// Sets the daemon-shutdown flag first: from here on no module is
2055    /// respawned (crash restart, operator restart, or swap), and every child
2056    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2057    /// the module exits on the EOF this shutdown gives it or is signalled by a
2058    /// service manager that kills the whole cgroup. Then writes the journal's
2059    /// shutdown marker, which records the instant and closes this daemon
2060    /// incarnation's stretch of the journal.
2061    #[cfg(unix)]
2062    pub(crate) fn begin_daemon_shutdown(&self) {
2063        self.child_roster.close();
2064        if let Some(journal) = &self.terminal_journal {
2065            journal.stamp_shutdown();
2066        }
2067    }
2068
2069    /// Announce a cut while established connections can still carry replies.
2070    /// These budgets promise notice and a bounded wait, not child completion;
2071    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2072    #[cfg(unix)]
2073    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2074        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2075        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2076        let Some(forwarding) = &self.forwarding else {
2077            return Ok(());
2078        };
2079        let module_ids = forwarding
2080            .begin_daemon_drain()
2081            .map_err(SuperviseError::Forwarding)?;
2082        let deadline_ms =
2083            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2084        let mut notices = tokio::task::JoinSet::new();
2085        let mut drains = Vec::new();
2086        for module_id in module_ids {
2087            let Some(target) = forwarding
2088                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2089                .map_err(SuperviseError::Forwarding)?
2090            else {
2091                continue;
2092            };
2093            let routes = forwarding
2094                .endpoint_routes(target.endpoint)
2095                .map_err(SuperviseError::Forwarding)?;
2096            // Restart allows deployed consumers to reopen after the new daemon
2097            // appears. The wire reason stays `restart`; what tells a daemon cut
2098            // apart from a module restart afterwards is the terminal record
2099            // itself, whose disposition is `daemon_shutdown` for every exit
2100            // observed once `begin_daemon_shutdown` has run.
2101            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2102                reason: RouteCloseReason::Restart,
2103                deadline_ms,
2104            })
2105            .expect("module draining serializes");
2106            let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2107                module_id: module_id.clone(),
2108                reason: RouteCloseReason::Restart,
2109            })
2110            .expect("route closing serializes");
2111            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2112            let mut seen = std::collections::HashSet::new();
2113            for route in routes {
2114                let client = route.goodbye_target;
2115                if seen.insert(client.connection_id) {
2116                    recipients.push((client.sink, client.negotiated_ver, closing.clone()));
2117                }
2118            }
2119            for (sink, version, body) in recipients {
2120                notices.spawn(async move {
2121                    let frame = Frame::build_with_version(
2122                        version,
2123                        FrameType::Push,
2124                        control_flags(),
2125                        0,
2126                        0,
2127                        0,
2128                        body,
2129                    )
2130                    .expect("bounded lifecycle notice frame builds");
2131                    sink.send_flushed(frame).await
2132                });
2133            }
2134            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2135            drains.push((module_id, target.endpoint, gauges));
2136        }
2137        // A quiet forwarding table is not proof that queued notices reached the
2138        // socket. Wait for writer flush acknowledgements before testing quiescence.
2139        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2140        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2141            if !matches!(result, Ok(Ok(()))) {
2142                warn!(?result, "daemon shutdown notice delivery failed");
2143            }
2144        }
2145        notices.abort_all();
2146        let deadline = Instant::now() + DRAIN_BUDGET;
2147        let mut waits = tokio::task::JoinSet::new();
2148        for (module_id, endpoint, gauges) in drains {
2149            let forwarding = Arc::clone(forwarding);
2150            let mut runtime = self.runtime_config();
2151            runtime.health.cadence = Duration::from_millis(100);
2152            waits.spawn(async move {
2153                wait_for_forwarding_quiescence(
2154                    &forwarding,
2155                    &module_id,
2156                    &runtime,
2157                    endpoint,
2158                    deadline,
2159                    &gauges,
2160                    DrainScope::Active,
2161                )
2162                .await
2163            });
2164        }
2165        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2166            if !matches!(result, Ok(Ok(true))) {
2167                warn!(?result, "daemon shutdown drain did not reach quiescence");
2168            }
2169        }
2170        Ok(())
2171    }
2172
2173    /// The last step of an announced daemon shutdown, after the notice and the
2174    /// drain: send every registered module a module GOODBYE, the same planned
2175    /// stop signal `ck module stop` gives, then close every connection so each
2176    /// subc module sees EOF and starts its own teardown, then end every
2177    /// supervised child that has not exited
2178    /// by its own deadline (its drain budget, capped). Modules lead their own
2179    /// process groups, so a
2180    /// service manager's group kill no longer reaches them; without this a
2181    /// child that does not stop on EOF (every `protocol: "none"` child, which
2182    /// has no connection) would outlive the daemon. Every wait is bounded (see
2183    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2184    #[cfg(unix)]
2185    pub(crate) async fn end_children_for_daemon_shutdown(
2186        &self,
2187        already_escalated: bool,
2188        escalate: impl std::future::Future<Output = ()>,
2189    ) {
2190        tokio::pin!(escalate);
2191        let mut escalated = already_escalated;
2192        if let Some(forwarding) = &self.forwarding {
2193            let reason = CloseReason::new(
2194                "daemon_shutdown",
2195                "the daemon is exiting after its shutdown notice and drain",
2196            );
2197            if escalated {
2198                // The operator asked to stop waiting: queue the GOODBYEs but
2199                // do not wait for them to be written.
2200                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2201            } else {
2202                tokio::select! {
2203                    biased;
2204                    _ = escalate.as_mut() => {
2205                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2206                        escalated = true;
2207                    }
2208                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2209                }
2210            }
2211            let closed = forwarding.close_all_connections(&reason);
2212            debug!(closed, "closed established connections for daemon shutdown");
2213        }
2214        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2215        // already completed and must not be polled again; the child shutdown
2216        // wait is told it is escalated and gets a future that never fires.
2217        let escalated_here = escalated && !already_escalated;
2218        let remaining_escalate = async move {
2219            if escalated_here {
2220                std::future::pending::<()>().await;
2221            } else {
2222                escalate.await;
2223            }
2224        };
2225        crate::child_roster::end_children_for_daemon_shutdown(
2226            &self.child_roster,
2227            escalated,
2228            remaining_escalate,
2229        )
2230        .await;
2231    }
2232
2233    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2234        Self {
2235            registry,
2236            restart_policy,
2237            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2238            connection_file_path: None,
2239            capture_logs_dir: None,
2240            forwarding: None,
2241            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2242            supervisor_handle: None,
2243            health: HealthConfig::default(),
2244            daemon_start_clock: crate::clock::StartClock::capture(),
2245            terminal_journal: None,
2246            spawn_events: SpawnEventFeed::default(),
2247            provenance_probe: ExecutableIdentityProbe::default(),
2248            child_roster: ChildRoster::default(),
2249            #[cfg(target_os = "linux")]
2250            cgroup_placement: None,
2251        }
2252    }
2253
2254    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2255        self.drain_timeout = drain_timeout;
2256        self
2257    }
2258
2259    pub fn with_process_liveness(
2260        mut self,
2261        process_liveness: Arc<SupervisorProcessLiveness>,
2262    ) -> Self {
2263        self.process_liveness = process_liveness;
2264        self
2265    }
2266
2267    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2268        self.connection_file_path = Some(connection_file_path.into());
2269        self
2270    }
2271
2272    /// Enables daemon-owned capture files for supervised stdout and stderr.
2273    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2274        self.capture_logs_dir = Some(logs_dir.into());
2275        self
2276    }
2277
2278    /// Names this daemon lifetime in spawn events, independently of whether a
2279    /// terminal journal is configured.
2280    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2281        // A millisecond start stamp can repeat after clock rollback or a rapid
2282        // restart. Use the connection file's random daemon_id instead: it already
2283        // identifies this daemon lifetime independently of the wall clock.
2284        self.spawn_events.configure_incarnation(daemon_incarnation);
2285        self
2286    }
2287
2288    /// Enables best-effort history shared by every supervised module. Without
2289    /// it, terminal history is kept only in each module's in-memory ring.
2290    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2291        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2292        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2293            path,
2294            daemon_incarnation,
2295        )));
2296        this
2297    }
2298
2299    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2300        self.forwarding = Some(forwarding);
2301        self
2302    }
2303
2304    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2305        self.spawn_events = supervisor_handle.spawn_events.clone();
2306        self.supervisor_handle = Some(supervisor_handle);
2307        self
2308    }
2309
2310    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2311        self.health = health;
2312        self
2313    }
2314
2315    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2316    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2317    /// record is kept.
2318    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2319        self.child_roster.record_to(path.into());
2320        self
2321    }
2322
2323    #[cfg(target_os = "linux")]
2324    pub fn with_cgroup_placement(
2325        mut self,
2326        cgroup_placement: Option<subc_cgroup::Placement>,
2327    ) -> Self {
2328        self.cgroup_placement = cgroup_placement;
2329        self
2330    }
2331
2332    /// Spawn `spec.program` and start monitoring it.
2333    ///
2334    /// The child is expected to parse `--subc <connection-file-path>`, read the
2335    /// TCP+key connection file, authenticate to the already-running listener, and
2336    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2337    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2338        validate_spec(&spec)?;
2339
2340        let runtime = self.runtime_config();
2341        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2342        let child = spawn_child(
2343            &spec,
2344            runtime.connection_file_path.as_deref(),
2345            self.supervisor_handle.as_ref(),
2346            &runtime.stderr_ring,
2347            runtime.capture_logs_dir.as_deref(),
2348            &runtime.child_roster,
2349            #[cfg(target_os = "linux")]
2350            runtime.cgroup_placement.as_ref(),
2351        )?;
2352        set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2353        self.process_liveness
2354            .track(spec.module_id.clone(), Arc::clone(&snapshot));
2355
2356        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2357    }
2358
2359    /// Start supervising a module declared in daemon configuration.
2360    ///
2361    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2362    /// failures in the supervisor handle so operator-facing `supervisor.list`
2363    /// reflects every configured module while daemon startup continues.
2364    pub fn supervise_configured(
2365        &self,
2366        spec: ModuleSpec,
2367        enabled: bool,
2368    ) -> Result<SupervisedModule, SuperviseError> {
2369        validate_spec(&spec)?;
2370
2371        let runtime = self.runtime_config();
2372        if !enabled {
2373            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2374            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2375        }
2376
2377        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2378        match spawn_child(
2379            &spec,
2380            runtime.connection_file_path.as_deref(),
2381            self.supervisor_handle.as_ref(),
2382            &runtime.stderr_ring,
2383            runtime.capture_logs_dir.as_deref(),
2384            &runtime.child_roster,
2385            #[cfg(target_os = "linux")]
2386            runtime.cgroup_placement.as_ref(),
2387        ) {
2388            Ok(child) => {
2389                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2390                self.process_liveness
2391                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2392                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2393            }
2394            Err(err) => {
2395                error!(
2396                    module_id = %spec.module_id,
2397                    program = %spec.program.display(),
2398                    error = %err,
2399                    "configured module failed to spawn; marking failed and continuing"
2400                );
2401                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2402                Ok(self.supervised_module(spec, runtime, snapshot, None))
2403            }
2404        }
2405    }
2406
2407    /// Supervise a configured module with its own health, drain, and crash
2408    /// budget. The restart policy is per-module because the config file is:
2409    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2410    /// module that is expensive to restart should not be forced onto the same
2411    /// budget as one that is cheap.
2412    pub fn supervise_configured_with_health(
2413        &self,
2414        spec: ModuleSpec,
2415        enabled: bool,
2416        health: HealthConfig,
2417        drain_timeout_ms: Option<u64>,
2418        restart_policy: RestartPolicy,
2419    ) -> Result<SupervisedModule, SuperviseError> {
2420        validate_spec(&spec)?;
2421
2422        let mut runtime = self.runtime_config();
2423        runtime.health = health;
2424        runtime.restart_policy = restart_policy;
2425        if let Some(ms) = drain_timeout_ms {
2426            runtime.drain_timeout = Duration::from_millis(ms);
2427            *runtime
2428                .effective_drain_timeout
2429                .lock()
2430                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2431        }
2432        if !enabled {
2433            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2434            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2435        }
2436
2437        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2438        match spawn_child(
2439            &spec,
2440            runtime.connection_file_path.as_deref(),
2441            self.supervisor_handle.as_ref(),
2442            &runtime.stderr_ring,
2443            runtime.capture_logs_dir.as_deref(),
2444            &runtime.child_roster,
2445            #[cfg(target_os = "linux")]
2446            runtime.cgroup_placement.as_ref(),
2447        ) {
2448            Ok(child) => {
2449                set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2450                self.process_liveness
2451                    .track(spec.module_id.clone(), Arc::clone(&snapshot));
2452                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2453            }
2454            Err(err) => {
2455                if health.critical {
2456                    error!(
2457                        module_id = %spec.module_id,
2458                        program = %spec.program.display(),
2459                        error = %err,
2460                        "critical configured module failed to spawn; marking failed and alerting"
2461                    );
2462                } else {
2463                    error!(
2464                        module_id = %spec.module_id,
2465                        program = %spec.program.display(),
2466                        error = %err,
2467                        "configured module failed to spawn; marking failed and continuing"
2468                    );
2469                }
2470                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2471                Ok(self.supervised_module(spec, runtime, snapshot, None))
2472            }
2473        }
2474    }
2475
2476    fn runtime_config(&self) -> SupervisorRuntimeConfig {
2477        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2478        SupervisorRuntimeConfig {
2479            restart_policy: self.restart_policy,
2480            drain_timeout: self.drain_timeout,
2481            // Shared with this module's roster copy: daemon shutdown waits on
2482            // each child for the module's own drain budget, as resolved now.
2483            child_roster: self
2484                .child_roster
2485                .for_module(Arc::clone(&effective_drain_timeout)),
2486            effective_drain_timeout,
2487            default_drain_timeout: self.drain_timeout,
2488            health: self.health,
2489            connection_file_path: self.connection_file_path.clone(),
2490            capture_logs_dir: self.capture_logs_dir.clone(),
2491            forwarding: self.forwarding.clone(),
2492            supervisor_handle: self.supervisor_handle.clone(),
2493            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
2494            terminal_ring: Arc::new(Mutex::new(
2495                TerminalRing::new(
2496                    TerminalRingConfig::default(),
2497                    self.daemon_start_clock.started_at_ms(),
2498                )
2499                .with_start_clock(self.daemon_start_clock)
2500                .with_journal(self.terminal_journal.clone())
2501                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
2502            )),
2503            spawn_events: self.spawn_events.clone(),
2504            #[cfg(target_os = "linux")]
2505            cgroup_placement: self.cgroup_placement.clone(),
2506            #[cfg(test)]
2507            test_seed_stale_facts_before_enable_spawn: false,
2508        }
2509    }
2510
2511    fn supervised_module(
2512        &self,
2513        spec: ModuleSpec,
2514        runtime: SupervisorRuntimeConfig,
2515        snapshot: SharedSnapshot,
2516        child: Option<SupervisedChild>,
2517    ) -> SupervisedModule {
2518        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
2519            spec: spec.clone(),
2520            health: runtime.health,
2521        }));
2522        let stderr_ring = Arc::clone(&runtime.stderr_ring);
2523        let terminal_ring = Arc::clone(&runtime.terminal_ring);
2524        // The module's OWN policy, which may be its per-module config rather than
2525        // the supervisor-wide one; status must report the budget the supervise
2526        // loop actually enforces.
2527        let restart_policy = runtime.restart_policy;
2528        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
2529        let (tx, rx) = mpsc::channel(4);
2530        let monitor = tokio::spawn(supervise_loop(
2531            spec.clone(),
2532            runtime,
2533            Arc::clone(&self.registry),
2534            Arc::clone(&self.process_liveness),
2535            Arc::clone(&snapshot),
2536            child,
2537            rx,
2538        ));
2539
2540        let module_id = spec.module_id.clone();
2541        let module = SupervisedModule {
2542            inner: Arc::new(SupervisedModuleInner {
2543                module_id: module_id.clone(),
2544                registry: Arc::clone(&self.registry),
2545                snapshot,
2546                configuration,
2547                stderr_ring,
2548                terminal_ring,
2549                commands: tx,
2550                monitor: Mutex::new(Some(monitor)),
2551                restart_policy,
2552                effective_drain_timeout,
2553                provenance_probe: self.provenance_probe.clone(),
2554            }),
2555        };
2556        if let Some(supervisor_handle) = &self.supervisor_handle {
2557            supervisor_handle.apply_identity_configuration(&spec);
2558            supervisor_handle.insert(module.clone());
2559        }
2560        module
2561    }
2562}
2563
2564impl Default for Supervisor {
2565    fn default() -> Self {
2566        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
2567    }
2568}
2569
2570/// Handle to one supervised child process.
2571#[derive(Clone)]
2572pub struct SupervisedModule {
2573    inner: Arc<SupervisedModuleInner>,
2574}
2575
2576struct SupervisedModuleInner {
2577    module_id: String,
2578    registry: Arc<Registry>,
2579    snapshot: SharedSnapshot,
2580    configuration: Arc<Mutex<SupervisedConfiguration>>,
2581    stderr_ring: Arc<Mutex<StderrRing>>,
2582    terminal_ring: Arc<Mutex<TerminalRing>>,
2583    commands: mpsc::Sender<SupervisorCommand>,
2584    monitor: Mutex<Option<JoinHandle<()>>>,
2585    /// Copied from the supervisor's runtime config at spawn so `status()` can
2586    /// report the restart budget without reaching back into the supervisor. The
2587    /// policy is fixed for the process's lifetime, so a copy cannot drift.
2588    restart_policy: RestartPolicy,
2589    effective_drain_timeout: Arc<Mutex<Duration>>,
2590    provenance_probe: ExecutableIdentityProbe,
2591}
2592
2593impl fmt::Debug for SupervisedModule {
2594    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2595        f.debug_struct("SupervisedModule")
2596            .field("module_id", &self.inner.module_id)
2597            .field("status", &self.status())
2598            .finish_non_exhaustive()
2599    }
2600}
2601
2602impl SupervisedModule {
2603    pub fn module_id(&self) -> &str {
2604        &self.inner.module_id
2605    }
2606
2607    /// Test-only: put one probe miss on the streak, the way
2608    /// `handle_health_probe_failure` does, so tests can assert what a later
2609    /// event does to the streak without driving the whole probe loop.
2610    #[cfg(test)]
2611    pub(crate) fn record_health_probe_failure_for_test(
2612        &self,
2613        detail: &str,
2614    ) -> Result<(), SuperviseError> {
2615        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
2616            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
2617            state.health.detail = Some(detail.to_string());
2618        })
2619    }
2620
2621    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
2622        Ok(lock_snapshot(&self.inner.snapshot)?.state)
2623    }
2624
2625    /// The module's retained stderr, newest lines last.
2626    ///
2627    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
2628    /// module, `supervisor.list` renders every module, and putting it in the
2629    /// shared snapshot would make each status read carry a payload almost nobody
2630    /// asked for. Callers that want the text ask for it.
2631    pub fn stderr_tail(
2632        &self,
2633        max_lines: Option<usize>,
2634        max_bytes: Option<usize>,
2635    ) -> StderrTailSnapshot {
2636        self.inner
2637            .stderr_ring
2638            .lock()
2639            .unwrap_or_else(|poisoned| poisoned.into_inner())
2640            .snapshot(max_lines, max_bytes)
2641    }
2642
2643    /// The module's bounded terminal history, oldest retained exit first.
2644    ///
2645    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
2646    /// daemon whose in-memory history was necessarily reset.
2647    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
2648        self.inner
2649            .terminal_ring
2650            .lock()
2651            .unwrap_or_else(|poisoned| poisoned.into_inner())
2652            .snapshot()
2653    }
2654
2655    /// Retained observations from the current ring and all journal generations.
2656    ///
2657    /// Blocking: this reads the journal files. Async callers use
2658    /// [`Self::read_durable_terminal_history`].
2659    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
2660        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
2661    }
2662
2663    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
2664    /// read (up to every retained generation) never occupies a runtime worker.
2665    /// Fails only if the blocking task could not finish (runtime shutdown or a
2666    /// panic in the read).
2667    pub(crate) async fn read_durable_terminal_history(
2668        &self,
2669    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
2670        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
2671        let module_id = self.inner.module_id.clone();
2672        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
2673            .await
2674    }
2675
2676    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
2677        self.status_with_snapshot_lock(&self.inner.snapshot, None)
2678    }
2679
2680    pub(crate) fn record_deliberate_severance(
2681        &self,
2682        identity: ProcessIdentity,
2683    ) -> Result<bool, SuperviseError> {
2684        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2685        if snapshot.pid != Some(identity.pid)
2686            || snapshot.process_start_time != Some(identity.start_time)
2687        {
2688            return Ok(false);
2689        }
2690        snapshot.deliberate_severance = Some(identity);
2691        Ok(true)
2692    }
2693
2694    /// Read status for a channel-0 renderer and report a contended snapshot lock.
2695    ///
2696    /// Internal supervision callers use [`Self::status`] so writer-side machinery
2697    /// does not produce reader-observability logs.
2698    pub(crate) fn status_for_control(
2699        &self,
2700        caller: &'static str,
2701    ) -> Result<ModuleStatus, SuperviseError> {
2702        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
2703    }
2704
2705    fn status_with_snapshot_lock(
2706        &self,
2707        snapshot: &SharedSnapshot,
2708        caller: Option<&'static str>,
2709    ) -> Result<ModuleStatus, SuperviseError> {
2710        let mut guard = match caller {
2711            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
2712            None => lock_snapshot(snapshot)?,
2713        };
2714        // Read the budget through the pruning path so a reader sees the same
2715        // in-window count the restart decision would use, not a stale total.
2716        let restart_count =
2717            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
2718        let snapshot = guard.clone();
2719        drop(guard);
2720        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
2721            SuperviseError::StatePoisoned {
2722                module_id: Some(self.inner.module_id.clone()),
2723            }
2724        })?;
2725        let registration_active = self
2726            .inner
2727            .registry
2728            .get_module(&self.inner.module_id)
2729            .map_err(SuperviseError::Registry)?
2730            .is_some();
2731        let protocol = self.declared_protocol()?;
2732        let running_process =
2733            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
2734        // Registration is the difference between the two protocols and the only
2735        // one: a subc module that has not registered cannot serve a request even
2736        // though its process is up, and a `none` module never registers at all,
2737        // so requiring it there would pin `live` to false for the whole life of
2738        // a perfectly healthy process.
2739        let live = match protocol {
2740            ModuleProtocol::Subc => running_process && registration_active,
2741            ModuleProtocol::None => running_process,
2742        };
2743
2744        Ok(ModuleStatus {
2745            module_id: self.inner.module_id.clone(),
2746            state: snapshot.state,
2747            enabled: snapshot.enabled,
2748            process_alive: snapshot.process_alive,
2749            registration_active,
2750            protocol,
2751            live,
2752            restart_count,
2753            lifetime_restarts: snapshot.lifetime_restarts,
2754            spawn_generation: snapshot.spawn_generation,
2755            max_restarts: self.inner.restart_policy.max_restarts,
2756            restart_window: self.inner.restart_policy.window,
2757            drain_timeout,
2758            restart_backoff: self.inner.restart_policy.backoff,
2759            restart_max_backoff: self.inner.restart_policy.max_backoff,
2760            pid: snapshot.pid,
2761            spawned_at_ms: snapshot.spawned_at_ms,
2762            spawned_from: snapshot.spawned_from,
2763            process_start_time: snapshot.process_start_time,
2764            last_exit: snapshot.last_exit,
2765            health: snapshot.health,
2766        })
2767    }
2768
2769    #[cfg(test)]
2770    pub(crate) fn hold_snapshot_for_test(
2771        &self,
2772        acquired: std::sync::mpsc::Sender<()>,
2773        hold: Duration,
2774    ) -> std::thread::JoinHandle<()> {
2775        let snapshot = Arc::clone(&self.inner.snapshot);
2776        std::thread::spawn(move || {
2777            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
2778            acquired
2779                .send(())
2780                .expect("test receiver waits for snapshot lock");
2781            std::thread::sleep(hold);
2782        })
2783    }
2784
2785    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
2786        let snapshot = match lock_snapshot(&self.inner.snapshot) {
2787            Ok(snapshot) => snapshot.clone(),
2788            Err(_) => {
2789                return subc_control::RunningImageAgreement::Unavailable {
2790                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
2791                };
2792            }
2793        };
2794        self.inner
2795            .provenance_probe
2796            .observe(
2797                snapshot.pid,
2798                snapshot.spawned_from.as_deref(),
2799                snapshot.spawned_file_identity,
2800                snapshot.process_start_time,
2801            )
2802            .await
2803    }
2804
2805    /// Memory and CPU time of the module's current process, read now. Only the
2806    /// process the supervisor spawned is read, not processes it has started.
2807    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
2808        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
2809            Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
2810            Err(_) => {
2811                return subc_control::ChildResourceUsage::Unavailable {
2812                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
2813                }
2814            }
2815        };
2816        crate::child_resources::read(pid, start_time)
2817    }
2818
2819    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
2820        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2821        Ok(match snapshot.state {
2822            ModuleState::Restarting => true,
2823            ModuleState::Failed | ModuleState::Disabled => false,
2824            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
2825        })
2826    }
2827
2828    #[cfg(test)]
2829    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
2830        self.is_warming_with_snapshot_lock(None)
2831    }
2832
2833    pub(crate) fn is_warming_for_control(
2834        &self,
2835        caller: &'static str,
2836    ) -> Result<bool, SuperviseError> {
2837        self.is_warming_with_snapshot_lock(Some(caller))
2838    }
2839
2840    fn is_warming_with_snapshot_lock(
2841        &self,
2842        caller: Option<&'static str>,
2843    ) -> Result<bool, SuperviseError> {
2844        let snapshot = match caller {
2845            Some(caller) => {
2846                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
2847            }
2848            None => lock_snapshot(&self.inner.snapshot)?,
2849        }
2850        .clone();
2851        Ok(matches!(
2852            snapshot.state,
2853            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
2854        ))
2855    }
2856
2857    /// Drain the module and stop monitoring it.
2858    pub async fn drain(&self) -> Result<(), SuperviseError> {
2859        self.stop().await
2860    }
2861
2862    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
2863        match self.state()? {
2864            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2865            ModuleState::Starting
2866            | ModuleState::Running
2867            | ModuleState::Unresponsive
2868            | ModuleState::Restarting
2869            | ModuleState::Draining
2870            | ModuleState::Disabled => {}
2871        }
2872
2873        let (reply_tx, reply_rx) = oneshot::channel();
2874        self.inner
2875            .commands
2876            .send(SupervisorCommand::Retire { reply: reply_tx })
2877            .await
2878            .map_err(|_| SuperviseError::CommandClosed {
2879                module_id: self.inner.module_id.clone(),
2880            })?;
2881        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2882            module_id: self.inner.module_id.clone(),
2883        })?
2884    }
2885
2886    pub async fn stop(&self) -> Result<(), SuperviseError> {
2887        match self.state()? {
2888            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2889            ModuleState::Starting
2890            | ModuleState::Running
2891            | ModuleState::Unresponsive
2892            | ModuleState::Restarting
2893            | ModuleState::Draining
2894            | ModuleState::Disabled => {}
2895        }
2896
2897        let (reply_tx, reply_rx) = oneshot::channel();
2898        self.inner
2899            .commands
2900            .send(SupervisorCommand::Drain { reply: reply_tx })
2901            .await
2902            .map_err(|_| SuperviseError::CommandClosed {
2903                module_id: self.inner.module_id.clone(),
2904            })?;
2905        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2906            module_id: self.inner.module_id.clone(),
2907        })?
2908    }
2909
2910    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
2911        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
2912        let (reply_tx, reply_rx) = oneshot::channel();
2913        self.inner
2914            .commands
2915            .send(SupervisorCommand::Restart {
2916                drain_timeout_ms,
2917                received_at_generation,
2918                queued_at: Instant::now(),
2919                reply: reply_tx,
2920            })
2921            .await
2922            .map_err(|_| SuperviseError::CommandClosed {
2923                module_id: self.inner.module_id.clone(),
2924            })?;
2925        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2926            module_id: self.inner.module_id.clone(),
2927        })?
2928    }
2929
2930    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
2931    /// `supervisor_swap` module. Returns once the swap has cut over (the old
2932    /// process then drains in the background of the supervise loop) or has
2933    /// failed, leaving the old process serving.
2934    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
2935        let (reply_tx, reply_rx) = oneshot::channel();
2936        self.inner
2937            .commands
2938            .send(SupervisorCommand::Swap {
2939                ready_timeout,
2940                reply: reply_tx,
2941            })
2942            .await
2943            .map_err(|_| SuperviseError::CommandClosed {
2944                module_id: self.inner.module_id.clone(),
2945            })?;
2946        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2947            module_id: self.inner.module_id.clone(),
2948        })?
2949    }
2950
2951    pub async fn reload(&self) -> Result<(), SuperviseError> {
2952        let (reply_tx, reply_rx) = oneshot::channel();
2953        self.inner
2954            .commands
2955            .send(SupervisorCommand::Reload { reply: reply_tx })
2956            .await
2957            .map_err(|_| SuperviseError::CommandClosed {
2958                module_id: self.inner.module_id.clone(),
2959            })?;
2960        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2961            module_id: self.inner.module_id.clone(),
2962        })?
2963    }
2964
2965    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
2966        let (reply_tx, reply_rx) = oneshot::channel();
2967        self.inner
2968            .commands
2969            .send(SupervisorCommand::SetEnabled {
2970                enabled,
2971                reply: reply_tx,
2972            })
2973            .await
2974            .map_err(|_| SuperviseError::CommandClosed {
2975                module_id: self.inner.module_id.clone(),
2976            })?;
2977        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2978            module_id: self.inner.module_id.clone(),
2979        })?
2980    }
2981
2982    /// This module's declared protocol, read from the same stored configuration
2983    /// the rescan diff compares and `update_configuration` rewrites, so a status
2984    /// read and the supervise loop can never disagree about which protocol is in
2985    /// force.
2986    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
2987        Ok(self
2988            .inner
2989            .configuration
2990            .lock()
2991            .map_err(|_| SuperviseError::StatePoisoned {
2992                module_id: Some(self.inner.module_id.clone()),
2993            })?
2994            .spec
2995            .protocol)
2996    }
2997
2998    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
2999        let configuration =
3000            self.inner
3001                .configuration
3002                .lock()
3003                .map_err(|_| SuperviseError::StatePoisoned {
3004                    module_id: Some(self.inner.module_id.clone()),
3005                })?;
3006        Ok((configuration.spec.clone(), configuration.health))
3007    }
3008
3009    /// Replace this module's launch spec, keeping its health and drain policy,
3010    /// the way a rescan does for a changed config entry. The running process is
3011    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3012    #[cfg(any(test, feature = "test-support"))]
3013    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3014        let (_, health) = self.configuration()?;
3015        let drain_timeout_ms = u64::try_from(
3016            self.inner
3017                .effective_drain_timeout
3018                .lock()
3019                .unwrap_or_else(|poisoned| poisoned.into_inner())
3020                .as_millis(),
3021        )
3022        .ok();
3023        self.update_configuration(spec, health, drain_timeout_ms)
3024            .await
3025    }
3026
3027    pub(crate) async fn update_configuration(
3028        &self,
3029        spec: ModuleSpec,
3030        health: HealthConfig,
3031        drain_timeout_ms: Option<u64>,
3032    ) -> Result<(), SuperviseError> {
3033        if spec.module_id != self.inner.module_id {
3034            return Err(SuperviseError::InvalidSpec {
3035                reason: "a supervised module's module_id cannot be changed".to_string(),
3036            });
3037        }
3038        validate_spec(&spec)?;
3039        let (reply_tx, reply_rx) = oneshot::channel();
3040        self.inner
3041            .commands
3042            .send(SupervisorCommand::UpdateConfiguration {
3043                spec: spec.clone(),
3044                health,
3045                drain_timeout_ms,
3046                reply: reply_tx,
3047            })
3048            .await
3049            .map_err(|_| SuperviseError::CommandClosed {
3050                module_id: self.inner.module_id.clone(),
3051            })?;
3052        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3053            module_id: self.inner.module_id.clone(),
3054        })?;
3055        let mut configuration =
3056            self.inner
3057                .configuration
3058                .lock()
3059                .map_err(|_| SuperviseError::StatePoisoned {
3060                    module_id: Some(self.inner.module_id.clone()),
3061                })?;
3062        configuration.spec = spec;
3063        configuration.health = health;
3064        Ok(())
3065    }
3066}
3067
3068impl Drop for SupervisedModuleInner {
3069    fn drop(&mut self) {
3070        let Ok(mut monitor) = self.monitor.lock() else {
3071            return;
3072        };
3073        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3074            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3075                state.state = ModuleState::Stopped;
3076                clear_current_process_facts(state);
3077            });
3078            monitor.abort();
3079        }
3080        let _ = monitor.take();
3081    }
3082}
3083
3084#[derive(Debug)]
3085enum SupervisorCommand {
3086    Drain {
3087        reply: oneshot::Sender<Result<(), SuperviseError>>,
3088    },
3089    Retire {
3090        reply: oneshot::Sender<Result<(), SuperviseError>>,
3091    },
3092    Restart {
3093        /// Operator override for this one restart's drain budget, in ms. `None`
3094        /// uses the module's configured/default budget; `Some(0)` cuts
3095        /// immediately (wedge bounce: a stuck request never settles, so
3096        /// waiting only delays recovery).
3097        drain_timeout_ms: Option<u64>,
3098        /// The module's `spawn_generation` when the request was received, before
3099        /// it waited in the command queue. A queued restart whose module has
3100        /// since spawned a newer process is already satisfied (see the handler).
3101        received_at_generation: u64,
3102        /// When the request entered the command queue, so the handler can log
3103        /// how long it waited behind the loop's other work.
3104        queued_at: Instant,
3105        reply: oneshot::Sender<Result<(), SuperviseError>>,
3106    },
3107    Reload {
3108        reply: oneshot::Sender<Result<(), SuperviseError>>,
3109    },
3110    SetEnabled {
3111        enabled: bool,
3112        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3113    },
3114    UpdateConfiguration {
3115        spec: ModuleSpec,
3116        health: HealthConfig,
3117        /// Per-module drain override from the new config; `None` re-resolves to
3118        /// the supervisor-wide default.
3119        drain_timeout_ms: Option<u64>,
3120        reply: oneshot::Sender<()>,
3121    },
3122    Swap {
3123        /// How long the candidate may take to register and declare itself
3124        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3125        ready_timeout: Option<Duration>,
3126        /// Answered at cutover or failure; the incumbent's drain follows.
3127        reply: oneshot::Sender<Result<(), SuperviseError>>,
3128    },
3129}
3130
3131#[derive(Debug)]
3132pub enum SuperviseError {
3133    InvalidSpec {
3134        reason: String,
3135    },
3136    Spawn {
3137        program: PathBuf,
3138        source: io::Error,
3139        cgroup_path: Option<PathBuf>,
3140    },
3141    Cgroup {
3142        module_id: String,
3143        source: io::Error,
3144    },
3145    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3146    /// than spawn a reserved module without its identity binding.
3147    LaunchNonce {
3148        reason: String,
3149    },
3150    Wait {
3151        module_id: String,
3152        source: io::Error,
3153    },
3154    Kill {
3155        module_id: String,
3156        source: io::Error,
3157    },
3158    Forwarding(ForwardingError),
3159    Registry(RegistryError),
3160    ReloadUnavailable {
3161        module_id: String,
3162        reason: String,
3163    },
3164    /// An operator restart/reload was requested for a module that is currently
3165    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3166    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3167    /// by a restart, so these commands are rejected instead of re-enabling it.
3168    Disabled {
3169        module_id: String,
3170    },
3171    ReloadFailed {
3172        module_id: String,
3173        reason: String,
3174    },
3175    RegistrationStillActive {
3176        module_id: String,
3177        waited: Duration,
3178    },
3179    StatePoisoned {
3180        module_id: Option<String>,
3181    },
3182    CommandClosed {
3183        module_id: String,
3184    },
3185    /// A restart or reload arrived while a swap's candidate was warming. The
3186    /// swap owns the module until it cuts over or fails; a stop or disable
3187    /// would have aborted it instead.
3188    SwapInProgress {
3189        module_id: String,
3190    },
3191    /// A swap was refused before anything was spawned.
3192    SwapRefused {
3193        module_id: String,
3194        reason: SwapRefusal,
3195    },
3196    /// A swap spawned a candidate and gave up on it. The candidate has been
3197    /// killed and its slot freed; the incumbent was left serving and was never
3198    /// drained, except in the one `CutoverLost` case described on that arm.
3199    SwapFailed {
3200        module_id: String,
3201        arm: SwapFailureArm,
3202        detail: String,
3203        /// How the candidate exited, when it exited on its own before the
3204        /// supervisor gave up on it.
3205        candidate_exit: Option<ExitReport>,
3206    },
3207}
3208
3209/// Why a swap was refused before a candidate was spawned.
3210#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3211pub enum SwapRefusal {
3212    /// The module's config does not declare `overlap: "safe"`.
3213    OverlapExclusive,
3214    /// The module is not registered, so there is no incumbent to keep serving
3215    /// and nothing a swap would improve on; a plain restart is the tool.
3216    NotRegistered,
3217    /// The module does not speak the subc wire, so a candidate could never
3218    /// register or declare itself ready.
3219    ProtocolNone,
3220    /// The supervisor lacks the forwarding table (to cut routes over) or the
3221    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3222    NotConfigured,
3223    /// A swap is already open for this module.
3224    AlreadySwapping,
3225}
3226
3227impl SwapRefusal {
3228    pub fn as_str(self) -> &'static str {
3229        match self {
3230            Self::OverlapExclusive => "overlap_exclusive",
3231            Self::NotRegistered => "not_registered",
3232            Self::ProtocolNone => "protocol_none",
3233            Self::NotConfigured => "not_configured",
3234            Self::AlreadySwapping => "already_swapping",
3235        }
3236    }
3237}
3238
3239/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3240/// serving and undrained; see `CutoverLost`.
3241#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3242pub enum SwapFailureArm {
3243    /// The candidate process could not be started.
3244    SpawnFailed,
3245    /// The candidate did not register within the readiness budget.
3246    NeverRegistered,
3247    /// The candidate registered but did not declare itself ready in time.
3248    NeverReady,
3249    /// The candidate exited before cutover.
3250    CandidateExited,
3251    /// The candidate declared itself ready but failed its health probe.
3252    CandidateUnhealthy,
3253    /// An operator stop, disable or retire arrived while the candidate warmed.
3254    /// The candidate was killed and the operator's command then carried out on
3255    /// the incumbent.
3256    Interrupted,
3257    /// The candidate's connection closed at the moment of cutover. If it
3258    /// closed before forwarding moved, the incumbent is untouched. If it closed
3259    /// between the forwarding and registry halves of cutover, forwarding can no
3260    /// longer route to the incumbent, so the module is restarted plainly.
3261    CutoverLost,
3262}
3263
3264impl SwapFailureArm {
3265    pub fn as_str(self) -> &'static str {
3266        match self {
3267            Self::SpawnFailed => "spawn_failed",
3268            Self::NeverRegistered => "never_registered",
3269            Self::NeverReady => "never_ready",
3270            Self::CandidateExited => "candidate_exited",
3271            Self::CandidateUnhealthy => "candidate_unhealthy",
3272            Self::Interrupted => "interrupted",
3273            Self::CutoverLost => "cutover_lost",
3274        }
3275    }
3276}
3277
3278impl fmt::Display for SuperviseError {
3279    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3280        match self {
3281            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3282            Self::Spawn {
3283                program,
3284                source,
3285                cgroup_path: Some(cgroup_path),
3286            } => write!(
3287                f,
3288                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3289                cgroup_path.display(),
3290                program.display()
3291            ),
3292            Self::Spawn {
3293                program,
3294                source,
3295                cgroup_path: None,
3296            } => write!(
3297                f,
3298                "failed to spawn module '{}': {source}",
3299                program.display()
3300            ),
3301            Self::Cgroup { module_id, source } => {
3302                write!(
3303                    f,
3304                    "failed to prepare cgroup for module '{module_id}': {source}"
3305                )
3306            }
3307            Self::LaunchNonce { reason } => {
3308                write!(
3309                    f,
3310                    "failed to generate reserved-module launch nonce: {reason}"
3311                )
3312            }
3313            Self::Wait { module_id, source } => {
3314                write!(f, "failed to wait for module '{module_id}': {source}")
3315            }
3316            Self::Kill { module_id, source } => {
3317                write!(f, "failed to kill module '{module_id}': {source}")
3318            }
3319            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3320            Self::Registry(err) => write!(f, "registry error: {err}"),
3321            Self::ReloadUnavailable { module_id, reason } => {
3322                write!(f, "reload unavailable for module '{module_id}': {reason}")
3323            }
3324            Self::Disabled { module_id } => {
3325                write!(
3326                    f,
3327                    "module '{module_id}' is disabled; enable it before restart or reload"
3328                )
3329            }
3330            Self::ReloadFailed { module_id, reason } => {
3331                write!(f, "reload failed for module '{module_id}': {reason}")
3332            }
3333            Self::RegistrationStillActive { module_id, waited } => write!(
3334                f,
3335                "module '{module_id}' registration remained active after waiting {waited:?}"
3336            ),
3337            Self::StatePoisoned { module_id } => match module_id {
3338                Some(module_id) => {
3339                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3340                }
3341                None => write!(f, "supervisor state was poisoned"),
3342            },
3343            Self::CommandClosed { module_id } => {
3344                write!(
3345                    f,
3346                    "supervisor command channel for module '{module_id}' is closed"
3347                )
3348            }
3349            Self::SwapInProgress { module_id } => write!(
3350                f,
3351                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3352            ),
3353            Self::SwapRefused { module_id, reason } => match reason {
3354                SwapRefusal::OverlapExclusive => write!(
3355                    f,
3356                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3357                ),
3358                SwapRefusal::NotRegistered => write!(
3359                    f,
3360                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3361                ),
3362                SwapRefusal::ProtocolNone => write!(
3363                    f,
3364                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3365                ),
3366                SwapRefusal::NotConfigured => write!(
3367                    f,
3368                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3369                ),
3370                SwapRefusal::AlreadySwapping => {
3371                    write!(f, "module '{module_id}' is already being swapped")
3372                }
3373            },
3374            Self::SwapFailed {
3375                module_id,
3376                arm,
3377                detail,
3378                ..
3379            } => write!(
3380                f,
3381                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3382                arm.as_str()
3383            ),
3384        }
3385    }
3386}
3387
3388impl Error for SuperviseError {
3389    fn source(&self) -> Option<&(dyn Error + 'static)> {
3390        match self {
3391            Self::Spawn { source, .. }
3392            | Self::Cgroup { source, .. }
3393            | Self::Wait { source, .. }
3394            | Self::Kill { source, .. } => Some(source),
3395            Self::Forwarding(err) => Some(err),
3396            Self::Registry(err) => Some(err),
3397            Self::LaunchNonce { .. }
3398            | Self::InvalidSpec { .. }
3399            | Self::ReloadUnavailable { .. }
3400            | Self::Disabled { .. }
3401            | Self::ReloadFailed { .. }
3402            | Self::RegistrationStillActive { .. }
3403            | Self::StatePoisoned { .. }
3404            | Self::CommandClosed { .. }
3405            | Self::SwapInProgress { .. }
3406            | Self::SwapRefused { .. }
3407            | Self::SwapFailed { .. } => None,
3408        }
3409    }
3410}
3411
3412pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3413    if spec.module_id.trim().is_empty() {
3414        return Err(SuperviseError::InvalidSpec {
3415            reason: "module_id must not be empty".to_string(),
3416        });
3417    }
3418
3419    Ok(())
3420}
3421
3422#[derive(Debug, Default)]
3423struct HealthProbeRuntime {
3424    registered_connection: Option<crate::ConnectionId>,
3425    advertised: bool,
3426    next_probe_at: Option<Instant>,
3427    probe_index: u64,
3428}
3429
3430impl HealthProbeRuntime {
3431    fn refresh_registration(
3432        &mut self,
3433        spec: &ModuleSpec,
3434        runtime: &SupervisorRuntimeConfig,
3435        registry: &Registry,
3436        snapshot: &SharedSnapshot,
3437    ) {
3438        // THE PROBE GATE FOR A MODULE THAT SPEAKS NO SUBC WIRE, placed here
3439        // because this is the only place that ever arms a probe: leaving
3440        // `advertised` false and `next_probe_at` empty makes `due()` false
3441        // forever, so `run_health_probe_cycle` -- and with it every arm of
3442        // `probe_module_health`, including the one that reads an absent
3443        // registration as proof the module is gone and escalates to a restart --
3444        // is unreachable for this module.
3445        //
3446        // That arm is right for a subc module and is exactly wrong here: a
3447        // `protocol: "none"` module never registers by declaration, so the
3448        // absence it would classify is the module working as configured.
3449        if spec.protocol == ModuleProtocol::None {
3450            self.registered_connection = None;
3451            self.advertised = false;
3452            self.next_probe_at = None;
3453            return;
3454        }
3455
3456        let registration = match registry.get_module(&spec.module_id) {
3457            Ok(registration) => registration,
3458            Err(err) => {
3459                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3460                self.advertised = false;
3461                self.next_probe_at = None;
3462                return;
3463            }
3464        };
3465
3466        let Some(registration) = registration else {
3467            self.registered_connection = None;
3468            self.advertised = false;
3469            self.next_probe_at = None;
3470            return;
3471        };
3472
3473        let advertised = registration
3474            .control_ops
3475            .iter()
3476            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
3477        if !advertised {
3478            self.registered_connection = Some(registration.connection_id);
3479            self.advertised = false;
3480            self.next_probe_at = None;
3481            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3482                state.health.status = SupervisorHealthStatus::Unknown;
3483                state.health.consecutive_failures = 0;
3484                state.health.last_probe_ms = None;
3485                state.health.detail = None;
3486                state.health.metrics = None;
3487            });
3488            return;
3489        }
3490
3491        let reregistered = self.registered_connection != Some(registration.connection_id);
3492        self.registered_connection = Some(registration.connection_id);
3493        self.advertised = true;
3494        if reregistered || self.next_probe_at.is_none() {
3495            self.probe_index = 0;
3496            self.next_probe_at = Some(
3497                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3498            );
3499            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3500                state.health.status = SupervisorHealthStatus::Unknown;
3501                state.health.consecutive_failures = 0;
3502                state.health.detail = None;
3503                state.health.metrics = None;
3504            });
3505        }
3506    }
3507
3508    fn wake_after(&self) -> Duration {
3509        if !self.advertised {
3510            return REGISTRY_RELEASE_POLL;
3511        }
3512        self.next_probe_at
3513            .map(|next| next.saturating_duration_since(Instant::now()))
3514            .unwrap_or(REGISTRY_RELEASE_POLL)
3515    }
3516
3517    fn due(&self) -> bool {
3518        self.advertised
3519            && self
3520                .next_probe_at
3521                .is_some_and(|next| Instant::now() >= next)
3522    }
3523
3524    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
3525        self.probe_index = self.probe_index.wrapping_add(1);
3526        self.next_probe_at = Some(
3527            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
3528        );
3529    }
3530}
3531
3532/// What a failed health probe actually OBSERVED, kept apart from how it reads.
3533///
3534/// This was a struct with a single `message: String`, and every one of the
3535/// fifteen construction sites collapsed into it. Each site knows exactly what it
3536/// saw -- the lane is gone, the module did not answer in time, the module
3537/// answered with the wrong thing -- and `handle_health_probe_failure` then
3538/// treated all of them identically: increment a counter, compare to a threshold,
3539/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
3540/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
3541///
3542/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
3543///
3544/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
3545///   answer on it again.
3546/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
3547///   AND with a perfectly healthy one that lost a CPU race -- which is what
3548///   happens under machine load, and is how this supervisor killed a healthy
3549///   module three times in one day.
3550/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
3551///   Restarting on it is defensible, but it is not the silence case and should
3552///   never be counted as one.
3553/// * `Misconfigured` is a daemon-side fault. The module has not been asked
3554///   anything, so it cannot be evidence about the module at all.
3555///
3556/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
3557/// one that fires most often, and while every variant collapsed into one string
3558/// it carried the same weight as the strongest.
3559///
3560/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
3561/// DESIGN and a reader stopping at it gets the build backwards: the restart
3562/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
3563/// probes still increment the failure streak and drive escalation at the
3564/// threshold (see `is_proof_of_death` below for why that is deliberate and
3565/// what gates the change). Absence of evidence restarts modules today.
3566#[derive(Debug)]
3567enum HealthProbeEvidence {
3568    /// The module's control lane is gone. Proof of death.
3569    LaneDead,
3570    /// No reply within the deadline. Proves nothing about the module's state.
3571    NoAnswer,
3572    /// The module replied, but not with a usable health report. Proves it is alive.
3573    BadAnswer,
3574    /// The daemon could not ask. Says nothing about the module.
3575    Misconfigured,
3576}
3577
3578#[derive(Debug)]
3579struct HealthProbeError {
3580    evidence: HealthProbeEvidence,
3581    message: String,
3582}
3583
3584impl HealthProbeError {
3585    fn lane_dead(message: impl Into<String>) -> Self {
3586        Self::with(HealthProbeEvidence::LaneDead, message)
3587    }
3588
3589    fn no_answer(message: impl Into<String>) -> Self {
3590        Self::with(HealthProbeEvidence::NoAnswer, message)
3591    }
3592
3593    fn bad_answer(message: impl Into<String>) -> Self {
3594        Self::with(HealthProbeEvidence::BadAnswer, message)
3595    }
3596
3597    fn misconfigured(message: impl Into<String>) -> Self {
3598        Self::with(HealthProbeEvidence::Misconfigured, message)
3599    }
3600
3601    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
3602        Self {
3603            evidence,
3604            message: message.into(),
3605        }
3606    }
3607
3608    /// Whether this observation is proof the module cannot serve.
3609    ///
3610    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
3611    /// variant that fires under CPU starvation, and treating it as proof is the
3612    /// defect this enum exists to make impossible to reintroduce silently.
3613    ///
3614    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
3615    /// to restart also needs a bound for the case it excludes -- a genuinely
3616    /// wedged module, alive but never answering -- and that bound must come from
3617    /// the distribution of real late-answer latencies, which nothing measures
3618    /// yet. Landing the classification first makes the later change a one-line
3619    /// decision against evidence that already exists, rather than two unproven
3620    /// changes at once.
3621    #[allow(dead_code)]
3622    fn is_proof_of_death(&self) -> bool {
3623        matches!(self.evidence, HealthProbeEvidence::LaneDead)
3624    }
3625
3626    /// Short stable label for logs and the health snapshot.
3627    ///
3628    /// An operator reading `ck health` currently cannot tell "the module is gone"
3629    /// from "the module did not answer in five seconds", because both render as
3630    /// prose in the same field. These labels are what make the two
3631    /// distinguishable at a glance, and they are what a later restart-policy
3632    /// change will be argued from.
3633    fn label(&self) -> &'static str {
3634        match self.evidence {
3635            HealthProbeEvidence::LaneDead => "lane-dead",
3636            HealthProbeEvidence::NoAnswer => "no-answer",
3637            HealthProbeEvidence::BadAnswer => "bad-answer",
3638            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
3639        }
3640    }
3641}
3642
3643impl fmt::Display for HealthProbeError {
3644    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3645        f.write_str(&self.message)
3646    }
3647}
3648
3649async fn run_health_probe_cycle(
3650    spec: &ModuleSpec,
3651    runtime: &SupervisorRuntimeConfig,
3652    registry: &Registry,
3653    process_liveness: &SupervisorProcessLiveness,
3654    snapshot: &SharedSnapshot,
3655    child: &mut Option<SupervisedChild>,
3656) {
3657    let now_ms = unix_ms_now();
3658    match probe_module_health(&spec.module_id, runtime, None).await {
3659        Ok(report) => {
3660            handle_health_report(
3661                spec,
3662                runtime,
3663                registry,
3664                process_liveness,
3665                snapshot,
3666                child,
3667                report,
3668                now_ms,
3669            )
3670            .await;
3671        }
3672        Err(err) => {
3673            handle_health_probe_failure(
3674                spec,
3675                runtime,
3676                registry,
3677                process_liveness,
3678                snapshot,
3679                child,
3680                err,
3681                now_ms,
3682            )
3683            .await;
3684        }
3685    }
3686}
3687
3688async fn probe_module_health(
3689    module_id: &str,
3690    runtime: &SupervisorRuntimeConfig,
3691    drain_deadline: Option<Instant>,
3692) -> Result<HealthReport, HealthProbeError> {
3693    let Some(forwarding) = runtime.forwarding.as_ref() else {
3694        return Err(HealthProbeError::misconfigured(
3695            "supervisor was not configured with a forwarding table",
3696        ));
3697    };
3698    let probe_started_at = Instant::now();
3699    let mut deadline = probe_started_at + runtime.health.deadline;
3700    if let Some(drain_deadline) = drain_deadline {
3701        deadline = deadline.min(drain_deadline);
3702    }
3703    let pending = if drain_deadline.is_some() {
3704        forwarding.begin_drain_health_probe_rpc_for(
3705            module_id,
3706            MODULE_CONTROL_OP_HEALTH_CHECK,
3707            probe_started_at,
3708            deadline,
3709        )
3710    } else {
3711        forwarding.begin_health_probe_rpc_for(
3712            module_id,
3713            MODULE_CONTROL_OP_HEALTH_CHECK,
3714            probe_started_at,
3715            deadline,
3716        )
3717    }
3718    .map_err(|err| {
3719        // The endpoint is not registered, so there is no live control lane to
3720        // ask. That is the module being absent, not slow.
3721        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3722    })?;
3723    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3724}
3725
3726/// [`probe_module_health`] for one endpoint rather than the id's active one.
3727///
3728/// A swap probes two processes that no by-id lookup reaches: its candidate
3729/// before cutover, and its superseded incumbent (for busy gauges) while the
3730/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
3731/// bounds the by-id drain probe.
3732async fn probe_endpoint_health(
3733    endpoint: crate::ModuleEndpointId,
3734    runtime: &SupervisorRuntimeConfig,
3735    deadline_cap: Option<Instant>,
3736) -> Result<HealthReport, HealthProbeError> {
3737    let Some(forwarding) = runtime.forwarding.as_ref() else {
3738        return Err(HealthProbeError::misconfigured(
3739            "supervisor was not configured with a forwarding table",
3740        ));
3741    };
3742    let probe_started_at = Instant::now();
3743    let mut deadline = probe_started_at + runtime.health.deadline;
3744    if let Some(cap) = deadline_cap {
3745        deadline = deadline.min(cap);
3746    }
3747    let pending = forwarding
3748        .begin_endpoint_health_probe_rpc_for(
3749            endpoint,
3750            MODULE_CONTROL_OP_HEALTH_CHECK,
3751            probe_started_at,
3752            deadline,
3753        )
3754        .map_err(|err| {
3755            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3756        })?;
3757    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3758}
3759
3760/// Send a begun health probe and classify its answer.
3761async fn await_health_probe(
3762    forwarding: &ForwardingTable,
3763    pending: PendingModuleControlRpc,
3764    deadline: Instant,
3765    probe_budget: Duration,
3766) -> Result<HealthReport, HealthProbeError> {
3767    let PendingModuleControlRpc {
3768        endpoint,
3769        module_sink,
3770        negotiated_ver,
3771        corr,
3772        receiver,
3773    } = pending;
3774    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
3775        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
3776    })?;
3777    let frame = Frame::build_with_version(
3778        negotiated_ver,
3779        FrameType::Request,
3780        control_flags(),
3781        0,
3782        0,
3783        corr,
3784        body,
3785    )
3786    .map_err(|err| {
3787        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
3788    })?;
3789
3790    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
3791    // blocks waiting for capacity when the module's egress queue is full, and an
3792    // unbounded await here freezes the whole supervision actor (it stops polling
3793    // Child::wait and supervisor commands), making the module unrecoverable
3794    // in-band. On timeout the probe fails like any transport failure.
3795    match timeout_at(deadline, module_sink.send(frame)).await {
3796        Ok(Ok(())) => {}
3797        Ok(Err(err)) => {
3798            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3799            // A closed sink means the module's egress channel is gone -- the
3800            // receiving half is dropped when its connection tears down. Proof.
3801            return Err(HealthProbeError::lane_dead(format!(
3802                "failed to send health.check: {err}"
3803            )));
3804        }
3805        Err(_elapsed) => {
3806            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3807            // A full egress queue means the module is not draining its socket, which
3808            // is consistent with a wedged module AND with one whose reader is merely
3809            // starved. Silence, not proof.
3810            return Err(HealthProbeError::no_answer(
3811                "health.check send timed out before enqueue (module egress full)",
3812            ));
3813        }
3814    }
3815
3816    match timeout_at(deadline, receiver).await {
3817        // Each arm records WHAT WAS OBSERVED. Four of them are the module
3818        // demonstrably answering -- rejected, non-health, malformed, wrong op --
3819        // and those prove it is alive even though the probe failed.
3820        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
3821            response.health_report().ok_or_else(|| {
3822                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
3823            })
3824        }
3825        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
3826            format!("health.check rejected: {}", body.message),
3827        )),
3828        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
3829            Err(HealthProbeError::lane_dead(message))
3830        }
3831        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
3832            Err(HealthProbeError::bad_answer(message))
3833        }
3834        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
3835            Err(HealthProbeError::bad_answer(format!(
3836                "expected module-control op '{expected}', got '{actual}'"
3837            )))
3838        }
3839        // A reply that crosses the deadline before this waiter observes it is
3840        // still proof of life. The forwarding path records its end-to-end latency
3841        // before delivering this classification.
3842        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
3843            "module answered health.check after its daemon deadline",
3844        )),
3845        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
3846            "health.check waiter was canceled before the module responded",
3847        )),
3848        Err(_) => {
3849            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
3850            Err(HealthProbeError::no_answer(format!(
3851                "module did not answer health.check within {probe_budget:?}"
3852            )))
3853        }
3854    }
3855}
3856
3857#[allow(clippy::too_many_arguments)]
3858async fn handle_health_report(
3859    spec: &ModuleSpec,
3860    runtime: &SupervisorRuntimeConfig,
3861    registry: &Registry,
3862    process_liveness: &SupervisorProcessLiveness,
3863    snapshot: &SharedSnapshot,
3864    child: &mut Option<SupervisedChild>,
3865    report: HealthReport,
3866    now_ms: u64,
3867) {
3868    let status = supervisor_health_status(report.status);
3869    let detail = report.detail.clone();
3870    let metrics = truncate_health_metrics(report.metrics);
3871    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3872        state.health.status = status;
3873        state.health.last_probe_ms = Some(now_ms);
3874        state.health.detail = detail.clone();
3875        state.health.metrics = metrics.clone();
3876        state.health.consecutive_failures = 0;
3877    });
3878
3879    let action = match report.status {
3880        HealthStatus::Ok => return,
3881        HealthStatus::Degraded => runtime.health.on_degraded,
3882        HealthStatus::Failing => runtime.health.on_failing,
3883    };
3884    apply_l3_health_action(
3885        spec,
3886        runtime,
3887        registry,
3888        process_liveness,
3889        snapshot,
3890        child,
3891        status,
3892        detail.as_deref(),
3893        action,
3894        now_ms,
3895    )
3896    .await;
3897}
3898
3899#[allow(clippy::too_many_arguments)]
3900async fn handle_health_probe_failure(
3901    spec: &ModuleSpec,
3902    runtime: &SupervisorRuntimeConfig,
3903    registry: &Registry,
3904    process_liveness: &SupervisorProcessLiveness,
3905    snapshot: &SharedSnapshot,
3906    child: &mut Option<SupervisedChild>,
3907    err: HealthProbeError,
3908    now_ms: u64,
3909) {
3910    let threshold = runtime.health.failure_threshold.max(1);
3911    let mut failures = 0;
3912    // Carry the evidence class into the operator-visible detail. Without it,
3913    // "module did not answer within 5s" and "the control lane is gone" are two
3914    // prose strings in the same field, and the reader has to know the codebase to
3915    // tell which one is proof of anything.
3916    let detail = format!("[{}] {err}", err.label());
3917    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3918        state.health.last_probe_ms = Some(now_ms);
3919        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3920        state.health.detail = Some(detail.clone());
3921        state.health.metrics = None;
3922        failures = state.health.consecutive_failures;
3923    });
3924
3925    if failures < threshold {
3926        warn!(
3927            module_id = %spec.module_id,
3928            consecutive_failures = failures,
3929            threshold,
3930            evidence = err.label(),
3931            detail = %detail,
3932            "health.check probe failed"
3933        );
3934        return;
3935    }
3936
3937    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3938        state.state = ModuleState::Unresponsive;
3939        state.health.status = SupervisorHealthStatus::Unresponsive;
3940    });
3941    // The evidence class is logged at the kill site because this is the line an
3942    // operator reads after an unexplained restart. A streak of `no-answer` under
3943    // machine load is the known false-positive shape; a `lane-dead` is not.
3944    if runtime.health.critical {
3945        error!(
3946            module_id = %spec.module_id,
3947            status = "unresponsive",
3948            evidence = err.label(),
3949            detail = %detail,
3950            "critical module health alert"
3951        );
3952    } else {
3953        warn!(
3954            module_id = %spec.module_id,
3955            status = "unresponsive",
3956            evidence = err.label(),
3957            detail = %detail,
3958            "module health threshold breached"
3959        );
3960    }
3961    if let Err(err) = health_restart_child(
3962        spec,
3963        runtime,
3964        registry,
3965        process_liveness,
3966        snapshot,
3967        child,
3968        SupervisorHealthStatus::Unresponsive,
3969        Some(&detail),
3970        now_ms,
3971    )
3972    .await
3973    {
3974        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
3975    }
3976}
3977
3978#[allow(clippy::too_many_arguments)]
3979async fn apply_l3_health_action(
3980    spec: &ModuleSpec,
3981    runtime: &SupervisorRuntimeConfig,
3982    registry: &Registry,
3983    process_liveness: &SupervisorProcessLiveness,
3984    snapshot: &SharedSnapshot,
3985    child: &mut Option<SupervisedChild>,
3986    status: SupervisorHealthStatus,
3987    detail: Option<&str>,
3988    action: HealthAction,
3989    now_ms: u64,
3990) {
3991    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
3992    match action {
3993        HealthAction::Report => {
3994            info!(
3995                module_id = %spec.module_id,
3996                status = ?status,
3997                detail,
3998                "module reported non-ok health"
3999            );
4000        }
4001        HealthAction::Alert => {
4002            error!(
4003                module_id = %spec.module_id,
4004                status = ?status,
4005                detail,
4006                "module health alert"
4007            );
4008        }
4009        HealthAction::Restart => {
4010            if let Err(err) = health_restart_child(
4011                spec,
4012                runtime,
4013                registry,
4014                process_liveness,
4015                snapshot,
4016                child,
4017                status,
4018                detail,
4019                now_ms,
4020            )
4021            .await
4022            {
4023                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4024            }
4025        }
4026    }
4027}
4028
4029#[allow(clippy::too_many_arguments)]
4030async fn health_restart_child(
4031    spec: &ModuleSpec,
4032    runtime: &SupervisorRuntimeConfig,
4033    registry: &Registry,
4034    process_liveness: &SupervisorProcessLiveness,
4035    snapshot: &SharedSnapshot,
4036    child: &mut Option<SupervisedChild>,
4037    status: SupervisorHealthStatus,
4038    detail: Option<&str>,
4039    now_ms: u64,
4040) -> Result<(), SuperviseError> {
4041    let (enabled, schedule) = {
4042        let mut state = lock_snapshot(snapshot)?;
4043        let enabled = state.enabled;
4044        let schedule = if enabled {
4045            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4046        } else {
4047            None
4048        };
4049        (enabled, schedule)
4050    };
4051
4052    if !enabled {
4053        return Err(SuperviseError::Disabled {
4054            module_id: spec.module_id.clone(),
4055        });
4056    }
4057
4058    if schedule.is_none() {
4059        record_health_action(snapshot, &spec.module_id, "disabled".to_string(), now_ms);
4060        error!(
4061            module_id = %spec.module_id,
4062            status = ?status,
4063            detail,
4064            max_restarts = runtime.restart_policy.max_restarts,
4065            window_secs = runtime.restart_policy.window.as_secs(),
4066            "health restart budget exhausted; disabling module"
4067        );
4068        let stop_notice = begin_forwarding_drain_if_configured(
4069            spec,
4070            runtime,
4071            registry,
4072            snapshot,
4073            Some(false),
4074            RouteCloseReason::Disable,
4075        )
4076        .await?;
4077        drain_optional_child(
4078            &spec.module_id,
4079            spec.protocol,
4080            stop_notice,
4081            registry,
4082            snapshot,
4083            &runtime.terminal_ring,
4084            &runtime.spawn_events,
4085            child,
4086            runtime.drain_timeout,
4087            ModuleState::Disabled,
4088            Some(false),
4089        )
4090        .await?;
4091        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4092        return Ok(());
4093    }
4094
4095    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4096    let mut restart_count = 0;
4097    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4098        restart_count = state.crash_restarts.len();
4099        state.state = ModuleState::Unresponsive;
4100        state.health.status = status;
4101        state.health.last_action = Some(HealthAction::Restart.to_string());
4102        state.health.last_action_ms = Some(now_ms);
4103    })?;
4104    warn!(
4105        module_id = %spec.module_id,
4106        status = ?status,
4107        detail,
4108        restart_count,
4109        restart_in_window = schedule.restart_in_window,
4110        delay_ms = schedule.delay.as_millis() as u64,
4111        "health-triggered module restart"
4112    );
4113
4114    let stop_notice = begin_forwarding_drain_if_configured(
4115        spec,
4116        runtime,
4117        registry,
4118        snapshot,
4119        Some(true),
4120        RouteCloseReason::Restart,
4121    )
4122    .await?;
4123    drain_optional_child(
4124        &spec.module_id,
4125        spec.protocol,
4126        stop_notice,
4127        registry,
4128        snapshot,
4129        &runtime.terminal_ring,
4130        &runtime.spawn_events,
4131        child,
4132        runtime.drain_timeout,
4133        ModuleState::Restarting,
4134        Some(true),
4135    )
4136    .await?;
4137    sleep(schedule.delay).await;
4138    // The backoff may have outlasted the restart it was counting down to: an
4139    // operator disable or drain in between moves the snapshot out of
4140    // `Restarting`, and that stop must win over this respawn.
4141    if !respawn_still_pending(snapshot) {
4142        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4143        return Ok(());
4144    }
4145    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
4146    match spawn_and_mark_running(spec, runtime, snapshot) {
4147        Ok(next_child) => {
4148            *child = Some(next_child);
4149            Ok(())
4150        }
4151        Err(err) => {
4152            fail_snapshot(snapshot, Some(&spec.module_id), None);
4153            process_liveness.untrack_if_current(&spec.module_id, snapshot);
4154            *child = None;
4155            Err(err)
4156        }
4157    }
4158}
4159
4160fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4161    let _ = update_snapshot(snapshot, Some(module_id), |state| {
4162        state.health.last_action = Some(action);
4163        state.health.last_action_ms = Some(now_ms);
4164    });
4165}
4166
4167fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4168    match status {
4169        HealthStatus::Ok => SupervisorHealthStatus::Ok,
4170        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4171        HealthStatus::Failing => SupervisorHealthStatus::Failing,
4172    }
4173}
4174
4175/// Caps the metrics blob stored in the cached supervisor snapshot, which is
4176/// returned to every `supervisor.list` and `supervisor.health` caller.
4177///
4178/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
4179/// path: that request exists to return a module's complete metrics object, and
4180/// `ck health <module-id>` documents it as the way to see what the cached view
4181/// truncates. The asymmetry is the feature.
4182///
4183/// So a new caller must decide which side it is on rather than assume the cap is
4184/// universal. Reaching for it on a fresh-probe path would silently reintroduce
4185/// the truncation that path exists to avoid.
4186fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4187    let metrics = metrics?;
4188    match serde_json::to_vec(&metrics) {
4189        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4190            "truncated": true,
4191            "original_bytes": encoded.len(),
4192        })),
4193        Ok(_) | Err(_) => Some(metrics),
4194    }
4195}
4196
4197/// Spread health probes so a fleet-wide restart does not converge them.
4198///
4199/// The delay is derived from the module id and probe index rather than a random
4200/// source, so it is deterministic per module: a module keeps its own offset
4201/// across daemon restarts instead of re-rolling into a collision.
4202fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4203    if cadence.is_zero() {
4204        return Duration::ZERO;
4205    }
4206    let cadence_ms = cadence.as_millis() as u64;
4207    // This early return is REDUNDANT, deliberately, and a mutation run will show
4208    // it surviving removal. Recording why here so the next person to notice does
4209    // not have to re-derive it:
4210    //
4211    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
4212    //   a zero cadence and builds the Duration from whole milliseconds, so a
4213    //   sub-millisecond cadence cannot come from config.
4214    // - Even if reached it changes no answer. The `.max(1)` below makes the span
4215    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
4216    //   -- exactly what this returns.
4217    //
4218    // Kept as a guard against a future widening of the config parser (accepting
4219    // microseconds, say), which would make the sub-millisecond case reachable.
4220    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
4221    // divides by zero. Remove this and nothing changes.
4222    if cadence_ms == 0 {
4223        return cadence;
4224    }
4225    // Note that this never returns less than one cadence, including for the FIRST
4226    // probe. So a freshly registered module reports health `unknown` for a full
4227    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
4228    // ready to answer.
4229    //
4230    // That is a property of the supervisor's schedule, not of any module: an
4231    // operator watching a restart sees `unknown` and cannot tell it from a module
4232    // that is slow to warm. Measured on two unrelated modules, both flipping to
4233    // `ok` between 22s and 32s after restart.
4234    //
4235    // Left as-is because spreading the first probe is what keeps a fleet-wide
4236    // restart from firing fourteen simultaneous probes into a cold machine. The
4237    // alternative -- probe at t+0 and jitter only from the second onward -- trades
4238    // that thundering herd for a faster first reading.
4239    let jitter_span = (cadence_ms / 10).max(1);
4240    let hash = module_id.as_bytes().iter().fold(
4241        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
4242        |acc, byte| {
4243            acc.wrapping_mul(1099511628211)
4244                .wrapping_add(u64::from(*byte))
4245        },
4246    );
4247    cadence + Duration::from_millis(hash % jitter_span)
4248}
4249
4250#[cfg(test)]
4251mod tests {
4252    use super::*;
4253
4254    #[test]
4255    fn readding_a_module_clears_its_rescan_removal_tombstone() {
4256        let handle = SupervisorHandle::new();
4257        let module_id = "readded-tombstone";
4258        handle.record_rescan_removal(module_id);
4259        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
4260
4261        handle.apply_identity_configuration(&ModuleSpec {
4262            launch_nonce_env: true,
4263            module_id: module_id.to_string(),
4264            program: PathBuf::from("/test/module"),
4265            args: Vec::new(),
4266            env: Vec::new(),
4267            reserved: false,
4268            reserved_prefixes: Vec::new(),
4269            protocol: ModuleProtocol::Subc,
4270            overlap: Default::default(),
4271        });
4272
4273        assert!(
4274            handle.removal_tombstone_age_ms(module_id).is_none(),
4275            "a re-added module must not retain a stale removal tombstone"
4276        );
4277    }
4278
4279    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
4280        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
4281        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
4282            snapshot.process_alive = true;
4283            snapshot.pid = Some(41);
4284            snapshot.spawned_at_ms = Some(42);
4285            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
4286            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
4287                device: 43,
4288                inode: 44,
4289            });
4290        })
4291        .unwrap();
4292        snapshot
4293    }
4294
4295    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
4296        let snapshot = lock_snapshot(snapshot).unwrap();
4297        assert!(!snapshot.process_alive);
4298        assert_eq!(snapshot.pid, None);
4299        assert_eq!(snapshot.spawned_at_ms, None);
4300        assert_eq!(snapshot.spawned_from, None);
4301        assert_eq!(snapshot.spawned_file_identity, None);
4302    }
4303
4304    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4305    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
4306        let supervisor = Supervisor::default();
4307        let mut runtime = supervisor.runtime_config();
4308        runtime.test_seed_stale_facts_before_enable_spawn = true;
4309        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
4310        let mut child = None;
4311        let spec = ModuleSpec {
4312            launch_nonce_env: true,
4313            module_id: "failed-enable-clears-facts".to_string(),
4314            program: PathBuf::from("/definitely/missing/failed-enable-module"),
4315            args: Vec::new(),
4316            env: Vec::new(),
4317            reserved: false,
4318            reserved_prefixes: Vec::new(),
4319            protocol: ModuleProtocol::Subc,
4320            overlap: Default::default(),
4321        };
4322
4323        let result = set_child_enabled(
4324            &spec,
4325            &runtime,
4326            &supervisor.registry,
4327            &supervisor.process_liveness,
4328            &snapshot,
4329            &mut child,
4330            true,
4331        )
4332        .await;
4333
4334        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
4335        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4336        assert_snapshot_process_facts_cleared(&snapshot);
4337    }
4338
4339    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4340    async fn failed_reload_spawn_clears_current_process_facts() {
4341        let supervisor = Supervisor::default();
4342        let mut runtime = supervisor.runtime_config();
4343        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
4344        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4345        let mut child = None;
4346        let spec = ModuleSpec {
4347            launch_nonce_env: true,
4348            module_id: "failed-reload-clears-facts".to_string(),
4349            program: PathBuf::from("/unused/failed-reload-module"),
4350            args: Vec::new(),
4351            env: Vec::new(),
4352            reserved: false,
4353            reserved_prefixes: Vec::new(),
4354            protocol: ModuleProtocol::Subc,
4355            overlap: Default::default(),
4356        };
4357
4358        let result = handle_reload_spawn_failure(
4359            &spec,
4360            &runtime,
4361            &supervisor.process_liveness,
4362            &snapshot,
4363            &mut child,
4364            "forced reload spawn failure".to_string(),
4365        )
4366        .await;
4367
4368        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
4369        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4370        assert_snapshot_process_facts_cleared(&snapshot);
4371    }
4372
4373    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4374    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
4375        let supervisor = Supervisor::default();
4376        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4377        let module = supervisor.supervised_module(
4378            ModuleSpec {
4379                launch_nonce_env: true,
4380                module_id: "drop-clears-facts".to_string(),
4381                program: PathBuf::from("/unused/drop-module"),
4382                args: Vec::new(),
4383                env: Vec::new(),
4384                reserved: false,
4385                reserved_prefixes: Vec::new(),
4386                protocol: ModuleProtocol::Subc,
4387                overlap: Default::default(),
4388            },
4389            supervisor.runtime_config(),
4390            Arc::clone(&snapshot),
4391            None,
4392        );
4393        assert!(!module
4394            .inner
4395            .monitor
4396            .lock()
4397            .unwrap()
4398            .as_ref()
4399            .unwrap()
4400            .is_finished());
4401
4402        drop(module);
4403
4404        assert_eq!(
4405            lock_snapshot(&snapshot).unwrap().state,
4406            ModuleState::Stopped
4407        );
4408        assert_snapshot_process_facts_cleared(&snapshot);
4409    }
4410
4411    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4412    async fn configuration_update_does_not_replace_captured_running_process_facts() {
4413        let supervisor = Supervisor::default();
4414        let snapshot = stale_process_snapshot(ModuleState::Running, true);
4415        let initial = ModuleSpec {
4416            launch_nonce_env: true,
4417            module_id: "rescan-preserves-spawn-facts".to_string(),
4418            program: PathBuf::from("/spawned/module"),
4419            args: Vec::new(),
4420            env: Vec::new(),
4421            reserved: false,
4422            reserved_prefixes: Vec::new(),
4423            protocol: ModuleProtocol::Subc,
4424            overlap: Default::default(),
4425        };
4426        let module = supervisor.supervised_module(
4427            initial.clone(),
4428            supervisor.runtime_config(),
4429            snapshot,
4430            None,
4431        );
4432        let before = module.status().unwrap();
4433        let mut replacement = initial;
4434        replacement.program = PathBuf::from("/rescanned/replacement-module");
4435
4436        module
4437            .update_configuration(replacement, HealthConfig::default(), None)
4438            .await
4439            .unwrap();
4440
4441        let after = module.status().unwrap();
4442        assert_eq!(after.pid, before.pid);
4443        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
4444        assert_eq!(after.spawned_from, before.spawned_from);
4445        drop(module);
4446    }
4447}
4448
4449fn unix_ms_now() -> u64 {
4450    SystemTime::now()
4451        .duration_since(UNIX_EPOCH)
4452        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
4453        .unwrap_or(0)
4454}
4455
4456async fn supervise_loop(
4457    mut spec: ModuleSpec,
4458    mut runtime: SupervisorRuntimeConfig,
4459    registry: Arc<Registry>,
4460    process_liveness: Arc<SupervisorProcessLiveness>,
4461    snapshot: SharedSnapshot,
4462    mut child: Option<SupervisedChild>,
4463    mut commands: mpsc::Receiver<SupervisorCommand>,
4464) {
4465    let mut health_probe = HealthProbeRuntime::default();
4466    // Deadline of the crash respawn whose backoff is currently elapsing. While
4467    // it is set the loop serves commands instead of sleeping inside the exit
4468    // arm, so a disable or drain lands immediately and cancels the respawn.
4469    let mut pending_respawn: Option<Instant> = None;
4470    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
4471    // before anything else so a stop that interrupted a swap runs at once.
4472    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
4473    loop {
4474        if let Some(command) = requeued.pop_front() {
4475            if !handle_supervisor_command(
4476                command,
4477                &mut spec,
4478                &mut runtime,
4479                &registry,
4480                &process_liveness,
4481                &snapshot,
4482                &mut child,
4483                &mut commands,
4484                &mut requeued,
4485            )
4486            .await
4487            {
4488                return;
4489            }
4490            if child.is_some() || !respawn_still_pending(&snapshot) {
4491                pending_respawn = None;
4492            }
4493            continue;
4494        }
4495        if child.is_some() {
4496            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
4497            let probe_sleep = sleep(health_probe.wake_after());
4498            tokio::pin!(probe_sleep);
4499            let active_child = child.as_mut().expect("child checked above");
4500            tokio::select! {
4501                wait_result = active_child.wait() => {
4502                    // Every arm below that gives up on the CHILD must keep the
4503                    // supervision task itself alive (child = None, loop
4504                    // continues into command-serving mode). Returning here
4505                    // closes the command channel, which makes the module
4506                    // permanently unrestartable in-band: a clean child exit
4507                    // of an enabled module once wedged the fleet this way
4508                    // ('supervisor command channel is closed') and required a
4509                    // full daemon restart to recover.
4510                    let exit_report = match wait_result {
4511                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
4512                        Err(err) => {
4513                            active_child.drain_stderr(&spec.module_id).await;
4514                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4515                            // Every other exit path (on_child_exit's Clean/Crash arms,
4516                            // the reload-registration-failure path) records a terminal
4517                            // before moving on. Without one here, a module whose wait()
4518                            // itself errored (e.g. already reaped) leaves no terminal
4519                            // record at all -- an empty ring reads as "nothing died".
4520                            record_wait_error_terminal(
4521                                &spec.module_id,
4522                                &runtime.terminal_ring,
4523                                &runtime.spawn_events,
4524                            );
4525                            untrack_if_registration_released(
4526                                &process_liveness,
4527                                &registry,
4528                                &spec.module_id,
4529                                &snapshot,
4530                            );
4531                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
4532                            child = None;
4533                            continue;
4534                        }
4535                    };
4536                    active_child.drain_stderr(&spec.module_id).await;
4537
4538                    let next = on_child_exit(
4539                        &spec,
4540                        runtime.restart_policy,
4541                        &registry,
4542                        &snapshot,
4543                        &runtime.terminal_ring,
4544                        &runtime.spawn_events,
4545                        &runtime.child_roster,
4546                        exit_report,
4547                    ).await;
4548                    // The exit is recorded, so a daemon shutdown may stop
4549                    // waiting for this child (see `SupervisedChild::wait`).
4550                    active_child.release_roster();
4551                    match next {
4552                        NextAction::Stop { registration_released } => {
4553                            if registration_released {
4554                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4555                            }
4556                            child = None;
4557                        }
4558                        NextAction::Restart { schedule } => {
4559                            let delay = schedule.map_or(
4560                                runtime.restart_policy.delay_for_restart(0),
4561                                |schedule| schedule.delay,
4562                            );
4563                            if let Some(schedule) = schedule {
4564                                log_crash_respawn(&spec.module_id, schedule);
4565                            }
4566                            // The exited child is fully recorded at this point,
4567                            // so release it and count the backoff down in the
4568                            // command-serving branch below rather than sleeping
4569                            // here: commands cannot be received from inside this
4570                            // select arm, and an operator disable or drain that
4571                            // arrives during the backoff must cancel the pending
4572                            // respawn instead of waiting for it to spawn first.
4573                            child = None;
4574                            pending_respawn = Some(Instant::now() + delay);
4575                        }
4576                    }
4577                }
4578                command = commands.recv() => {
4579                    let Some(command) = command else {
4580                        return;
4581                    };
4582                    if !handle_supervisor_command(
4583                        command,
4584                        &mut spec,
4585                        &mut runtime,
4586                        &registry,
4587                        &process_liveness,
4588                        &snapshot,
4589                        &mut child,
4590                        &mut commands,
4591                        &mut requeued,
4592                    ).await {
4593                        return;
4594                    }
4595                }
4596                _ = &mut probe_sleep => {
4597                    if health_probe.due() {
4598                        run_health_probe_cycle(
4599                            &spec,
4600                            &runtime,
4601                            &registry,
4602                            &process_liveness,
4603                            &snapshot,
4604                            &mut child,
4605                        ).await;
4606                        if child.is_some() {
4607                            health_probe.schedule_next(&spec, runtime.health.cadence);
4608                        }
4609                    }
4610                }
4611            }
4612        } else if let Some(deadline) = pending_respawn {
4613            tokio::select! {
4614                _ = sleep_until(deadline) => {
4615                    pending_respawn = None;
4616                    // A command handled below while the backoff elapsed may
4617                    // have stopped the module; never respawn past an operator's
4618                    // disable or drain.
4619                    if !respawn_still_pending(&snapshot) {
4620                        continue;
4621                    }
4622                    // The daemon began shutting down during the backoff: the
4623                    // spawn would be refused anyway, and refusing it here
4624                    // leaves the module stopped instead of reporting a
4625                    // failed restart.
4626                    if runtime.child_roster.is_closed() {
4627                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4628                            state.state = ModuleState::Stopped;
4629                        });
4630                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
4631                        continue;
4632                    }
4633                    if let Err(err) = wait_for_registration_release(
4634                        &registry,
4635                        &spec.module_id,
4636                        REGISTRY_RELEASE_TIMEOUT,
4637                    ).await {
4638                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
4639                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
4640                        continue;
4641                    }
4642
4643                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
4644                        Ok(next_child) => {
4645                            child = Some(next_child);
4646                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
4647                        }
4648                        Err(err) => {
4649                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
4650                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4651                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
4652                        }
4653                    }
4654                }
4655                command = commands.recv() => {
4656                    let Some(command) = command else {
4657                        return;
4658                    };
4659                    if !handle_supervisor_command(
4660                        command,
4661                        &mut spec,
4662                        &mut runtime,
4663                        &registry,
4664                        &process_liveness,
4665                        &snapshot,
4666                        &mut child,
4667                        &mut commands,
4668                        &mut requeued,
4669                    ).await {
4670                        return;
4671                    }
4672                    // Reconcile the pending respawn with what the command did:
4673                    // a restart or reload has already spawned a fresh child,
4674                    // while a disable or drain moved the snapshot out of the
4675                    // state the respawn was counting down from.
4676                    if child.is_some() || !respawn_still_pending(&snapshot) {
4677                        pending_respawn = None;
4678                    }
4679                }
4680            }
4681        } else {
4682            let Some(command) = commands.recv().await else {
4683                return;
4684            };
4685            if !handle_supervisor_command(
4686                command,
4687                &mut spec,
4688                &mut runtime,
4689                &registry,
4690                &process_liveness,
4691                &snapshot,
4692                &mut child,
4693                &mut commands,
4694                &mut requeued,
4695            )
4696            .await
4697            {
4698                return;
4699            }
4700        }
4701    }
4702}
4703
4704fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
4705    info!(
4706        module_id,
4707        restart_in_window = schedule.restart_in_window,
4708        delay_ms = schedule.delay.as_millis() as u64,
4709        "respawning after crash"
4710    );
4711}
4712
4713/// Whether the respawn a backoff was counting down to is still wanted. A
4714/// disable or drain handled while the backoff elapsed moves the snapshot out
4715/// of `Restarting`, and the operator's stop must win over the pending respawn,
4716/// so every sleep-then-spawn path re-validates against the live snapshot
4717/// instead of assuming the state it left behind still holds.
4718fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
4719    matches!(
4720        lock_snapshot(snapshot),
4721        Ok(state) if state.enabled && state.state == ModuleState::Restarting
4722    )
4723}
4724
4725enum NextAction {
4726    Stop {
4727        registration_released: bool,
4728    },
4729    Restart {
4730        schedule: Option<CrashRestartSchedule>,
4731    },
4732}
4733
4734#[allow(clippy::too_many_arguments)]
4735async fn handle_supervisor_command(
4736    command: SupervisorCommand,
4737    spec: &mut ModuleSpec,
4738    runtime: &mut SupervisorRuntimeConfig,
4739    registry: &Registry,
4740    process_liveness: &SupervisorProcessLiveness,
4741    snapshot: &SharedSnapshot,
4742    child: &mut Option<SupervisedChild>,
4743    commands: &mut mpsc::Receiver<SupervisorCommand>,
4744    requeued: &mut VecDeque<SupervisorCommand>,
4745) -> bool {
4746    match command {
4747        SupervisorCommand::Drain { reply } => {
4748            // A plain stop runs no forwarding drain, so nothing reaches the
4749            // module over its connection before the wait: ask by signal.
4750            let result = drain_optional_child(
4751                &spec.module_id,
4752                spec.protocol,
4753                StopNotice::NotSent,
4754                registry,
4755                snapshot,
4756                &runtime.terminal_ring,
4757                &runtime.spawn_events,
4758                child,
4759                runtime.drain_timeout,
4760                ModuleState::Stopped,
4761                None,
4762            )
4763            .await;
4764            let registration_released = result.is_ok();
4765            let _ = reply.send(result);
4766            if registration_released {
4767                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4768            }
4769            false
4770        }
4771        SupervisorCommand::Retire { reply } => {
4772            let result = async {
4773                let stop_notice = begin_forwarding_drain_if_configured(
4774                    spec,
4775                    runtime,
4776                    registry,
4777                    snapshot,
4778                    None,
4779                    RouteCloseReason::Disable,
4780                )
4781                .await?;
4782                drain_optional_child(
4783                    &spec.module_id,
4784                    spec.protocol,
4785                    stop_notice,
4786                    registry,
4787                    snapshot,
4788                    &runtime.terminal_ring,
4789                    &runtime.spawn_events,
4790                    child,
4791                    runtime.drain_timeout,
4792                    ModuleState::Stopped,
4793                    None,
4794                )
4795                .await
4796            }
4797            .await;
4798            let registration_released = result.is_ok();
4799            let _ = reply.send(result);
4800            if registration_released {
4801                process_liveness.untrack_if_current(&spec.module_id, snapshot);
4802            }
4803            false
4804        }
4805        SupervisorCommand::Restart {
4806            drain_timeout_ms,
4807            received_at_generation,
4808            queued_at,
4809            reply,
4810        } => {
4811            // Without this line a restart that waited in the queue (behind a
4812            // health probe cycle or another command) was invisible: the log
4813            // showed only the drain timing out, minutes after the operator's call.
4814            info!(
4815                module_id = %spec.module_id,
4816                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
4817                "restart command dequeued"
4818            );
4819            // ACK AT INITIATION, not completion. The blocking form deadlocked any
4820            // caller whose own request lane rides the module being restarted: the
4821            // caller's in-flight request keeps the drain from quiescing, the drain
4822            // keeps the restart from completing, and the completion keeps the reply
4823            // from releasing the caller — so the drain always timed out and cut the
4824            // initiator with a GOODBYE, even on a healthy module. Replying once the
4825            // restart is validated lets a self-lane caller settle, which is exactly
4826            // what makes the drain succeed. Completion is observable via
4827            // supervisor.list / module status; a post-ack failure lands the module
4828            // in a visible terminal state below rather than in a reply nobody can
4829            // receive.
4830            let validation = match lock_snapshot(snapshot) {
4831                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
4832                    module_id: spec.module_id.clone(),
4833                }),
4834                Ok(_) => Ok(()),
4835                Err(err) => Err(err),
4836            };
4837            let initiated = validation.is_ok();
4838            let _ = reply.send(validation);
4839            // A restart asks for a fresh process. Commands run one at a time,
4840            // so a restart queued behind another restart (two operator calls
4841            // in quick succession) is dequeued the moment the first one has
4842            // spawned its replacement -- before that process has sent HELLO.
4843            // Running it would drain and kill the process the first restart
4844            // just produced, which is the opposite of what both callers asked
4845            // for. If a process spawned after this request was received is
4846            // still supervised, the request is already satisfied. Not when the
4847            // configuration changed since that spawn: then the newer process
4848            // predates the spec this restart may exist to apply.
4849            let satisfied_by_generation = if initiated && child.is_some() {
4850                lock_snapshot(snapshot).ok().and_then(|state| {
4851                    (state.spawn_generation > received_at_generation
4852                        && !state.configuration_updated_since_spawn)
4853                        .then_some(state.spawn_generation)
4854                })
4855            } else {
4856                None
4857            };
4858            if let Some(generation) = satisfied_by_generation {
4859                info!(
4860                    module_id = %spec.module_id,
4861                    received_at_generation,
4862                    "restart already satisfied by generation {generation}; not restarting again"
4863                );
4864            } else if initiated {
4865                // Precedence: this restart's operator override, else the module's
4866                // configured budget (already resolved into the runtime).
4867                let drain_timeout = drain_timeout_ms
4868                    .map(Duration::from_millis)
4869                    .unwrap_or(runtime.drain_timeout);
4870                if let Err(err) = restart_child(
4871                    spec,
4872                    runtime,
4873                    registry,
4874                    process_liveness,
4875                    snapshot,
4876                    child,
4877                    drain_timeout,
4878                )
4879                .await
4880                {
4881                    warn!(
4882                        module_id = %spec.module_id,
4883                        error = %err,
4884                        "operator restart failed after initiation ack; module state carries the outcome"
4885                    );
4886                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4887                        state.state = ModuleState::Failed;
4888                        clear_current_process_facts(state);
4889                    });
4890                }
4891            }
4892            true
4893        }
4894        SupervisorCommand::Reload { reply } => {
4895            let result =
4896                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
4897            let _ = reply.send(result);
4898            true
4899        }
4900        SupervisorCommand::SetEnabled { enabled, reply } => {
4901            let result = set_child_enabled(
4902                spec,
4903                runtime,
4904                registry,
4905                process_liveness,
4906                snapshot,
4907                child,
4908                enabled,
4909            )
4910            .await;
4911            let _ = reply.send(result);
4912            true
4913        }
4914        SupervisorCommand::UpdateConfiguration {
4915            spec: next_spec,
4916            health,
4917            drain_timeout_ms,
4918            reply,
4919        } => {
4920            if let Some(handle) = &runtime.supervisor_handle {
4921                handle.apply_identity_configuration(&next_spec);
4922            }
4923            *spec = next_spec;
4924            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4925                state.configuration_updated_since_spawn = true;
4926            });
4927            runtime.health = health;
4928            runtime.drain_timeout = drain_timeout_ms
4929                .map(Duration::from_millis)
4930                .unwrap_or(runtime.default_drain_timeout);
4931            *runtime
4932                .effective_drain_timeout
4933                .lock()
4934                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
4935            let _ = reply.send(());
4936            true
4937        }
4938        SupervisorCommand::Swap {
4939            ready_timeout,
4940            reply,
4941        } => {
4942            let end = swap::run_swap(
4943                spec,
4944                runtime,
4945                registry,
4946                process_liveness,
4947                snapshot,
4948                child,
4949                commands,
4950                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
4951                reply,
4952            )
4953            .await;
4954            requeued.extend(end.requeue);
4955            true
4956        }
4957    }
4958}
4959
4960async fn restart_child(
4961    spec: &ModuleSpec,
4962    runtime: &SupervisorRuntimeConfig,
4963    registry: &Registry,
4964    process_liveness: &SupervisorProcessLiveness,
4965    snapshot: &SharedSnapshot,
4966    child: &mut Option<SupervisedChild>,
4967    drain_timeout: Duration,
4968) -> Result<(), SuperviseError> {
4969    // Restart cycles a running module; it must not silently start a disabled one.
4970    if !lock_snapshot(snapshot)?.enabled {
4971        return Err(SuperviseError::Disabled {
4972            module_id: spec.module_id.clone(),
4973        });
4974    }
4975    let stop_notice = begin_forwarding_drain_with_timeout(
4976        spec,
4977        runtime,
4978        registry,
4979        snapshot,
4980        None,
4981        RouteCloseReason::Restart,
4982        drain_timeout,
4983    )
4984    .await?;
4985
4986    if child.is_some() {
4987        drain_optional_child(
4988            &spec.module_id,
4989            spec.protocol,
4990            stop_notice,
4991            registry,
4992            snapshot,
4993            &runtime.terminal_ring,
4994            &runtime.spawn_events,
4995            child,
4996            drain_timeout,
4997            ModuleState::Restarting,
4998            Some(true),
4999        )
5000        .await?;
5001    } else {
5002        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5003            state.enabled = true;
5004            state.state = ModuleState::Restarting;
5005            clear_current_process_facts(state);
5006        })?;
5007        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5008    }
5009
5010    reset_restart_count(snapshot, &spec.module_id)?;
5011    sleep(runtime.restart_policy.backoff).await;
5012    // A disable or drain that landed during the backoff cancels this respawn:
5013    // the operator's stop must win over the restart the sleep counted down to.
5014    if !respawn_still_pending(snapshot) {
5015        process_liveness.untrack_if_current(&spec.module_id, snapshot);
5016        return Ok(());
5017    }
5018    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5019    // Mirror health_restart_child's spawn-failure handling: of the four
5020    // spawn-failure sites this was the only one that propagated with the
5021    // snapshot still reading `Restarting` -- neither running nor failed, and
5022    // unrevivable by `set_enabled(true)` (issue #34). `Failed` is the state the
5023    // operator can see and heal.
5024    match spawn_and_mark_running(spec, runtime, snapshot) {
5025        Ok(next_child) => {
5026            *child = Some(next_child);
5027            debug!(module_id = %spec.module_id, "supervised module restarted by operator request");
5028            Ok(())
5029        }
5030        Err(err) => {
5031            fail_snapshot(snapshot, Some(&spec.module_id), None);
5032            process_liveness.untrack_if_current(&spec.module_id, snapshot);
5033            *child = None;
5034            Err(err)
5035        }
5036    }
5037}
5038
5039async fn reload_child(
5040    spec: &ModuleSpec,
5041    runtime: &SupervisorRuntimeConfig,
5042    registry: &Registry,
5043    process_liveness: &SupervisorProcessLiveness,
5044    snapshot: &SharedSnapshot,
5045    child: &mut Option<SupervisedChild>,
5046) -> Result<(), SuperviseError> {
5047    // Reload cycles a running module; it must not silently start a disabled one.
5048    if !lock_snapshot(snapshot)?.enabled {
5049        return Err(SuperviseError::Disabled {
5050            module_id: spec.module_id.clone(),
5051        });
5052    }
5053    let stop_notice = begin_forwarding_drain(
5054        spec,
5055        runtime,
5056        registry,
5057        snapshot,
5058        Some(true),
5059        RouteCloseReason::Reload,
5060    )
5061    .await?;
5062
5063    if child.is_some() {
5064        drain_optional_child(
5065            &spec.module_id,
5066            spec.protocol,
5067            stop_notice,
5068            registry,
5069            snapshot,
5070            &runtime.terminal_ring,
5071            &runtime.spawn_events,
5072            child,
5073            runtime.drain_timeout,
5074            ModuleState::Restarting,
5075            Some(true),
5076        )
5077        .await?;
5078    } else {
5079        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5080            state.enabled = true;
5081            state.state = ModuleState::Restarting;
5082            clear_current_process_facts(state);
5083        })?;
5084        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5085    }
5086
5087    reset_restart_count(snapshot, &spec.module_id)?;
5088    sleep(runtime.restart_policy.backoff).await;
5089    // A disable or drain that landed during the backoff cancels this respawn:
5090    // the operator's stop must win over the restart the sleep counted down to.
5091    if !respawn_still_pending(snapshot) {
5092        process_liveness.untrack_if_current(&spec.module_id, snapshot);
5093        return Ok(());
5094    }
5095    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5096    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5097        Ok(next_child) => next_child,
5098        Err(err) => {
5099            return handle_reload_spawn_failure(
5100                spec,
5101                runtime,
5102                process_liveness,
5103                snapshot,
5104                child,
5105                format!("new child failed to spawn: {err}"),
5106            )
5107            .await;
5108        }
5109    };
5110    *child = Some(next_child);
5111
5112    let wait_outcome = {
5113        let active_child = child.as_mut().expect("new reload child was just stored");
5114        wait_for_registration_after_reload(
5115            registry,
5116            &spec.module_id,
5117            snapshot,
5118            active_child,
5119            REGISTRY_RELEASE_TIMEOUT,
5120        )
5121        .await?
5122    };
5123
5124    match wait_outcome {
5125        RegistrationWaitOutcome::Registered => {
5126            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
5127            Ok(())
5128        }
5129        RegistrationWaitOutcome::Exited(exit_report) => {
5130            if let Some(active_child) = child.as_mut() {
5131                active_child.drain_stderr(&spec.module_id).await;
5132            }
5133            *child = None;
5134            handle_reload_child_registration_failure(
5135                spec,
5136                runtime,
5137                registry,
5138                process_liveness,
5139                snapshot,
5140                child,
5141                ReloadRegistrationFailure {
5142                    exit_report: registration_failure_exit_report(exit_report),
5143                    reason: "new child exited before registering".to_string(),
5144                },
5145            )
5146            .await
5147        }
5148        RegistrationWaitOutcome::TimedOut => {
5149            let mut timed_out_child = child
5150                .take()
5151                .expect("timed-out reload child is still running");
5152            timed_out_child
5153                .start_kill()
5154                .map_err(|source| SuperviseError::Kill {
5155                    module_id: spec.module_id.clone(),
5156                    source,
5157                })?;
5158            let status = timed_out_child
5159                .wait()
5160                .await
5161                .map_err(|source| SuperviseError::Wait {
5162                    module_id: spec.module_id.clone(),
5163                    source,
5164                })?;
5165            timed_out_child.drain_stderr(&spec.module_id).await;
5166            handle_reload_child_registration_failure(
5167                spec,
5168                runtime,
5169                registry,
5170                process_liveness,
5171                snapshot,
5172                child,
5173                ReloadRegistrationFailure {
5174                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
5175                        snapshot,
5176                        &timed_out_child,
5177                        &status,
5178                    )),
5179                    reason: format!(
5180                        "new child did not register within {:?}",
5181                        REGISTRY_RELEASE_TIMEOUT
5182                    ),
5183                },
5184            )
5185            .await
5186        }
5187    }
5188}
5189
5190async fn set_child_enabled(
5191    spec: &ModuleSpec,
5192    runtime: &SupervisorRuntimeConfig,
5193    registry: &Registry,
5194    process_liveness: &SupervisorProcessLiveness,
5195    snapshot: &SharedSnapshot,
5196    child: &mut Option<SupervisedChild>,
5197    enabled: bool,
5198) -> Result<bool, SuperviseError> {
5199    let (current_enabled, current_state) = {
5200        let state = lock_snapshot(snapshot)?;
5201        (state.enabled, state.state)
5202    };
5203    // `start` (enable on an already-enabled module) heals TERMINAL states instead
5204    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
5205    // clean (Stopped) has no live process and no other in-band recovery — the
5206    // operator's start is the explicit recovery act and resets the budget. Without
5207    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
5208    // which the 2026-07-14 aft outage proved is a trap when the failed module is
5209    // the one providing every agent's shell.
5210    let revive_terminal = enabled
5211        && current_enabled
5212        && child.is_none()
5213        && matches!(current_state, ModuleState::Failed | ModuleState::Stopped);
5214    if current_enabled == enabled && !revive_terminal {
5215        return Ok(false);
5216    }
5217
5218    if enabled {
5219        update_snapshot(snapshot, Some(&spec.module_id), |state| {
5220            state.enabled = true;
5221            state.state = ModuleState::Starting;
5222            clear_current_process_facts(state);
5223        })?;
5224        #[cfg(test)]
5225        if runtime.test_seed_stale_facts_before_enable_spawn {
5226            update_snapshot(snapshot, Some(&spec.module_id), |state| {
5227                state.process_alive = true;
5228                state.pid = Some(41);
5229                state.spawned_at_ms = Some(42);
5230                state.spawned_from = Some(PathBuf::from("/spawned/module"));
5231                state.spawned_file_identity = Some(SpawnedFileIdentity {
5232                    device: 43,
5233                    inode: 44,
5234                });
5235            })?;
5236        }
5237        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5238        reset_restart_count(snapshot, &spec.module_id)?;
5239        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5240        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5241            Ok(next_child) => next_child,
5242            Err(err) => {
5243                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5244                    state.state = ModuleState::Failed;
5245                    clear_current_process_facts(state);
5246                }) {
5247                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
5248                }
5249                process_liveness.untrack_if_current(&spec.module_id, snapshot);
5250                return Err(err);
5251            }
5252        };
5253        *child = Some(next_child);
5254        debug!(module_id = %spec.module_id, "supervised module enabled");
5255        Ok(true)
5256    } else {
5257        let stop_notice = begin_forwarding_drain_if_configured(
5258            spec,
5259            runtime,
5260            registry,
5261            snapshot,
5262            Some(false),
5263            RouteCloseReason::Disable,
5264        )
5265        .await?;
5266        drain_optional_child(
5267            &spec.module_id,
5268            spec.protocol,
5269            stop_notice,
5270            registry,
5271            snapshot,
5272            &runtime.terminal_ring,
5273            &runtime.spawn_events,
5274            child,
5275            runtime.drain_timeout,
5276            ModuleState::Disabled,
5277            Some(false),
5278        )
5279        .await?;
5280        debug!(module_id = %spec.module_id, "supervised module disabled");
5281        Ok(true)
5282    }
5283}
5284
5285#[allow(clippy::too_many_arguments)]
5286async fn on_child_exit(
5287    spec: &ModuleSpec,
5288    policy: RestartPolicy,
5289    registry: &Registry,
5290    snapshot: &SharedSnapshot,
5291    terminal_ring: &Arc<Mutex<TerminalRing>>,
5292    spawn_events: &SpawnEventFeed,
5293    roster: &ChildRoster,
5294    exit_report: ExitReport,
5295) -> NextAction {
5296    // Once the daemon has begun shutting down, no exit is a crash to recover
5297    // from: the module is exiting because the daemon is going away (EOF on its
5298    // connection, or a service manager signalling the whole cgroup). Record it
5299    // as such and never schedule a respawn, which would only start a process
5300    // for the shutdown to end again.
5301    if roster.is_closed() {
5302        return on_child_exit_during_daemon_shutdown(
5303            spec,
5304            registry,
5305            snapshot,
5306            terminal_ring,
5307            spawn_events,
5308            exit_report,
5309        )
5310        .await;
5311    }
5312    // Every stop the supervisor itself asks for (operator stop, disable,
5313    // restart, reload, swap, a health restart, a drain that runs out of budget)
5314    // takes the child out of the supervise loop and reaps it in
5315    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
5316    // that reaches this point was not requested by the daemon.
5317    //
5318    // For a subc-wire module a clean exit is still a stop: those modules are
5319    // written to re-raise SIGTERM, so a stray outside signal already reads as a
5320    // crash, and exiting 0 is a deliberate choice the module made. A
5321    // `protocol: "none"` module is a stock program we cannot change, and many
5322    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
5323    // stop would leave the module down for good after any stray signal, so it
5324    // goes through the crash path instead: it spends restart budget, respawns
5325    // with the crash backoff, and ends `failed` when the budget runs out.
5326    let unrequested_clean_exit_of_protocol_none =
5327        exit_report.kind == ExitKind::Clean && spec.protocol == ModuleProtocol::None;
5328    match exit_report.kind {
5329        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
5330            info!(
5331                module_id = %spec.module_id,
5332                exit_code = ?exit_report.code,
5333                exit_signal = ?exit_report.signal,
5334                "supervised module exited cleanly"
5335            );
5336            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5337                state.state = ModuleState::Stopped;
5338                clear_current_process_facts(state);
5339                state.last_exit = Some(exit_report.clone());
5340            }) {
5341                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
5342            }
5343            record_terminal(
5344                &spec.module_id,
5345                terminal_ring,
5346                spawn_events,
5347                &exit_report,
5348                TerminalDisposition::Stopped,
5349            );
5350            let registration_released = match wait_for_registration_release(
5351                registry,
5352                &spec.module_id,
5353                REGISTRY_RELEASE_TIMEOUT,
5354            )
5355            .await
5356            {
5357                Ok(()) => true,
5358                Err(err) => {
5359                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
5360                    false
5361                }
5362            };
5363            NextAction::Stop {
5364                registration_released,
5365            }
5366        }
5367        ExitKind::Clean | ExitKind::Crash => {
5368            if unrequested_clean_exit_of_protocol_none {
5369                warn!(
5370                    module_id = %spec.module_id,
5371                    exit_code = ?exit_report.code,
5372                    exit_signal = ?exit_report.signal,
5373                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
5374                );
5375            } else {
5376                warn!(
5377                    module_id = %spec.module_id,
5378                    exit_code = ?exit_report.code,
5379                    exit_signal = ?exit_report.signal,
5380                    "supervised module exited abnormally (crash)"
5381                );
5382            }
5383            let mut restart_schedule = None;
5384            let mut disposition = TerminalDisposition::Disabled;
5385            // Set only when the budget is what stopped the module, so the
5386            // terminal record says which limit was hit rather than leaving
5387            // `failed` to be read as "crashed once, badly".
5388            let mut disposition_detail = None;
5389            let now = Instant::now();
5390            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5391                clear_current_process_facts(state);
5392                state.last_exit = Some(exit_report.clone());
5393                if state.enabled {
5394                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
5395                        state.state = ModuleState::Restarting;
5396                        restart_schedule = Some(schedule);
5397                        disposition = TerminalDisposition::Restarting;
5398                    } else {
5399                        state.state = ModuleState::Failed;
5400                        disposition = TerminalDisposition::Failed;
5401                        disposition_detail = Some(policy.budget_exhausted_detail());
5402                    }
5403                } else {
5404                    state.state = ModuleState::Disabled;
5405                    disposition = TerminalDisposition::Disabled;
5406                }
5407            }) {
5408                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
5409                return NextAction::Stop {
5410                    registration_released: false,
5411                };
5412            }
5413            if disposition_detail.is_some() {
5414                // The window is in the message, not only in the fields: this line
5415                // is read in a scrollback where a bare `max_restarts=3` reads as a
5416                // lifetime cap and sends the operator looking for three crashes
5417                // that never happened together.
5418                error!(
5419                    module_id = %spec.module_id,
5420                    max_restarts = policy.max_restarts,
5421                    window_secs = policy.window.as_secs(),
5422                    "module stopped: {}",
5423                    policy.budget_exhausted_detail()
5424                );
5425            }
5426            record_terminal_with_detail(
5427                &spec.module_id,
5428                terminal_ring,
5429                spawn_events,
5430                &exit_report,
5431                disposition,
5432                disposition_detail,
5433            );
5434
5435            if let Some(schedule) = restart_schedule {
5436                NextAction::Restart {
5437                    schedule: Some(schedule),
5438                }
5439            } else {
5440                let registration_released = match wait_for_registration_release(
5441                    registry,
5442                    &spec.module_id,
5443                    REGISTRY_RELEASE_TIMEOUT,
5444                )
5445                .await
5446                {
5447                    Ok(()) => true,
5448                    Err(err) => {
5449                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
5450                        false
5451                    }
5452                };
5453                NextAction::Stop {
5454                    registration_released,
5455                }
5456            }
5457        }
5458        ExitKind::DeliberateSeverance => {
5459            warn!(
5460                module_id = %spec.module_id,
5461                exit_code = ?exit_report.code,
5462                exit_signal = ?exit_report.signal,
5463                "supervised module exited after deliberate connection severance"
5464            );
5465            let mut should_restart = false;
5466            let mut disposition = TerminalDisposition::Disabled;
5467            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5468                clear_current_process_facts(state);
5469                state.last_exit = Some(exit_report.clone());
5470                state.lifetime_restarts += 1;
5471                if state.enabled {
5472                    state.state = ModuleState::Restarting;
5473                    should_restart = true;
5474                    disposition = TerminalDisposition::Restarting;
5475                } else {
5476                    state.state = ModuleState::Disabled;
5477                }
5478            }) {
5479                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
5480                return NextAction::Stop {
5481                    registration_released: false,
5482                };
5483            }
5484            record_terminal(
5485                &spec.module_id,
5486                terminal_ring,
5487                spawn_events,
5488                &exit_report,
5489                disposition,
5490            );
5491
5492            if should_restart {
5493                NextAction::Restart { schedule: None }
5494            } else {
5495                let registration_released = match wait_for_registration_release(
5496                    registry,
5497                    &spec.module_id,
5498                    REGISTRY_RELEASE_TIMEOUT,
5499                )
5500                .await
5501                {
5502                    Ok(()) => true,
5503                    Err(err) => {
5504                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
5505                        false
5506                    }
5507                };
5508                NextAction::Stop {
5509                    registration_released,
5510                }
5511            }
5512        }
5513    }
5514}
5515
5516async fn on_child_exit_during_daemon_shutdown(
5517    spec: &ModuleSpec,
5518    registry: &Registry,
5519    snapshot: &SharedSnapshot,
5520    terminal_ring: &Arc<Mutex<TerminalRing>>,
5521    spawn_events: &SpawnEventFeed,
5522    exit_report: ExitReport,
5523) -> NextAction {
5524    info!(
5525        module_id = %spec.module_id,
5526        exit_code = ?exit_report.code,
5527        exit_signal = ?exit_report.signal,
5528        exit_kind = ?exit_report.kind,
5529        "supervised module exited during daemon shutdown; not restarting it"
5530    );
5531    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5532        state.state = ModuleState::Stopped;
5533        clear_current_process_facts(state);
5534        state.last_exit = Some(exit_report.clone());
5535    }) {
5536        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
5537    }
5538    record_terminal(
5539        &spec.module_id,
5540        terminal_ring,
5541        spawn_events,
5542        &exit_report,
5543        TerminalDisposition::DaemonShutdown,
5544    );
5545    let registration_released =
5546        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
5547            .await
5548            .is_ok();
5549    NextAction::Stop {
5550        registration_released,
5551    }
5552}
5553
5554fn record_wait_error_terminal(
5555    module_id: &str,
5556    terminal_ring: &Arc<Mutex<TerminalRing>>,
5557    spawn_events: &SpawnEventFeed,
5558) {
5559    record_terminal(
5560        module_id,
5561        terminal_ring,
5562        spawn_events,
5563        &wait_error_exit_report(),
5564        TerminalDisposition::Failed,
5565    );
5566}
5567
5568fn record_terminal(
5569    module_id: &str,
5570    terminal_ring: &Arc<Mutex<TerminalRing>>,
5571    spawn_events: &SpawnEventFeed,
5572    exit_report: &ExitReport,
5573    disposition: TerminalDisposition,
5574) {
5575    record_terminal_with_detail(
5576        module_id,
5577        terminal_ring,
5578        spawn_events,
5579        exit_report,
5580        disposition,
5581        None,
5582    );
5583}
5584
5585/// The ring lock is held only to capture the read (see
5586/// `TerminalJournal::capture_read`), so this module's exits keep recording
5587/// while the journal files are read. Blocking: it reads files.
5588fn durable_terminal_history_of(
5589    terminal_ring: &Mutex<TerminalRing>,
5590    module_id: &str,
5591) -> subc_control::TerminalHistory {
5592    let read = terminal_ring
5593        .lock()
5594        .unwrap_or_else(|p| p.into_inner())
5595        .capture_durable_history();
5596    read.read(module_id)
5597}
5598
5599fn record_terminal_with_detail(
5600    module_id: &str,
5601    terminal_ring: &Arc<Mutex<TerminalRing>>,
5602    spawn_events: &SpawnEventFeed,
5603    exit_report: &ExitReport,
5604    disposition: TerminalDisposition,
5605    disposition_detail: Option<String>,
5606) {
5607    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
5608    let record = TerminalRecord {
5609        exit_code: exit_report.code,
5610        exit_signal: exit_report.signal,
5611        at_ms: exit_report.at_ms,
5612        disposition,
5613        exit_kind: exit_report.kind.into(),
5614        disposition_detail,
5615    };
5616    terminal_ring
5617        .lock()
5618        .unwrap_or_else(|poisoned| poisoned.into_inner())
5619        .record_exit(module_id, record);
5620}
5621
5622fn untrack_if_registration_released(
5623    process_liveness: &SupervisorProcessLiveness,
5624    registry: &Registry,
5625    module_id: &str,
5626    snapshot: &SharedSnapshot,
5627) {
5628    match registry.get_module(module_id) {
5629        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
5630        Ok(Some(_)) => {}
5631        Err(err) => {
5632            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
5633        }
5634    }
5635}
5636
5637/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
5638/// then apply the module's configured entries minus daemon-private capture keys.
5639///
5640/// Separated from `spawn_child` only so it can be asserted without spawning a
5641/// process — a duplicate of this logic in a test would pass while the real one
5642/// drifted, which is the defect class this function exists to avoid.
5643/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
5644/// nonce. A `protocol: "none"` module gets neither, because it cannot use
5645/// either and the argument would stop a stock binary from starting at all.
5646/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
5647///
5648/// The plain-spawn form, kept for the tests that assert its plan; spawns go
5649/// through [`apply_wire_spawn_args_for_role`].
5650#[cfg(test)]
5651fn apply_wire_spawn_args(
5652    command: &mut Command,
5653    spec: &ModuleSpec,
5654    connection_file_path: Option<&std::path::Path>,
5655    handle: Option<&SupervisorHandle>,
5656) -> Result<Option<NonceHandoff>, SuperviseError> {
5657    apply_wire_spawn_args_for_role(
5658        command,
5659        spec,
5660        connection_file_path,
5661        handle,
5662        SpawnRole::Plain,
5663    )
5664}
5665
5666/// The read end of a spawn's launch-nonce pipe, prepared by
5667/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
5668/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
5669/// handoff and keeps only the environment copy.
5670#[cfg(unix)]
5671type NonceHandoff = subc_os::LaunchNonceHandoff;
5672#[cfg(not(unix))]
5673type NonceHandoff = std::convert::Infallible;
5674
5675/// [`apply_wire_spawn_args`] for either slot.
5676///
5677/// A plain spawn's nonce replaces the module's recorded spawn (and reserved)
5678/// nonce, as every respawn always has. A swap candidate's nonce must leave
5679/// those alone, because the incumbent is still serving and its consumers still
5680/// attest with its nonce; it is recorded as the open swap's candidate token
5681/// instead, and the recording happens before the process exists so its HELLO
5682/// can never arrive ahead of it.
5683///
5684/// The nonce goes to the child two ways. On macOS and Linux it is written into
5685/// a pipe whose read end the child gets as descriptor 3, named by
5686/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
5687/// process of the same user cannot read it with `ps eww`. That handoff is
5688/// returned rather than installed here, because installing it replaces
5689/// whatever the child has at descriptor 3 and so must be the last pre-exec
5690/// step, after the Linux cgroup placement that the caller registers later.
5691/// The environment copy `SUBC_LAUNCH_NONCE` is also set by default for older
5692/// readers. A module can withhold it on Unix with `launch_nonce_env: false`.
5693fn apply_wire_spawn_args_for_role(
5694    command: &mut Command,
5695    spec: &ModuleSpec,
5696    connection_file_path: Option<&std::path::Path>,
5697    handle: Option<&SupervisorHandle>,
5698    role: SpawnRole,
5699) -> Result<Option<NonceHandoff>, SuperviseError> {
5700    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
5701    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
5702    // included: a daemon started from a module's process tree inherits it,
5703    // and passing it on would point the child at a descriptor it does not
5704    // have.
5705    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
5706    // Remove inherited or configured copies too: withholding must mean absent.
5707    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
5708    if spec.protocol == ModuleProtocol::None {
5709        return Ok(None);
5710    }
5711    if let Some(connection_file_path) = connection_file_path {
5712        command.arg(SUBC_ARG).arg(connection_file_path);
5713    }
5714
5715    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
5716    // route.open attestation. Reserved modules additionally use the same nonce
5717    // for HELLO id-squatting protection. A respawn rotates both records.
5718    let nonce = generate_launch_nonce()?;
5719    if let Some(handle) = handle {
5720        match role {
5721            SpawnRole::Plain => {
5722                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
5723                if spec.reserved {
5724                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
5725                }
5726            }
5727            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
5728        }
5729    }
5730    #[cfg(unix)]
5731    let handoff = {
5732        let handoff =
5733            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
5734                program: spec.program.clone(),
5735                source,
5736                cgroup_path: None,
5737            })?;
5738        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
5739        Some(handoff)
5740    };
5741    #[cfg(not(unix))]
5742    let handoff = None;
5743    if !cfg!(unix) || spec.launch_nonce_env {
5744        command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
5745    }
5746    Ok(handoff)
5747}
5748
5749fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
5750    command.env_remove(CK_LOG_ENV);
5751    // The spawn role is the supervisor's to set, and only on a swap candidate
5752    // (see `apply_spawn_role`). Removing it here, rather than just not setting
5753    // it, is what makes it absent on a plain spawn: the daemon's own
5754    // environment could carry it, and so could a spec built outside daemon
5755    // config (config refuses it as an `env` key). A module reading it on a
5756    // plain restart would pick the long swap budget and leave callers waiting.
5757    command.env_remove(SUBC_SPAWN_ROLE_ENV);
5758    for (key, value) in &spec.env {
5759        // cortexkit-log currently exposes retention only as a Rust struct, not
5760        // environment names. These values are daemon-private sink metadata and
5761        // must never become a public child-process contract by being inherited.
5762        if matches!(
5763            key.as_str(),
5764            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
5765        ) || key == SUBC_SPAWN_ROLE_ENV
5766        {
5767            continue;
5768        }
5769        command.env(key, value);
5770    }
5771}
5772
5773/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
5774/// of a blue/green swap.
5775#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5776enum SpawnRole {
5777    Plain,
5778    SwapCandidate,
5779}
5780
5781/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
5782/// `apply_child_env` has already removed the variable for every spawn.
5783fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
5784    if role == SpawnRole::SwapCandidate {
5785        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
5786    }
5787}
5788
5789fn spawn_child(
5790    spec: &ModuleSpec,
5791    connection_file_path: Option<&std::path::Path>,
5792    handle: Option<&SupervisorHandle>,
5793    ring: &Arc<Mutex<StderrRing>>,
5794    capture_logs_dir: Option<&std::path::Path>,
5795    roster: &ChildRoster,
5796    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5797) -> Result<SupervisedChild, SuperviseError> {
5798    spawn_child_in_slot(
5799        spec,
5800        connection_file_path,
5801        handle,
5802        ring,
5803        capture_logs_dir,
5804        roster,
5805        #[cfg(target_os = "linux")]
5806        cgroup_placement,
5807        SpawnRole::Plain,
5808        false,
5809    )
5810}
5811
5812/// Spawn one process of `spec` into a slot.
5813///
5814/// `alternate_slot` picks the process's cgroup name (see `swap::cgroup_name`).
5815/// A swap candidate needs a different cgroup from the process it is replacing,
5816/// which is still alive: in the same cgroup the two would be one kill domain,
5817/// and killing a failed candidate could take the incumbent with it.
5818///
5819/// The stderr capture file is `<module_id>.stderr.log` for every process of
5820/// the module, whichever slot it is in, because that is the one file
5821/// `ck module logs` reads. During a swap's overlap both processes append to it;
5822/// the daemon writes whole lines, so the two interleave by line, which is also
5823/// the merged view an operator wants while a swap runs.
5824#[allow(clippy::too_many_arguments)]
5825fn spawn_child_in_slot(
5826    spec: &ModuleSpec,
5827    connection_file_path: Option<&std::path::Path>,
5828    handle: Option<&SupervisorHandle>,
5829    ring: &Arc<Mutex<StderrRing>>,
5830    capture_logs_dir: Option<&std::path::Path>,
5831    roster: &ChildRoster,
5832    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5833    role: SpawnRole,
5834    alternate_slot: bool,
5835) -> Result<SupervisedChild, SuperviseError> {
5836    if roster.is_closed() {
5837        return Err(SuperviseError::Spawn {
5838            program: spec.program.clone(),
5839            source: io::Error::other("the daemon is shutting down; not starting a new process"),
5840            cgroup_path: None,
5841        });
5842    }
5843    #[cfg(target_os = "linux")]
5844    let cgroup_name = swap::cgroup_name(&spec.module_id, alternate_slot);
5845    #[cfg(not(target_os = "linux"))]
5846    let _ = alternate_slot;
5847    let mut command = Command::new(&spec.program);
5848    command.args(&spec.args);
5849    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
5850    // that is the whole of the intent, so remove that one key rather than the
5851    // environment.
5852    //
5853    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
5854    // and took the POSIX environment with it. Modules spawned that way had no
5855    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
5856    // logging:
5857    //
5858    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
5859    //     both unset it fell back to the temp dir alone and `ck` could not find
5860    //     a daemon running on the same machine from inside any module's process
5861    //     tree — reporting a path the file has never lived at, which reads as
5862    //     "the daemon did not write its file".
5863    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
5864    //     the RELATIVE `.local/share`, so a module deriving its own store path
5865    //     resolved it against its own CWD. That is the store-fragmentation
5866    //     defect the daemon already refuses in config (`parse_doc` rejects a
5867    //     relative `storage.data_home`) arriving by derivation instead.
5868    //   * anything a module spawns inherited it: git without ~/.gitconfig,
5869    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
5870    //     quietly rather than erroring.
5871    //
5872    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
5873    // offered one candidate under /tmp while the file sat in /run/user/1000.
5874    //
5875    // A configured module is unaffected either way: `module_spec()` puts the
5876    // resolved CK_LOG into `spec.env`, which is applied below and therefore
5877    // wins over anything ambient.
5878    apply_child_env(&mut command, spec);
5879    apply_spawn_role(&mut command, role);
5880    let nonce_handoff =
5881        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
5882
5883    #[cfg(target_os = "linux")]
5884    let cgroup_path = cgroup_placement
5885        .map(|placement| placement.module_path(&cgroup_name))
5886        .transpose()
5887        .map_err(|source| SuperviseError::Cgroup {
5888            module_id: spec.module_id.clone(),
5889            source,
5890        })?;
5891    #[cfg(not(target_os = "linux"))]
5892    let cgroup_path: Option<PathBuf> = None;
5893    #[cfg(target_os = "linux")]
5894    if let Some(path) = &cgroup_path {
5895        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
5896            if let Some(placement) = cgroup_placement {
5897                remove_module_cgroup(placement, &cgroup_name);
5898            }
5899            return Err(error);
5900        }
5901    }
5902
5903    let output_sink = if let Some(logs_dir) = capture_logs_dir {
5904        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
5905        match ChildOutputSink::open(&path, capture_retention(spec)) {
5906            Ok(sink) => sink,
5907            Err(error) => {
5908                warn!(
5909                    module_id = %spec.module_id,
5910                    path = %path.display(),
5911                    error = %error,
5912                    "could not open child output capture file; forwarding to stderr"
5913                );
5914                ChildOutputSink::Stderr
5915            }
5916        }
5917    } else {
5918        ChildOutputSink::Stderr
5919    };
5920
5921    command.stdout(Stdio::piped());
5922    command.stderr(Stdio::piped());
5923    command.kill_on_drop(true);
5924    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
5925    // before exec). In the daemon's group, a service manager that kills the
5926    // job's process group when the daemon exits (launchd's default) killed
5927    // every module at the same moment its control connection closed, so no
5928    // module ever ran its EOF teardown on a daemon stop. Outside that group a
5929    // module is reached only by the daemon: the EOF it sees when its
5930    // connection closes, and the bounded stop in `child_roster` for anything
5931    // still running after that. On Linux this composes with the cgroup
5932    // placement above: that is a pre_exec write to cgroup.procs, std performs
5933    // setpgid in the child before running pre_exec callbacks, and the two
5934    // change independent process attributes.
5935    //
5936    // stdin is /dev/null because a process outside the terminal's foreground
5937    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
5938    // by hand would otherwise hand down. Under a service manager stdin is
5939    // already /dev/null.
5940    #[cfg(unix)]
5941    command.process_group(0);
5942    command.stdin(Stdio::null());
5943    // The LAST pre-exec step, after the cgroup placement above: installing the
5944    // nonce at descriptor 3 replaces whatever the child had there, which could
5945    // be the descriptor an earlier step writes through.
5946    #[cfg(unix)]
5947    if let Some(handoff) = nonce_handoff {
5948        handoff.install_last(command.as_std_mut());
5949    }
5950    #[cfg(not(unix))]
5951    let _ = nonce_handoff;
5952
5953    // Containment, step 1 of 3 (issue #109): create the child suspended so it
5954    // cannot run a single instruction -- and therefore cannot spawn a
5955    // grandchild -- before it is in the job. See `contain_spawned_child` for the
5956    // other two steps and why the window matters.
5957    #[cfg(windows)]
5958    subc_jobobject::suspend_on_create_async(&mut command);
5959    let mut child = match command.spawn() {
5960        Ok(child) => child,
5961        Err(source) => {
5962            #[cfg(target_os = "linux")]
5963            if let Some(placement) = cgroup_placement {
5964                remove_module_cgroup(placement, &cgroup_name);
5965            }
5966            return Err(SuperviseError::Spawn {
5967                program: spec.program.clone(),
5968                source,
5969                cgroup_path,
5970            });
5971        }
5972    };
5973
5974    // Containment, steps 2 and 3: assign while suspended, then resume.
5975    #[cfg(windows)]
5976    let job = contain_spawned_child(&child, spec)?;
5977    let spawned_at_ms = unix_ms_now();
5978    let spawned_from = spec.program.clone();
5979    let spawned_file_identity = spawned_file_identity(&spawned_from);
5980    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
5981        program: spec.program.clone(),
5982        source: io::Error::other("spawned child exposed no live pid"),
5983        cgroup_path: cgroup_path.clone(),
5984    })?;
5985    let process_start_time = crate::provenance::process_start_time(pid);
5986    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
5987    // The executable identity is the spawned path's, read above, not the
5988    // running image's: right after spawn the child may not have finished its
5989    // exec yet and would still report this daemon's own image.
5990    #[cfg(target_os = "linux")]
5991    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
5992    #[cfg(not(target_os = "linux"))]
5993    let recorded_cgroup_name = None;
5994    let roster_guard = roster.admit(
5995        spec.module_id.clone(),
5996        pid,
5997        spec.protocol,
5998        process_start_time,
5999        crate::child_roster::RecordedIdentity {
6000            start_time: subc_os::start_time(pid),
6001            executable: spawned_file_identity.map(|identity| {
6002                crate::live_children::ExecutableIdentity {
6003                    device: identity.device,
6004                    inode: identity.inode,
6005                }
6006            }),
6007            cgroup_name: recorded_cgroup_name,
6008        },
6009    );
6010    // The check at the top of this function can pass just before daemon
6011    // shutdown begins, and the process is only in the roster from here on.
6012    // The shutdown stop returns as soon as it finds the roster empty, so a
6013    // process admitted after that look would outlive the daemon. The roster
6014    // is closed before the stop first reads it and admission happens under
6015    // the roster's lock, so either the stop sees this process or this check
6016    // sees the roster closed: end the process now rather than start a module
6017    // the daemon is about to stop.
6018    if roster.is_closed() {
6019        if let Err(error) = child.start_kill() {
6020            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
6021        }
6022        drop(roster_guard);
6023        return Err(SuperviseError::Spawn {
6024            program: spec.program.clone(),
6025            source: io::Error::other(
6026                "the daemon began shutting down while this process was starting; ended it",
6027            ),
6028            cgroup_path,
6029        });
6030    }
6031
6032    let stdout_pump = match child.stdout.take() {
6033        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
6034        None => {
6035            warn!(
6036                module_id = %spec.module_id,
6037                "spawned child exposed no stdout pipe; file capture will be incomplete"
6038            );
6039            None
6040        }
6041    };
6042    let stderr_pump = match child.stderr.take() {
6043        Some(stderr) => {
6044            let generation = ring
6045                .lock()
6046                .unwrap_or_else(|poisoned| poisoned.into_inner())
6047                .begin_process();
6048            Some(StderrPump {
6049                task: tokio::spawn(pump_stderr_to(
6050                    stderr,
6051                    Arc::clone(ring),
6052                    generation,
6053                    output_sink,
6054                )),
6055                generation,
6056            })
6057        }
6058        None => {
6059            // Spawning succeeded but the pipe did not materialise. Recording it as
6060            // uncaptured keeps the tail honest: the alternative is an empty tail
6061            // that reads as a module which printed nothing.
6062            ring.lock()
6063                .unwrap_or_else(|poisoned| poisoned.into_inner())
6064                .mark_not_captured("stderr pipe was not available on spawn");
6065            warn!(
6066                module_id = %spec.module_id,
6067                "spawned child exposed no stderr pipe; tail will be unavailable"
6068            );
6069            None
6070        }
6071    };
6072
6073    Ok(SupervisedChild {
6074        child,
6075        #[cfg(target_os = "linux")]
6076        module_id: cgroup_name,
6077        #[cfg(target_os = "linux")]
6078        cgroup_placement: cgroup_placement.cloned(),
6079        #[cfg(windows)]
6080        job,
6081        stdout_pump,
6082        stderr_pump,
6083        stderr_ring: Arc::clone(ring),
6084        spawned_at_ms,
6085        spawned_from,
6086        spawned_file_identity,
6087        process_start_time,
6088        process_identity,
6089        pid,
6090        roster_guard: Some(roster_guard),
6091    })
6092}
6093
6094/// Contain a freshly spawned Windows child and start it.
6095///
6096/// Steps 2 and 3 of the suspended-create contract: the job is created and the
6097/// child assigned **while it is still suspended** (step 1 is
6098/// `suspend_on_create_async` at the spawn site), then the child is resumed.
6099///
6100/// A child that is never resumed hangs forever holding a pid, so a resume
6101/// failure kills the child and fails the spawn rather than returning a
6102/// `SupervisedChild` that can never run.
6103///
6104/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
6105/// it did before this existed, whereas refusing to start one would be a new
6106/// outage. It is logged at warn because it means a helper process could leak.
6107#[cfg(windows)]
6108fn contain_spawned_child(
6109    child: &Child,
6110    spec: &ModuleSpec,
6111) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
6112    let module_id = spec.module_id.as_str();
6113    let Some(pid) = child.id() else {
6114        // The child exited between spawn and here. Its tree, if it made one,
6115        // needs no containment: nothing is left to contain.
6116        warn!(
6117            module_id,
6118            "spawned child had already exited before containment; no job object attached"
6119        );
6120        return Ok(None);
6121    };
6122
6123    let job = match subc_jobobject::JobObject::new() {
6124        Ok(job) => job,
6125        Err(source) => {
6126            warn!(
6127                module_id,
6128                error = %source,
6129                "could not create a job object; this module's helper processes will not be \
6130                 reaped on teardown"
6131            );
6132            // Resume regardless: leaving the child suspended would turn a
6133            // containment gap into a hung module.
6134            resume_suspended_child(pid, spec)?;
6135            return Ok(None);
6136        }
6137    };
6138
6139    if let Err(source) = job.assign(child) {
6140        warn!(
6141            module_id,
6142            error = %source,
6143            "could not assign the child to its job object; this module's helper processes \
6144             will not be reaped on teardown"
6145        );
6146        resume_suspended_child(pid, spec)?;
6147        return Ok(None);
6148    }
6149
6150    resume_suspended_child(pid, spec)?;
6151    Ok(Some(job))
6152}
6153
6154/// Resume a suspended child, killing it if it cannot be started.
6155///
6156/// A suspended process holds a pid and does nothing, so there is no useful
6157/// state to return: the caller gets an error and the spawn fails.
6158#[cfg(windows)]
6159fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
6160    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
6161        // Kill it here rather than leaving a suspended process for the caller
6162        // to notice; `kill_on_drop` would eventually do this, but the module
6163        // would have been reported as running in between.
6164        let _ = std::process::Command::new("taskkill.exe")
6165            .args(["/PID", &pid.to_string(), "/T", "/F"])
6166            .stdin(Stdio::null())
6167            .stdout(Stdio::null())
6168            .stderr(Stdio::null())
6169            .status();
6170        return Err(SuperviseError::Spawn {
6171            program: spec.program.clone(),
6172            source,
6173            cgroup_path: None,
6174        });
6175    }
6176    Ok(())
6177}
6178
6179#[cfg(target_os = "linux")]
6180fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
6181    match placement.remove_module(module_id) {
6182        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
6183        Err(error) => warn!(
6184            module_id,
6185            error = %error,
6186            "could not remove module cgroup after process exit; continuing teardown"
6187        ),
6188    }
6189}
6190
6191#[cfg(target_os = "linux")]
6192fn apply_cgroup_placement(
6193    command: &mut Command,
6194    spec: &ModuleSpec,
6195    path: &std::path::Path,
6196) -> Result<(), SuperviseError> {
6197    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
6198        module_id: spec.module_id.clone(),
6199        source,
6200    })
6201}
6202
6203fn capture_retention(spec: &ModuleSpec) -> Retention {
6204    let defaults = Retention::default();
6205    let value = |name: &str| {
6206        spec.env
6207            .iter()
6208            .rev()
6209            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
6210    };
6211    Retention {
6212        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
6213            .and_then(|value| value.parse().ok())
6214            .unwrap_or(defaults.max_file_mb),
6215        keep: value(CAPTURE_KEEP_ENV)
6216            .and_then(|value| value.parse().ok())
6217            .unwrap_or(defaults.keep),
6218        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
6219            .and_then(|value| value.parse().ok())
6220            .unwrap_or(defaults.max_age_days),
6221    }
6222}
6223
6224/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
6225/// module's registration to the exact process the supervisor spawned.
6226fn generate_launch_nonce() -> Result<String, SuperviseError> {
6227    let mut bytes = [0u8; 32];
6228    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
6229        reason: source.to_string(),
6230    })?;
6231    let mut hex = String::with_capacity(64);
6232    for b in bytes {
6233        use std::fmt::Write;
6234        let _ = write!(hex, "{b:02x}");
6235    }
6236    Ok(hex)
6237}
6238
6239/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
6240/// signal about how many leading bytes matched.
6241fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
6242    if a.len() != b.len() {
6243        return false;
6244    }
6245    let mut diff = 0u8;
6246    for (x, y) in a.iter().zip(b.iter()) {
6247        diff |= x ^ y;
6248    }
6249    diff == 0
6250}
6251
6252fn spawn_and_mark_running(
6253    spec: &ModuleSpec,
6254    runtime: &SupervisorRuntimeConfig,
6255    snapshot: &SharedSnapshot,
6256) -> Result<SupervisedChild, SuperviseError> {
6257    let child = spawn_child(
6258        spec,
6259        runtime.connection_file_path.as_deref(),
6260        runtime.supervisor_handle.as_ref(),
6261        &runtime.stderr_ring,
6262        runtime.capture_logs_dir.as_deref(),
6263        &runtime.child_roster,
6264        #[cfg(target_os = "linux")]
6265        runtime.cgroup_placement.as_ref(),
6266    )?;
6267    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
6268    Ok(child)
6269}
6270
6271enum RegistrationWaitOutcome {
6272    Registered,
6273    Exited(ExitReport),
6274    TimedOut,
6275}
6276
6277struct ReloadRegistrationFailure {
6278    exit_report: ExitReport,
6279    reason: String,
6280}
6281
6282#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6283enum BusyGaugeObservation {
6284    Quiescent,
6285    Busy,
6286    Omitted,
6287}
6288
6289fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
6290    let Some(metrics) = metrics.and_then(Value::as_object) else {
6291        return BusyGaugeObservation::Omitted;
6292    };
6293    let mut sum = 0u128;
6294    for gauge in gauges {
6295        let Some(value) = metrics.get(gauge) else {
6296            return BusyGaugeObservation::Omitted;
6297        };
6298        let Some(value) = value.as_u64() else {
6299            return BusyGaugeObservation::Busy;
6300        };
6301        sum = sum.saturating_add(u128::from(value));
6302    }
6303    if sum == 0 {
6304        BusyGaugeObservation::Quiescent
6305    } else {
6306        BusyGaugeObservation::Busy
6307    }
6308}
6309
6310fn declared_busy_gauges(
6311    registry: &Registry,
6312    module_id: &str,
6313) -> Result<Vec<String>, SuperviseError> {
6314    busy_gauges_of(
6315        registry
6316            .get_module(module_id)
6317            .map_err(SuperviseError::Registry)?,
6318    )
6319}
6320
6321/// [`declared_busy_gauges`] for the registration a connection holds, in any
6322/// slot: after cutover the incumbent is no longer the id's active
6323/// registration, and its own manifest is the one that names its gauges.
6324fn declared_busy_gauges_for_connection(
6325    registry: &Registry,
6326    connection_id: ConnectionId,
6327) -> Result<Vec<String>, SuperviseError> {
6328    busy_gauges_of(
6329        registry
6330            .get_module_by_connection(connection_id)
6331            .map_err(SuperviseError::Registry)?,
6332    )
6333}
6334
6335fn busy_gauges_of(
6336    registration: Option<crate::registry::ModuleRegistration>,
6337) -> Result<Vec<String>, SuperviseError> {
6338    let Some(registration) = registration else {
6339        return Ok(Vec::new());
6340    };
6341    let Some(self_signals) = registration.manifest.self_signals else {
6342        return Ok(Vec::new());
6343    };
6344
6345    let mut gauges = Vec::new();
6346    for declaration in self_signals {
6347        if declaration.kind != SelfSignalKind::Busy {
6348            continue;
6349        }
6350        match declaration.anchored_to {
6351            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
6352                gauges.extend(declared)
6353            }
6354            _ => {
6355                // An invalid Busy anchor is fail-safe: the empty name cannot be
6356                // present in a conforming health report, so this drain stays busy.
6357                gauges.push(String::new());
6358            }
6359        }
6360    }
6361    Ok(gauges)
6362}
6363
6364/// Wait for `endpoint` to have nothing in flight and, when the module declares
6365/// busy gauges, for a health probe to report them quiet. The probe is addressed
6366/// by `scope`: a swap's superseded incumbent must be asked about its own
6367/// gauges, and by module id the probe would reach the promoted candidate.
6368async fn wait_for_forwarding_quiescence(
6369    forwarding: &ForwardingTable,
6370    module_id: &str,
6371    runtime: &SupervisorRuntimeConfig,
6372    endpoint: crate::ModuleEndpointId,
6373    deadline: Instant,
6374    busy_gauges: &[String],
6375    scope: DrainScope,
6376) -> Result<bool, SuperviseError> {
6377    let mut gauges_quiescent = busy_gauges.is_empty();
6378    let mut next_probe_at = Instant::now();
6379    let mut omission_counted = false;
6380
6381    loop {
6382        let now = Instant::now();
6383        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
6384            let report = match scope {
6385                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
6386                DrainScope::Endpoint(endpoint) => {
6387                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
6388                }
6389            };
6390            gauges_quiescent = match report {
6391                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
6392                    BusyGaugeObservation::Quiescent => true,
6393                    BusyGaugeObservation::Busy => false,
6394                    BusyGaugeObservation::Omitted => {
6395                        if !omission_counted {
6396                            forwarding
6397                                .counters()
6398                                .increment_drains_with_undeclared_gauge();
6399                            omission_counted = true;
6400                        }
6401                        false
6402                    }
6403                },
6404                Err(err) => {
6405                    warn!(
6406                        module_id,
6407                        error = %err,
6408                        "drain health.check did not produce declared busy gauges; treating module as busy"
6409                    );
6410                    false
6411                }
6412            };
6413            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
6414        }
6415
6416        let in_flight = forwarding
6417            .endpoint_in_flight_count(endpoint)
6418            .map_err(SuperviseError::Forwarding)?;
6419        if in_flight == 0 && gauges_quiescent {
6420            return Ok(true);
6421        }
6422
6423        let now = Instant::now();
6424        if now >= deadline {
6425            return Ok(false);
6426        }
6427        let mut wait = deadline
6428            .saturating_duration_since(now)
6429            .min(REGISTRY_RELEASE_POLL);
6430        if !busy_gauges.is_empty() {
6431            wait = wait.min(next_probe_at.saturating_duration_since(now));
6432        }
6433        sleep(wait).await;
6434    }
6435}
6436
6437/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
6438///
6439/// `Ok` is always honest and passed straight through -- the wait actually measured
6440/// in-flight state. `Err` means the wait produced no measurement at all (the
6441/// forwarding table's lock was poisoned), so `false` is reported as the one honest
6442/// constant: the drain did not complete. Never recomputed from route state, never a
6443/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
6444fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
6445    match wait_result {
6446        Ok(drained) => *drained,
6447        Err(_) => false,
6448    }
6449}
6450
6451fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
6452    for released in released_routes {
6453        let frame = match Frame::build_with_version(
6454            released.negotiated_ver,
6455            FrameType::Goodbye,
6456            control_flags(),
6457            released.channel,
6458            released.epoch,
6459            0,
6460            Vec::new(),
6461        ) {
6462            Ok(frame) => frame,
6463            Err(err) => {
6464                warn!(
6465                    route_channel = released.channel,
6466                    error = %err,
6467                    "failed to build supervisor drain route GOODBYE frame"
6468                );
6469                continue;
6470            }
6471        };
6472        if !released.close_on_delivery_failure() {
6473            crate::forwarding::send_module_route_goodbye(
6474                &forwarding.counters(),
6475                &released.sink,
6476                frame,
6477                released.module_id.as_deref(),
6478                "supervisor drain",
6479            );
6480            continue;
6481        }
6482        if let Err(err) = released.sink.try_send(frame) {
6483            warn!(
6484                target_connection_id = released.connection_id.get(),
6485                route_channel = released.channel,
6486                error = %err,
6487                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
6488            );
6489            let _ = forwarding.escalate_client_delivery_failure(
6490                released.connection_id,
6491                released.channel,
6492                released.epoch,
6493                CloseReason::new(
6494                    "route_goodbye_delivery_failed",
6495                    format!(
6496                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
6497                        released.channel
6498                    ),
6499                ),
6500                crate::forwarding::UndeliveredFrame {
6501                    module_id: released.module_id.as_deref(),
6502                    sink: &released.sink,
6503                },
6504            );
6505        }
6506    }
6507}
6508
6509fn send_module_draining(
6510    module_id: &str,
6511    reason: RouteCloseReason,
6512    deadline_ms: u64,
6513    target: &ModuleDrainTarget,
6514) {
6515    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
6516        reason,
6517        deadline_ms,
6518    }) {
6519        Ok(body) => body,
6520        Err(err) => {
6521            warn!(
6522                module_id,
6523                error = %err,
6524                "failed to encode module draining command"
6525            );
6526            return;
6527        }
6528    };
6529    let frame = match Frame::build_with_version(
6530        target.negotiated_ver,
6531        FrameType::Push,
6532        control_flags(),
6533        0,
6534        0,
6535        0,
6536        body,
6537    ) {
6538        Ok(frame) => frame,
6539        Err(err) => {
6540            warn!(
6541                module_id,
6542                error = %err,
6543                "failed to build module draining command frame"
6544            );
6545            return;
6546        }
6547    };
6548    if let Err(err) = target.sink.try_send(frame) {
6549        warn!(
6550            module_id,
6551            target_connection_id = target.endpoint.connection_id.get(),
6552            error = %err,
6553            "module draining command was not delivered to peer"
6554        );
6555    }
6556}
6557
6558/// The channel-0 GOODBYE that tells a module its stop is planned.
6559fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
6560    match Frame::build_with_version(
6561        negotiated_ver,
6562        FrameType::Goodbye,
6563        control_flags(),
6564        0,
6565        0,
6566        0,
6567        Vec::new(),
6568    ) {
6569        Ok(frame) => Some(frame),
6570        Err(err) => {
6571            warn!(
6572                module_id,
6573                error = %err,
6574                "failed to build module GOODBYE frame"
6575            );
6576            None
6577        }
6578    }
6579}
6580
6581/// Send every registered module connection its module GOODBYE at daemon
6582/// shutdown, then request that connection's close.
6583///
6584/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
6585/// before EOF, so the GOODBYE must reach the socket before the close. A close
6586/// request does not wait for the connection's queued frames: its writer gets a
6587/// bounded grace after the close, is aborted if it overruns it, and the daemon
6588/// process may exit before that grace ends. So with `wait_for_flush`, each
6589/// connection is closed only after its writer has acknowledged writing the
6590/// GOODBYE, or once a short shared budget runs out, so one module that is not
6591/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
6592/// are only queued, for a shutdown the operator has told to stop waiting.
6593/// A connection that is already gone is skipped.
6594#[cfg(unix)]
6595async fn send_module_goodbyes_for_daemon_shutdown(
6596    forwarding: &Arc<ForwardingTable>,
6597    reason: &CloseReason,
6598    wait_for_flush: bool,
6599) {
6600    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
6601    let targets = match forwarding.module_connections() {
6602        Ok(targets) => targets,
6603        Err(err) => {
6604            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
6605            return;
6606        }
6607    };
6608    let deadline = Instant::now() + GOODBYE_BUDGET;
6609    let mut sends = tokio::task::JoinSet::new();
6610    for target in targets {
6611        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
6612            continue;
6613        };
6614        if !wait_for_flush {
6615            if let Err(err) = target.sink.try_send(frame) {
6616                debug!(
6617                    module_id = %target.module_id,
6618                    error = %err,
6619                    "shutdown module GOODBYE was not queued"
6620                );
6621            }
6622            continue;
6623        }
6624        let forwarding = Arc::clone(forwarding);
6625        let reason = reason.clone();
6626        sends.spawn(async move {
6627            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
6628                Ok(Ok(())) => {}
6629                Ok(Err(err)) => debug!(
6630                    module_id = %target.module_id,
6631                    error = %err,
6632                    "module connection closed before its shutdown GOODBYE was written"
6633                ),
6634                Err(_) => warn!(
6635                    module_id = %target.module_id,
6636                    budget = ?GOODBYE_BUDGET,
6637                    "shutdown module GOODBYE was not written within its budget; closing anyway"
6638                ),
6639            }
6640            forwarding.request_connection_close(target.endpoint.connection_id, reason);
6641        });
6642    }
6643    // Every task ends by the shared deadline, so this wait is bounded too.
6644    while sends.join_next().await.is_some() {}
6645}
6646
6647fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
6648    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
6649        return;
6650    };
6651    if let Err(err) = target.sink.try_send(frame) {
6652        warn!(
6653            module_id,
6654            target_connection_id = target.endpoint.connection_id.get(),
6655            error = %err,
6656            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
6657        );
6658        forwarding.request_connection_close(
6659            target.endpoint.connection_id,
6660            CloseReason::new(
6661                "module_goodbye_delivery_failed",
6662                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
6663            ),
6664        );
6665    }
6666}
6667
6668#[derive(Clone, Copy)]
6669struct ForwardingDrainContext<'a> {
6670    spec: &'a ModuleSpec,
6671    runtime: &'a SupervisorRuntimeConfig,
6672    registry: &'a Registry,
6673    scope: DrainScope,
6674}
6675
6676/// Which process a forwarding drain addresses.
6677#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6678enum DrainScope {
6679    /// Whatever endpoint is active for the module id: every plain stop,
6680    /// restart and reload. Also moves the module's state to `Draining`.
6681    Active,
6682    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
6683    /// module id would resolve to the promoted candidate and leave neither
6684    /// process routable. The module's state is left alone, since the promoted
6685    /// candidate is what it describes and that process is running.
6686    Endpoint(crate::ModuleEndpointId),
6687}
6688
6689/// Whether a child being drained has already been asked to stop by the time
6690/// its drain wait starts.
6691///
6692/// The drain wait is the same budget whatever this says. What it decides is
6693/// whether the supervisor must ask by signal before that wait begins: a child
6694/// that nobody asked will sit out the whole budget and then be SIGKILLed,
6695/// healthy or not.
6696#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6697enum StopNotice {
6698    /// The module was sent `module.draining` and a module GOODBYE over its own
6699    /// registered connection, and stops itself.
6700    SentOverConnection,
6701    /// The forwarding drain found no registered connection for the module: a
6702    /// subc child spawned moments ago that has not sent HELLO yet, or a
6703    /// `protocol: "none"` child, which never registers.
6704    NoConnection,
6705    /// This path sends nothing over the module's connection: the supervisor has
6706    /// no forwarding table, or the caller stops the child without a forwarding
6707    /// drain.
6708    NotSent,
6709}
6710
6711async fn begin_forwarding_drain(
6712    spec: &ModuleSpec,
6713    runtime: &SupervisorRuntimeConfig,
6714    registry: &Registry,
6715    snapshot: &SharedSnapshot,
6716    enabled: Option<bool>,
6717    reason: RouteCloseReason,
6718) -> Result<StopNotice, SuperviseError> {
6719    let Some(forwarding) = runtime.forwarding.as_ref() else {
6720        return Err(SuperviseError::ReloadUnavailable {
6721            module_id: spec.module_id.clone(),
6722            reason: "supervisor was not configured with a forwarding table".to_string(),
6723        });
6724    };
6725
6726    begin_forwarding_drain_with(
6727        forwarding,
6728        ForwardingDrainContext {
6729            spec,
6730            runtime,
6731            registry,
6732            scope: DrainScope::Active,
6733        },
6734        snapshot,
6735        enabled,
6736        reason,
6737        runtime.drain_timeout,
6738    )
6739    .await
6740}
6741
6742async fn begin_forwarding_drain_if_configured(
6743    spec: &ModuleSpec,
6744    runtime: &SupervisorRuntimeConfig,
6745    registry: &Registry,
6746    snapshot: &SharedSnapshot,
6747    enabled: Option<bool>,
6748    reason: RouteCloseReason,
6749) -> Result<StopNotice, SuperviseError> {
6750    begin_forwarding_drain_with_timeout(
6751        spec,
6752        runtime,
6753        registry,
6754        snapshot,
6755        enabled,
6756        reason,
6757        runtime.drain_timeout,
6758    )
6759    .await
6760}
6761
6762/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
6763/// budget, for paths where the operator overrides the module's configured one
6764/// (`supervisor.restart{drain_timeout_ms}`).
6765async fn begin_forwarding_drain_with_timeout(
6766    spec: &ModuleSpec,
6767    runtime: &SupervisorRuntimeConfig,
6768    registry: &Registry,
6769    snapshot: &SharedSnapshot,
6770    enabled: Option<bool>,
6771    reason: RouteCloseReason,
6772    drain_timeout: Duration,
6773) -> Result<StopNotice, SuperviseError> {
6774    let Some(forwarding) = runtime.forwarding.as_ref() else {
6775        return Ok(StopNotice::NotSent);
6776    };
6777
6778    begin_forwarding_drain_with(
6779        forwarding,
6780        ForwardingDrainContext {
6781            spec,
6782            runtime,
6783            registry,
6784            scope: DrainScope::Active,
6785        },
6786        snapshot,
6787        enabled,
6788        reason,
6789        drain_timeout,
6790    )
6791    .await
6792}
6793
6794async fn begin_forwarding_drain_with(
6795    forwarding: &ForwardingTable,
6796    context: ForwardingDrainContext<'_>,
6797    snapshot: &SharedSnapshot,
6798    enabled: Option<bool>,
6799    reason: RouteCloseReason,
6800    drain_timeout: Duration,
6801) -> Result<StopNotice, SuperviseError> {
6802    let ForwardingDrainContext {
6803        spec,
6804        runtime,
6805        registry,
6806        scope,
6807    } = context;
6808    debug_assert_ne!(reason, RouteCloseReason::Crash);
6809    let terminal = matches!(reason, RouteCloseReason::Disable);
6810    let drain_started_at = Instant::now();
6811    let drain_deadline = drain_started_at + drain_timeout;
6812    let deadline_ms =
6813        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
6814    let busy_gauges = match scope {
6815        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
6816        DrainScope::Endpoint(endpoint) => {
6817            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
6818        }
6819    };
6820
6821    // Admission gate first: route.open/commit and route REQUEST admission are closed
6822    // before the first quiescence check, so the outstanding count can only fall.
6823    let gate_started = Instant::now();
6824    let drain_target = match scope {
6825        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
6826        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
6827    }
6828    .map_err(SuperviseError::Forwarding)?;
6829    // The instant admission closed, and how long taking the forwarding write
6830    // lock to close it took. The timeout line reports only the quiescence
6831    // wait, so without this a drain that started late looked like one that
6832    // started on time.
6833    info!(
6834        module_id = %spec.module_id,
6835        ?reason,
6836        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
6837        connected = drain_target.is_some(),
6838        "module drain began; route admission closed"
6839    );
6840    if scope == DrainScope::Active {
6841        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6842            state.state = ModuleState::Draining;
6843            state.draining_to_replace =
6844                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
6845            if let Some(enabled) = enabled {
6846                state.enabled = enabled;
6847            }
6848        })?;
6849    }
6850
6851    let Some(target) = drain_target.as_ref() else {
6852        // Nothing was sent: the module has no registered connection to carry
6853        // `module.draining` or a GOODBYE. The caller must not assume the child
6854        // was asked to stop.
6855        return Ok(StopNotice::NoConnection);
6856    };
6857    {
6858        send_module_draining(&spec.module_id, reason, deadline_ms, target);
6859        let routes = forwarding
6860            .endpoint_routes(target.endpoint)
6861            .map_err(SuperviseError::Forwarding)?;
6862        let routes_notified = routes.len();
6863        crate::control::send_route_control_pushes(
6864            forwarding,
6865            routes.clone(),
6866            ClientControlPush::RouteClosing {
6867                module_id: spec.module_id.clone(),
6868                reason,
6869            },
6870        );
6871        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
6872
6873        // `route.closing` was just sent above: from here on every return path,
6874        // including an early one, MUST send `route.closed` before propagating
6875        // anything else. A client holds `closing` as a promise that a verdict is
6876        // coming; leaving early without `closed` strands it waiting forever, since
6877        // `closing` carries no timeout of its own.
6878        let wait_result = wait_for_forwarding_quiescence(
6879            forwarding,
6880            &spec.module_id,
6881            runtime,
6882            target.endpoint,
6883            drain_deadline,
6884            &busy_gauges,
6885            scope,
6886        )
6887        .await;
6888        let drained = drained_after_quiescence_wait(&wait_result);
6889        if let Err(err) = &wait_result {
6890            error!(
6891                module_id = %spec.module_id,
6892                ?reason,
6893                error = %err,
6894                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
6895            );
6896        } else if !drained {
6897            // Name what the drain waited on. Without it the line says only that
6898            // something did not settle, and "one wedged call" and "every
6899            // session's held stream" read the same; the first is a module bug,
6900            // the second is a module that should end its streams on
6901            // module.draining. Read before teardown releases the routes.
6902            let holdouts = forwarding
6903                .endpoint_drain_holdouts(target.endpoint)
6904                .unwrap_or_default();
6905            warn!(
6906                module_id = %spec.module_id,
6907                waited = ?drain_timeout,
6908                ?reason,
6909                held_requests = holdouts.requests,
6910                held_routes = holdouts.routes,
6911                total_routes = holdouts.total_routes,
6912                top_connections = ?holdouts.top_connections,
6913                // `module_channel:corr`, so the module can find each held request
6914                // in its own log; capped, so `held_requests` is the full count.
6915                held = %holdouts
6916                    .held
6917                    .iter()
6918                    .map(|(channel, corr)| format!("{channel}:{corr}"))
6919                    .collect::<Vec<_>>()
6920                    .join(","),
6921                "route drain timed out before request quiescence; forcing teardown"
6922            );
6923        }
6924        crate::control::send_route_control_pushes(
6925            forwarding,
6926            routes,
6927            ClientControlPush::RouteClosed {
6928                module_id: spec.module_id.clone(),
6929                reason,
6930                drained,
6931                abandoned: target.abandoned_bindings.len() as u32,
6932                excluded_subscriptions: target.excluded_subscriptions,
6933                terminal: Some(terminal),
6934            },
6935        );
6936        wait_result?;
6937
6938        // `route.closed` has now been sent unconditionally above. From here the
6939        // remaining steps are cleanup (route + module GOODBYE) rather than a
6940        // promise the client is waiting on, but a lock-poisoned
6941        // `release_module_endpoint_routes` would otherwise skip the module
6942        // GOODBYE silently too -- send it before propagating the error.
6943        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
6944            Ok(routes) => routes,
6945            Err(err) => {
6946                warn!(
6947                    module_id = %spec.module_id,
6948                    ?reason,
6949                    error = %err,
6950                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
6951                );
6952                send_module_goodbye(&spec.module_id, forwarding, target);
6953                return Err(SuperviseError::Forwarding(err));
6954            }
6955        };
6956        let route_goodbye_count = released_routes.len();
6957        send_route_goodbyes(forwarding, released_routes);
6958        send_module_goodbye(&spec.module_id, forwarding, target);
6959
6960        // The drain's happy path was previously silent: every emission above is
6961        // best-effort with only its failure arm logged, so "were consumers told"
6962        // was unprovable from the daemon log (surfaced by a 30-minute consumer
6963        // hang where the open question was exactly whether teardown notice went
6964        // out). One summary line makes that class decidable in one grep.
6965        info!(
6966            module_id = %spec.module_id,
6967            ?reason,
6968            routes_notified,
6969            route_goodbyes = route_goodbye_count,
6970            abandoned_reservations = target.abandoned_bindings.len(),
6971            excluded_subscriptions = target.excluded_subscriptions,
6972            drained,
6973            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
6974        );
6975    }
6976
6977    Ok(StopNotice::SentOverConnection)
6978}
6979
6980/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
6981/// the only slot a plain (non-swap) spawn can register into.
6982async fn wait_for_registration_after_reload(
6983    registry: &Registry,
6984    module_id: &str,
6985    snapshot: &SharedSnapshot,
6986    child: &mut SupervisedChild,
6987    wait: Duration,
6988) -> Result<RegistrationWaitOutcome, SuperviseError> {
6989    wait_for_slot_registration(
6990        registry,
6991        crate::registry::RegistrationSlot::Active(module_id),
6992        module_id,
6993        snapshot,
6994        child,
6995        wait,
6996    )
6997    .await
6998}
6999
7000/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
7001///
7002/// Keyed on the slot rather than the bare module id because during a swap the
7003/// id's active slot is already held by the incumbent: an id-keyed wait would
7004/// report the incumbent's registration as the candidate's and a candidate that
7005/// never registers would look registered. A swap candidate waits on
7006/// `crate::registry::RegistrationSlot::Candidate`.
7007async fn wait_for_slot_registration(
7008    registry: &Registry,
7009    slot: crate::registry::RegistrationSlot<'_>,
7010    module_id: &str,
7011    snapshot: &SharedSnapshot,
7012    child: &mut SupervisedChild,
7013    wait: Duration,
7014) -> Result<RegistrationWaitOutcome, SuperviseError> {
7015    let deadline = Instant::now() + wait;
7016    loop {
7017        if registry
7018            .registration(slot)
7019            .map_err(SuperviseError::Registry)?
7020            .is_some()
7021        {
7022            return Ok(RegistrationWaitOutcome::Registered);
7023        }
7024
7025        let now = Instant::now();
7026        if now >= deadline {
7027            return Ok(RegistrationWaitOutcome::TimedOut);
7028        }
7029        let remaining = deadline.saturating_duration_since(now);
7030        let poll = remaining.min(REGISTRY_RELEASE_POLL);
7031
7032        tokio::select! {
7033            wait_result = child.wait() => {
7034                let status = wait_result.map_err(|source| SuperviseError::Wait {
7035                    module_id: module_id.to_string(),
7036                    source,
7037                })?;
7038                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
7039                    snapshot,
7040                    child,
7041                    &status,
7042                )));
7043            }
7044            _ = sleep(poll) => {}
7045        }
7046    }
7047}
7048
7049fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
7050    // A replacement process that exits before HELLO did not provide service, even
7051    // if it used status 0. Count it against the restart cap as a new-binary failure.
7052    if exit_report.kind != ExitKind::DeliberateSeverance {
7053        exit_report.kind = ExitKind::Crash;
7054    }
7055    exit_report
7056}
7057
7058async fn handle_reload_child_registration_failure(
7059    spec: &ModuleSpec,
7060    runtime: &SupervisorRuntimeConfig,
7061    registry: &Registry,
7062    process_liveness: &SupervisorProcessLiveness,
7063    snapshot: &SharedSnapshot,
7064    child: &mut Option<SupervisedChild>,
7065    failure: ReloadRegistrationFailure,
7066) -> Result<(), SuperviseError> {
7067    let ReloadRegistrationFailure {
7068        exit_report,
7069        reason,
7070    } = failure;
7071    match on_child_exit(
7072        spec,
7073        runtime.restart_policy,
7074        registry,
7075        snapshot,
7076        &runtime.terminal_ring,
7077        &runtime.spawn_events,
7078        &runtime.child_roster,
7079        exit_report,
7080    )
7081    .await
7082    {
7083        NextAction::Stop {
7084            registration_released,
7085        } => {
7086            if registration_released {
7087                process_liveness.untrack_if_current(&spec.module_id, snapshot);
7088            }
7089        }
7090        NextAction::Restart { schedule } => {
7091            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
7092                schedule.delay
7093            });
7094            if let Some(schedule) = schedule {
7095                log_crash_respawn(&spec.module_id, schedule);
7096            }
7097            sleep(delay).await;
7098            // A disable or drain that landed during the backoff cancels this
7099            // policy retry: the operator's stop must win over the respawn the
7100            // sleep counted down to.
7101            if respawn_still_pending(snapshot) {
7102                if let Err(err) = wait_for_registration_release(
7103                    registry,
7104                    &spec.module_id,
7105                    REGISTRY_RELEASE_TIMEOUT,
7106                )
7107                .await
7108                {
7109                    fail_snapshot(snapshot, Some(&spec.module_id), None);
7110                    process_liveness.untrack_if_current(&spec.module_id, snapshot);
7111                    return Err(SuperviseError::ReloadFailed {
7112                        module_id: spec.module_id.clone(),
7113                        reason: format!(
7114                            "{reason}; registration did not release before policy retry: {err}"
7115                        ),
7116                    });
7117                }
7118                process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7119                match spawn_and_mark_running(spec, runtime, snapshot) {
7120                    Ok(next_child) => {
7121                        *child = Some(next_child);
7122                    }
7123                    Err(err) => {
7124                        fail_snapshot(snapshot, Some(&spec.module_id), None);
7125                        process_liveness.untrack_if_current(&spec.module_id, snapshot);
7126                        return Err(SuperviseError::ReloadFailed {
7127                            module_id: spec.module_id.clone(),
7128                            reason: format!("{reason}; policy retry spawn failed: {err}"),
7129                        });
7130                    }
7131                }
7132            }
7133        }
7134    }
7135
7136    Err(SuperviseError::ReloadFailed {
7137        module_id: spec.module_id.clone(),
7138        reason,
7139    })
7140}
7141
7142async fn handle_reload_spawn_failure(
7143    spec: &ModuleSpec,
7144    runtime: &SupervisorRuntimeConfig,
7145    process_liveness: &SupervisorProcessLiveness,
7146    snapshot: &SharedSnapshot,
7147    child: &mut Option<SupervisedChild>,
7148    reason: String,
7149) -> Result<(), SuperviseError> {
7150    let mut should_retry = false;
7151    let now = Instant::now();
7152    update_snapshot(snapshot, Some(&spec.module_id), |state| {
7153        clear_current_process_facts(state);
7154        if daemon_will_restart(state, &runtime.restart_policy, now) {
7155            state.record_crash_restart(&runtime.restart_policy, now);
7156            state.state = ModuleState::Restarting;
7157            should_retry = true;
7158        } else if state.enabled {
7159            state.state = ModuleState::Failed;
7160        } else {
7161            state.state = ModuleState::Disabled;
7162        }
7163    })?;
7164
7165    if should_retry {
7166        sleep(runtime.restart_policy.backoff).await;
7167        // A disable or drain that landed during the backoff cancels this
7168        // policy retry: the operator's stop must win over the respawn the
7169        // sleep counted down to.
7170        if respawn_still_pending(snapshot) {
7171            process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7172            match spawn_and_mark_running(spec, runtime, snapshot) {
7173                Ok(next_child) => {
7174                    *child = Some(next_child);
7175                }
7176                Err(err) => {
7177                    fail_snapshot(snapshot, Some(&spec.module_id), None);
7178                    process_liveness.untrack_if_current(&spec.module_id, snapshot);
7179                    return Err(SuperviseError::ReloadFailed {
7180                        module_id: spec.module_id.clone(),
7181                        reason: format!("{reason}; policy retry spawn failed: {err}"),
7182                    });
7183                }
7184            }
7185        }
7186    } else {
7187        process_liveness.untrack_if_current(&spec.module_id, snapshot);
7188    }
7189
7190    Err(SuperviseError::ReloadFailed {
7191        module_id: spec.module_id.clone(),
7192        reason,
7193    })
7194}
7195
7196fn control_flags() -> Flags {
7197    Flags::new(false, Priority::Passive, false)
7198}
7199
7200#[allow(clippy::too_many_arguments)]
7201async fn drain_optional_child(
7202    module_id: &str,
7203    protocol: ModuleProtocol,
7204    stop_notice: StopNotice,
7205    registry: &Registry,
7206    snapshot: &SharedSnapshot,
7207    terminal_ring: &Arc<Mutex<TerminalRing>>,
7208    spawn_events: &SpawnEventFeed,
7209    child: &mut Option<SupervisedChild>,
7210    drain_timeout: Duration,
7211    final_state: ModuleState,
7212    enabled: Option<bool>,
7213) -> Result<(), SuperviseError> {
7214    if let Some(child) = child.take() {
7215        drain_child_to_state(
7216            module_id,
7217            protocol,
7218            stop_notice,
7219            registry,
7220            snapshot,
7221            terminal_ring,
7222            spawn_events,
7223            child,
7224            drain_timeout,
7225            final_state,
7226            enabled,
7227        )
7228        .await
7229    } else {
7230        update_snapshot(snapshot, Some(module_id), |state| {
7231            state.state = final_state;
7232            if let Some(enabled) = enabled {
7233                state.enabled = enabled;
7234            }
7235            clear_current_process_facts(state);
7236        })?;
7237        wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7238    }
7239}
7240
7241#[allow(clippy::too_many_arguments)]
7242async fn drain_child_to_state(
7243    module_id: &str,
7244    protocol: ModuleProtocol,
7245    stop_notice: StopNotice,
7246    registry: &Registry,
7247    snapshot: &SharedSnapshot,
7248    terminal_ring: &Arc<Mutex<TerminalRing>>,
7249    spawn_events: &SpawnEventFeed,
7250    mut child: SupervisedChild,
7251    drain_timeout: Duration,
7252    final_state: ModuleState,
7253    enabled: Option<bool>,
7254) -> Result<(), SuperviseError> {
7255    update_snapshot(snapshot, Some(module_id), |state| {
7256        state.state = ModuleState::Draining;
7257        state.draining_to_replace = final_state == ModuleState::Restarting;
7258        if let Some(enabled) = enabled {
7259            state.enabled = enabled;
7260        }
7261    })?;
7262
7263    // The wait below is the same budget in every case; what differs is
7264    // whether anything has ASKED the child to stop before it starts. Only a
7265    // forwarding drain that reached the module's registered connection has
7266    // (`module.draining`, then a module GOODBYE). Every other child was told
7267    // nothing: a `protocol: "none"` module, which never registers; a subc
7268    // module spawned moments ago that has not sent HELLO yet; or a stop that
7269    // runs no forwarding drain. Without a signal the budget is only a delay
7270    // in front of SIGKILL -- and the not-yet-registered child is the worst
7271    // case, because it registers into a module that is already draining,
7272    // is never told, and is killed while healthy.
7273    if stop_notice != StopNotice::SentOverConnection {
7274        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
7275            info!(
7276                module_id,
7277                pid = child.pid,
7278                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7279                "module has no connection yet; requesting stop by signal"
7280            );
7281        }
7282        request_graceful_stop(module_id, &child);
7283    }
7284
7285    let exit_report = match timeout(drain_timeout, child.wait()).await {
7286        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
7287        Ok(Err(source)) => {
7288            fail_snapshot(snapshot, Some(module_id), None);
7289            return Err(SuperviseError::Wait {
7290                module_id: module_id.to_string(),
7291                source,
7292            });
7293        }
7294        Err(_) => {
7295            // Mirror the sibling arm above: state is already `Draining`, and an
7296            // error propagated from here would strand it there -- a state
7297            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
7298            // `Failed | Stopped`), leaving an operator Restart as the only exit.
7299            // `Failed` before `?` keeps the module operator-visible and
7300            // revivable. Trigger is an ESRCH race (process exits between the
7301            // drain timeout firing and the kill) or a post-kill wait failure
7302            // (issue #34).
7303            //
7304            // Logged because the kill is otherwise visible only as signal 9 in
7305            // the terminal ring, and the budget it follows can be long enough
7306            // that consumers see a stretch of refusals with no stated cause.
7307            warn!(
7308                module_id,
7309                pid = child.pid,
7310                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7311                reason = ?final_state,
7312                ?stop_notice,
7313                "drain budget expired before the module exited; killing it"
7314            );
7315            child.start_kill().map_err(|source| {
7316                fail_snapshot(snapshot, Some(module_id), None);
7317                SuperviseError::Kill {
7318                    module_id: module_id.to_string(),
7319                    source,
7320                }
7321            })?;
7322            let status = child.wait().await.map_err(|source| {
7323                fail_snapshot(snapshot, Some(module_id), None);
7324                SuperviseError::Wait {
7325                    module_id: module_id.to_string(),
7326                    source,
7327                }
7328            })?;
7329            classify_reaped_child_exit(snapshot, &child, &status)
7330        }
7331    };
7332
7333    update_snapshot(snapshot, Some(module_id), |state| {
7334        state.state = final_state;
7335        if let Some(enabled) = enabled {
7336            state.enabled = enabled;
7337        }
7338        clear_current_process_facts(state);
7339        state.last_exit = Some(exit_report.clone());
7340        if exit_report.kind == ExitKind::DeliberateSeverance {
7341            state.lifetime_restarts += 1;
7342        }
7343    })?;
7344    record_terminal(
7345        module_id,
7346        terminal_ring,
7347        spawn_events,
7348        &exit_report,
7349        terminal_disposition(final_state),
7350    );
7351    child.drain_stderr(module_id).await;
7352
7353    wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7354}
7355
7356/// Ask a child that nothing else has asked to stop, by signal.
7357///
7358/// A registered subc module is asked over its own connection: the drain sends
7359/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
7360/// module GOODBYE, and the module stops itself. A module that speaks no subc
7361/// wire receives none of that, and neither does a subc module that has not
7362/// registered yet, so for them the drain budget would be pure delay in front of
7363/// a SIGKILL -- and for a process with a store to flush (JetStream is the
7364/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
7365/// into a recovery on the next start.
7366///
7367/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
7368/// rule rather than an optimisation: that module's graceful stop is already
7369/// running by the time its child is drained, and a signal would race it.
7370///
7371/// Best-effort by construction. A child that has already exited is the ordinary
7372/// case rather than an error (the kill lands on a reaped or exiting pid), so a
7373/// failure is logged at debug and the wait-then-kill below still decides the
7374/// outcome.
7375#[cfg(unix)]
7376fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
7377    let Some(pid) = child
7378        .id()
7379        .and_then(|pid| i32::try_from(pid).ok())
7380        .and_then(rustix::process::Pid::from_raw)
7381    else {
7382        debug!(
7383            module_id,
7384            "no pid to signal for teardown; falling through to the drain wait"
7385        );
7386        return;
7387    };
7388    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
7389        Ok(()) => debug!(
7390            module_id,
7391            "sent SIGTERM to a module nothing else asked to stop"
7392        ),
7393        Err(err) => debug!(
7394            module_id,
7395            error = %err,
7396            "SIGTERM to module failed; the drain wait and kill still apply"
7397        ),
7398    }
7399}
7400
7401/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
7402/// Windows does offer need cooperation this supervisor cannot assume: a console
7403/// control event requires sharing a console with the child, and `WM_CLOSE`
7404/// requires the child to pump a message loop. A supervised server process does
7405/// neither, so there is nothing to send and teardown is the wait followed by the
7406/// kill. Emulating a signal here would mean inventing a stop protocol, which is
7407/// the thing `protocol: "none"` exists to avoid.
7408#[cfg(not(unix))]
7409fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
7410    debug!(
7411        module_id,
7412        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
7413    );
7414}
7415
7416fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
7417    match final_state {
7418        ModuleState::Stopped => TerminalDisposition::Stopped,
7419        ModuleState::Disabled => TerminalDisposition::Disabled,
7420        ModuleState::Restarting => TerminalDisposition::Restarting,
7421        ModuleState::Failed => TerminalDisposition::Failed,
7422        ModuleState::Starting
7423        | ModuleState::Running
7424        | ModuleState::Unresponsive
7425        | ModuleState::Draining => {
7426            unreachable!("terminal exits only finish in terminal or restarting states")
7427        }
7428    }
7429}
7430
7431/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
7432/// plain stop or restart waits for before it spawns a replacement.
7433async fn wait_for_registration_release(
7434    registry: &Registry,
7435    module_id: &str,
7436    wait: Duration,
7437) -> Result<(), SuperviseError> {
7438    wait_for_slot_registration_release(
7439        registry,
7440        crate::registry::RegistrationSlot::Active(module_id),
7441        wait,
7442    )
7443    .await
7444}
7445
7446/// Wait for the registration in `slot` to go away.
7447///
7448/// Keyed on the slot rather than the bare module id because a successful swap
7449/// never empties the id's active slot (the promoted candidate is in it), so an
7450/// id-keyed wait for the incumbent's release would always time out. Draining a
7451/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
7452/// incumbent's connection instead.
7453async fn wait_for_slot_registration_release(
7454    registry: &Registry,
7455    slot: crate::registry::RegistrationSlot<'_>,
7456    wait: Duration,
7457) -> Result<(), SuperviseError> {
7458    let deadline = Instant::now() + wait;
7459    let mut release_events = registration_release_events().subscribe();
7460    let still_active = |registration: &crate::registry::ModuleRegistration| {
7461        SuperviseError::RegistrationStillActive {
7462            module_id: registration.manifest.module_id.clone(),
7463            waited: wait,
7464        }
7465    };
7466    loop {
7467        let _observed_generation = *release_events.borrow_and_update();
7468        let Some(registration) = registry
7469            .registration(slot)
7470            .map_err(SuperviseError::Registry)?
7471        else {
7472            return Ok(());
7473        };
7474
7475        let now = Instant::now();
7476        if now >= deadline {
7477            return Err(still_active(&registration));
7478        }
7479
7480        let remaining = deadline.saturating_duration_since(now);
7481        match timeout(remaining, release_events.changed()).await {
7482            Ok(Ok(())) | Ok(Err(_)) => {}
7483            Err(_) => return Err(still_active(&registration)),
7484        }
7485    }
7486}
7487
7488#[cfg(test)]
7489mod slot_registration_wait_tests {
7490    use super::*;
7491    use crate::registry::{ConnectionId, RegistrationSlot};
7492    use subc_protocol::manifest::ModuleManifest;
7493
7494    const INCUMBENT: u64 = 1;
7495    const CANDIDATE: u64 = 2;
7496
7497    fn swapped_registry() -> Arc<Registry> {
7498        let registry = Arc::new(Registry::default());
7499        let manifest = ModuleManifest::builder("m", "0.1.0").build();
7500        registry
7501            .register_with_control_ops(
7502                manifest.clone(),
7503                1,
7504                ConnectionId::new(INCUMBENT),
7505                Vec::new(),
7506            )
7507            .unwrap();
7508        registry
7509            .register_candidate_with_control_ops(
7510                manifest,
7511                1,
7512                ConnectionId::new(CANDIDATE),
7513                Vec::new(),
7514            )
7515            .unwrap();
7516        registry
7517    }
7518
7519    /// After a promotion the id's active slot is held by the new process, so an
7520    /// id-keyed wait for the incumbent's release can never succeed; the
7521    /// connection-keyed wait completes as soon as the incumbent deregisters.
7522    #[tokio::test]
7523    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
7524        let registry = swapped_registry();
7525        registry.promote_candidate("m").unwrap().unwrap();
7526
7527        assert!(matches!(
7528            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
7529            Err(SuperviseError::RegistrationStillActive { .. })
7530        ));
7531
7532        // Still held while the incumbent's connection has not deregistered.
7533        assert!(matches!(
7534            wait_for_slot_registration_release(
7535                &registry,
7536                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7537                Duration::from_millis(50),
7538            )
7539            .await,
7540            Err(SuperviseError::RegistrationStillActive { .. })
7541        ));
7542
7543        let releaser = Arc::clone(&registry);
7544        let release = tokio::spawn(async move {
7545            sleep(Duration::from_millis(20)).await;
7546            releaser
7547                .deregister_connection(ConnectionId::new(INCUMBENT))
7548                .unwrap();
7549            notify_registration_release();
7550        });
7551        wait_for_slot_registration_release(
7552            &registry,
7553            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7554            Duration::from_secs(5),
7555        )
7556        .await
7557        .expect("the incumbent's own registration is released");
7558        release.await.unwrap();
7559        assert!(registry.get_module("m").unwrap().is_some());
7560    }
7561
7562    /// The candidate slot is waited on separately from the active slot: the
7563    /// incumbent's registration neither holds up nor stands in for it.
7564    #[tokio::test]
7565    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
7566        let registry = swapped_registry();
7567        assert!(matches!(
7568            wait_for_slot_registration_release(
7569                &registry,
7570                RegistrationSlot::Candidate("m"),
7571                Duration::from_millis(50),
7572            )
7573            .await,
7574            Err(SuperviseError::RegistrationStillActive { .. })
7575        ));
7576        registry
7577            .deregister_connection(ConnectionId::new(CANDIDATE))
7578            .unwrap();
7579        wait_for_slot_registration_release(
7580            &registry,
7581            RegistrationSlot::Candidate("m"),
7582            Duration::from_millis(50),
7583        )
7584        .await
7585        .expect("a candidate slot with no candidate is released");
7586        assert!(registry
7587            .registration(RegistrationSlot::Active("m"))
7588            .unwrap()
7589            .is_some());
7590    }
7591}
7592
7593fn classify_exit(status: &ExitStatus) -> ExitReport {
7594    ExitReport {
7595        kind: if status.success() {
7596            ExitKind::Clean
7597        } else {
7598            ExitKind::Crash
7599        },
7600        code: status.code(),
7601        signal: exit_signal(status),
7602        at_ms: unix_ms_now(),
7603    }
7604}
7605
7606/// The terminal record for a module whose `wait()` call itself errored (e.g. the
7607/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
7608/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
7609/// disposition still must be `Failed` so the terminal ring is not silently missing
7610/// an entry, matching what `fail_snapshot` records for this same arm.
7611fn wait_error_exit_report() -> ExitReport {
7612    ExitReport {
7613        kind: ExitKind::Crash,
7614        code: None,
7615        signal: None,
7616        at_ms: unix_ms_now(),
7617    }
7618}
7619
7620#[cfg(unix)]
7621fn exit_signal(status: &ExitStatus) -> Option<i32> {
7622    use std::os::unix::process::ExitStatusExt;
7623
7624    status.signal()
7625}
7626
7627#[cfg(not(unix))]
7628fn exit_signal(_status: &ExitStatus) -> Option<i32> {
7629    None
7630}
7631
7632/// Give an operator-touched module its full crash budget back.
7633///
7634/// Named for the counter it used to zero; it now empties the in-window ring,
7635/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
7636/// ledger of what happened survives every operator action.
7637fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
7638    update_snapshot(snapshot, Some(module_id), |state| {
7639        state.clear_crash_restarts();
7640    })
7641}
7642
7643fn set_running(
7644    snapshot: &SharedSnapshot,
7645    child: &SupervisedChild,
7646    module_id: &str,
7647    spawn_events: &SpawnEventFeed,
7648) -> Result<(), SuperviseError> {
7649    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7650        module_id: Some(module_id.to_string()),
7651    })?;
7652    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
7653    // Every caller of this is a plain spawn, which always uses the primary key;
7654    // a promoted swap candidate sets the flag itself after this returns.
7655    state.in_alternate_slot = false;
7656    state.configuration_updated_since_spawn = false;
7657    state.state = ModuleState::Running;
7658    state.enabled = true;
7659    state.process_alive = true;
7660    state.pid = child.id();
7661    state.spawned_at_ms = Some(child.spawned_at_ms);
7662    state.spawned_from = Some(child.spawned_from.clone());
7663    state.spawned_file_identity = child.spawned_file_identity;
7664    state.process_start_time = child.process_start_time;
7665    Ok(())
7666}
7667
7668fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
7669    state.process_alive = false;
7670    state.pid = None;
7671    state.spawned_at_ms = None;
7672    state.spawned_from = None;
7673    state.spawned_file_identity = None;
7674    state.process_start_time = None;
7675    state.deliberate_severance = None;
7676}
7677
7678#[cfg(test)]
7679fn record_deliberate_severance(
7680    snapshot: &SharedSnapshot,
7681    identity: ProcessIdentity,
7682) -> Result<(), SuperviseError> {
7683    update_snapshot(snapshot, None, |state| {
7684        state.deliberate_severance = Some(identity);
7685    })
7686}
7687
7688fn apply_deliberate_severance_marker(
7689    snapshot: &SharedSnapshot,
7690    exited_identity: Option<ProcessIdentity>,
7691    mut exit_report: ExitReport,
7692) -> ExitReport {
7693    let marker = lock_snapshot(snapshot)
7694        .ok()
7695        .and_then(|mut state| state.deliberate_severance.take());
7696    if marker.is_some() && marker == exited_identity {
7697        exit_report.kind = ExitKind::DeliberateSeverance;
7698    }
7699    exit_report
7700}
7701
7702fn classify_reaped_child_exit(
7703    snapshot: &SharedSnapshot,
7704    child: &SupervisedChild,
7705    status: &ExitStatus,
7706) -> ExitReport {
7707    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
7708}
7709
7710fn fail_snapshot(
7711    snapshot: &SharedSnapshot,
7712    module_id: Option<&str>,
7713    last_exit: Option<ExitReport>,
7714) {
7715    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
7716        state.state = ModuleState::Failed;
7717        clear_current_process_facts(state);
7718        if let Some(last_exit) = last_exit {
7719            state.last_exit = Some(last_exit);
7720        }
7721    }) {
7722        error!(error = %err, "failed to mark supervisor state failed");
7723    }
7724}
7725
7726fn update_snapshot(
7727    snapshot: &SharedSnapshot,
7728    module_id: Option<&str>,
7729    update: impl FnOnce(&mut SupervisorSnapshot),
7730) -> Result<(), SuperviseError> {
7731    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7732        module_id: module_id.map(ToOwned::to_owned),
7733    })?;
7734    update(&mut state);
7735    Ok(())
7736}
7737
7738const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
7739
7740fn lock_snapshot_for_control<'a>(
7741    snapshot: &'a SharedSnapshot,
7742    module_id: &str,
7743    caller: &'static str,
7744) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
7745    let started_at = Instant::now();
7746    let guard = lock_snapshot(snapshot)?;
7747    let waited = started_at.elapsed();
7748    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
7749        warn!(
7750            module_id = %module_id,
7751            waited_ms = waited.as_millis() as u64,
7752            caller = %caller,
7753            "slow snapshot lock"
7754        );
7755    }
7756    Ok(guard)
7757}
7758
7759fn lock_snapshot(
7760    snapshot: &SharedSnapshot,
7761) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
7762    snapshot
7763        .lock()
7764        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
7765}
7766
7767#[cfg(test)]
7768mod terminal_history_tests {
7769    use std::{
7770        path::PathBuf,
7771        sync::Arc,
7772        time::{Duration, Instant},
7773    };
7774
7775    use tokio::time::sleep;
7776
7777    use super::{
7778        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
7779        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
7780        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
7781        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
7782        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
7783        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
7784        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
7785    };
7786    // The supervisor's clock, distinct from the `std::time::Instant` these tests
7787    // use for their own wall-clock deadlines: crash-restart instants must be on
7788    // the same clock the production code stamps them with, which is tokio's (and
7789    // is what `start_paused` tests can move).
7790    use super::Instant as ClockInstant;
7791    use crate::{
7792        registry::Registry,
7793        terminal_ring::{TerminalRing, TerminalRingConfig},
7794    };
7795    use std::sync::Mutex;
7796    use subc_control::TerminalDisposition;
7797
7798    /// See the twin in `control.rs` for why this derives the path from
7799    /// `current_exe()` and why the existence check is here: `--lib` alone does
7800    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
7801    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
7802    fn fake_aft_stub_path() -> PathBuf {
7803        let mut path = std::env::current_exe().expect("current_exe available in tests");
7804        path.pop();
7805        path.pop();
7806        path.push(if cfg!(windows) {
7807            "fake-aft-stub.exe"
7808        } else {
7809            "fake-aft-stub"
7810        });
7811        assert!(
7812            path.exists(),
7813            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
7814             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
7815            path.display()
7816        );
7817        path
7818    }
7819
7820    #[test]
7821    fn reserved_never_spawned_refuses_every_hello() {
7822        // The canary hole: a reserved id whose module has never spawned had NO
7823        // gate entry and admitted anyone -- the reservation protected the nonce
7824        // holder, not the NAME. Now the entry is present with no legitimate
7825        // holder and refuses all comers.
7826        let supervisor = SupervisorHandle::default();
7827        supervisor.apply_identity_configuration(&ModuleSpec {
7828            launch_nonce_env: true,
7829            module_id: "never-spawned".to_string(),
7830            program: PathBuf::from("/usr/bin/false"),
7831            args: Vec::new(),
7832            env: Vec::new(),
7833            reserved: true,
7834            reserved_prefixes: Vec::new(),
7835            protocol: ModuleProtocol::Subc,
7836            overlap: Default::default(),
7837        });
7838        assert!(
7839            supervisor
7840                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
7841                .is_some(),
7842            "forged nonce must refuse on a reserved never-spawned id"
7843        );
7844        assert!(
7845            supervisor
7846                .reserved_hello_rejection("never-spawned", None)
7847                .is_some(),
7848            "absent nonce must refuse on a reserved never-spawned id"
7849        );
7850        // And a real spawn nonce minted later admits exactly that nonce.
7851        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
7852        supervisor.apply_identity_configuration(&ModuleSpec {
7853            launch_nonce_env: true,
7854            module_id: "never-spawned".to_string(),
7855            program: PathBuf::from("/usr/bin/false"),
7856            args: Vec::new(),
7857            env: Vec::new(),
7858            reserved: true,
7859            reserved_prefixes: Vec::new(),
7860            protocol: ModuleProtocol::Subc,
7861            overlap: Default::default(),
7862        });
7863        assert!(supervisor
7864            .reserved_hello_rejection("never-spawned", Some("minted"))
7865            .is_none());
7866        assert!(supervisor
7867            .reserved_hello_rejection("never-spawned", Some("forged"))
7868            .is_some());
7869    }
7870
7871    /// Put `count` crash restarts on a snapshot's ring as if they had all just
7872    /// happened, which is what "spent budget" looks like to every reader.
7873    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
7874        let now = ClockInstant::now();
7875        for _ in 0..count {
7876            state.crash_restarts.push_back(now);
7877        }
7878    }
7879
7880    /// Age the oldest recorded restart out of `window`, standing in for the hours
7881    /// that would otherwise have to pass. Injecting the instant is the point: a
7882    /// test that slept a real window would take ten minutes and still prove less.
7883    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
7884        let aged = state
7885            .crash_restarts
7886            .front()
7887            .expect("a crash restart must be recorded before it can be aged")
7888            .checked_sub(window + Duration::from_secs(1))
7889            .expect("the test clock is far enough from its origin to age an instant");
7890        state.crash_restarts[0] = aged;
7891    }
7892
7893    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
7894        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
7895        seed_crash_restarts(&mut state, count);
7896        state
7897    }
7898
7899    #[test]
7900    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
7901        let policy = RestartPolicy::new(3, Duration::ZERO);
7902        let now = ClockInstant::now();
7903        assert!(daemon_will_restart(
7904            &mut snapshot_with_restarts(true, 2),
7905            &policy,
7906            now
7907        ));
7908        assert!(!daemon_will_restart(
7909            &mut snapshot_with_restarts(true, 3),
7910            &policy,
7911            now
7912        ));
7913        assert!(!daemon_will_restart(
7914            &mut snapshot_with_restarts(false, 0),
7915            &policy,
7916            now
7917        ));
7918    }
7919
7920    #[test]
7921    fn crash_restart_backoff_escalates_with_in_window_count() {
7922        let policy = RestartPolicy::new(4, Duration::from_millis(100))
7923            .with_max_backoff(Duration::from_secs(30));
7924        let now = ClockInstant::now();
7925        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7926        let schedules = (0..4)
7927            .map(|_| {
7928                state
7929                    .next_crash_restart(&policy, now)
7930                    .expect("the test policy allows four crash restarts")
7931            })
7932            .collect::<Vec<_>>();
7933
7934        assert_eq!(
7935            schedules
7936                .iter()
7937                .map(|schedule| schedule.restart_in_window)
7938                .collect::<Vec<_>>(),
7939            vec![0, 1, 2, 3]
7940        );
7941        assert_eq!(
7942            schedules
7943                .iter()
7944                .map(|schedule| schedule.delay)
7945                .collect::<Vec<_>>(),
7946            vec![
7947                Duration::from_millis(100),
7948                Duration::from_secs(1),
7949                Duration::from_secs(10),
7950                Duration::from_secs(30),
7951            ]
7952        );
7953    }
7954
7955    #[test]
7956    fn crash_restart_backoff_resets_after_ring_clear() {
7957        let policy = RestartPolicy::new(3, Duration::from_millis(100));
7958        let now = ClockInstant::now();
7959        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7960        assert_eq!(
7961            state.next_crash_restart(&policy, now).unwrap().delay,
7962            Duration::from_millis(100)
7963        );
7964        assert_eq!(
7965            state.next_crash_restart(&policy, now).unwrap().delay,
7966            Duration::from_secs(1)
7967        );
7968
7969        state.clear_crash_restarts();
7970        let schedule = state
7971            .next_crash_restart(&policy, now)
7972            .expect("a cleared ring must allow another restart");
7973        assert_eq!(schedule.restart_in_window, 0);
7974        assert_eq!(schedule.delay, Duration::from_millis(100));
7975    }
7976
7977    #[test]
7978    fn crash_restart_backoff_ignores_aged_restarts() {
7979        let policy = RestartPolicy::new(3, Duration::from_millis(100));
7980        let now = ClockInstant::now();
7981        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7982        state
7983            .next_crash_restart(&policy, now)
7984            .expect("the first restart is allowed");
7985        state
7986            .next_crash_restart(&policy, now)
7987            .expect("the second restart is allowed");
7988        state.crash_restarts[0] = now
7989            .checked_sub(policy.window + Duration::from_secs(1))
7990            .expect("the fake clock can age a restart past the window");
7991
7992        let schedule = state
7993            .next_crash_restart(&policy, now)
7994            .expect("an aged restart must release its slot");
7995        assert_eq!(schedule.restart_in_window, 1);
7996        assert_eq!(schedule.delay, Duration::from_secs(1));
7997        assert_eq!(state.crash_restarts.len(), 2);
7998    }
7999
8000    /// The budget is a rate: the same three spent restarts refuse a respawn
8001    /// while they are recent and allow one once they have aged past the window.
8002    /// Nothing about the module changed in between, which is the whole point.
8003    #[test]
8004    fn a_budget_spent_before_the_window_no_longer_refuses() {
8005        let policy = RestartPolicy::new(3, Duration::ZERO);
8006        let mut state = snapshot_with_restarts(true, 3);
8007        let now = ClockInstant::now();
8008        assert!(!daemon_will_restart(&mut state, &policy, now));
8009
8010        assert!(daemon_will_restart(
8011            &mut state,
8012            &policy,
8013            now + policy.window + Duration::from_secs(1)
8014        ));
8015        assert!(
8016            state.crash_restarts.is_empty(),
8017            "reading the budget must drop the instants that left the window"
8018        );
8019    }
8020
8021    fn module_with_recovery_snapshot(
8022        state: ModuleState,
8023        enabled: bool,
8024        restart_count: u32,
8025    ) -> SupervisedModule {
8026        let registry = Arc::new(Registry::default());
8027        let supervisor =
8028            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
8029        let module = supervisor
8030            .spawn(ModuleSpec {
8031                launch_nonce_env: true,
8032                module_id: "recovery-snapshot".to_string(),
8033                program: fake_aft_stub_path(),
8034                args: Vec::new(),
8035                env: Vec::new(),
8036                reserved: false,
8037                reserved_prefixes: Vec::new(),
8038                protocol: ModuleProtocol::Subc,
8039                overlap: Default::default(),
8040            })
8041            .unwrap();
8042        update_snapshot(
8043            &module.inner.snapshot,
8044            Some("recovery-snapshot"),
8045            |snapshot| {
8046                snapshot.state = state;
8047                snapshot.enabled = enabled;
8048                seed_crash_restarts(snapshot, restart_count);
8049            },
8050        )
8051        .unwrap();
8052        module
8053    }
8054
8055    #[cfg(target_os = "linux")]
8056    #[tokio::test]
8057    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
8058        let supervisor = Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
8059            .with_cgroup_placement(None);
8060        let result = supervisor.spawn(ModuleSpec {
8061            launch_nonce_env: true,
8062            module_id: "no-cgroup-placement".to_string(),
8063            program: fake_aft_stub_path(),
8064            args: Vec::new(),
8065            env: Vec::new(),
8066            reserved: false,
8067            reserved_prefixes: Vec::new(),
8068            protocol: ModuleProtocol::Subc,
8069            overlap: Default::default(),
8070        });
8071
8072        assert!(
8073            result.is_ok(),
8074            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
8075        );
8076    }
8077
8078    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8079    async fn undecided_snapshot_uses_shared_restart_predicate() {
8080        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
8081            .will_recover_after_connection_loss()
8082            .unwrap());
8083        assert!(
8084            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
8085                .will_recover_after_connection_loss()
8086                .unwrap()
8087        );
8088    }
8089
8090    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8091    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
8092        assert!(
8093            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
8094                .will_recover_after_connection_loss()
8095                .unwrap()
8096        );
8097    }
8098
8099    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8100    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
8101        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
8102            .will_recover_after_connection_loss()
8103            .unwrap());
8104        assert!(
8105            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
8106                .will_recover_after_connection_loss()
8107                .unwrap()
8108        );
8109    }
8110
8111    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8112    async fn warming_snapshot_is_limited_to_startup_phases() {
8113        for state in [
8114            ModuleState::Starting,
8115            ModuleState::Running,
8116            ModuleState::Restarting,
8117        ] {
8118            assert!(
8119                module_with_recovery_snapshot(state, true, 0)
8120                    .is_warming()
8121                    .unwrap(),
8122                "{state:?} should be warming"
8123            );
8124        }
8125        for state in [
8126            ModuleState::Unresponsive,
8127            ModuleState::Draining,
8128            ModuleState::Stopped,
8129            ModuleState::Failed,
8130            ModuleState::Disabled,
8131        ] {
8132            assert!(
8133                !module_with_recovery_snapshot(state, true, 0)
8134                    .is_warming()
8135                    .unwrap(),
8136                "{state:?} should not be warming"
8137            );
8138        }
8139    }
8140
8141    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8142    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
8143        let registry = Arc::new(Registry::default());
8144        let supervisor =
8145            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
8146        let module = supervisor
8147            .spawn(ModuleSpec {
8148                launch_nonce_env: true,
8149                module_id: "terminal-history".to_string(),
8150                program: fake_aft_stub_path(),
8151                args: Vec::new(),
8152                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8153                reserved: false,
8154                reserved_prefixes: Vec::new(),
8155                protocol: ModuleProtocol::Subc,
8156                overlap: Default::default(),
8157            })
8158            .unwrap();
8159
8160        let deadline = Instant::now() + Duration::from_secs(5);
8161        loop {
8162            let history = module.terminal_history();
8163            if history.entries.len() == 2 {
8164                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
8165                assert_eq!(history.dropped, 0);
8166                assert_eq!(
8167                    history
8168                        .entries
8169                        .iter()
8170                        .map(|entry| entry.exit_code)
8171                        .collect::<Vec<_>>(),
8172                    vec![Some(23), Some(23)]
8173                );
8174                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
8175                return;
8176            }
8177            assert!(
8178                Instant::now() < deadline,
8179                "module did not retain two terminal exits: {history:?}"
8180            );
8181            sleep(Duration::from_millis(10)).await;
8182        }
8183    }
8184
8185    /// A disable issued while a crash respawn is still backing off must preempt
8186    /// that respawn: the operator's stop wins, the disable must not queue behind
8187    /// the backoff, and the module must never come back up afterwards.
8188    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8189    async fn disable_during_crash_backoff_cancels_pending_respawn() {
8190        let backoff = Duration::from_secs(2);
8191        let supervisor = Supervisor::new(
8192            Arc::new(Registry::default()),
8193            RestartPolicy::new(10, backoff),
8194        );
8195        let module = supervisor
8196            .spawn(ModuleSpec {
8197                launch_nonce_env: true,
8198                module_id: "disable-during-backoff".to_string(),
8199                program: fake_aft_stub_path(),
8200                args: Vec::new(),
8201                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8202                reserved: false,
8203                reserved_prefixes: Vec::new(),
8204                protocol: ModuleProtocol::Subc,
8205                overlap: Default::default(),
8206            })
8207            .unwrap();
8208
8209        // Wait for the first crash to put the module into its backoff window.
8210        let deadline = Instant::now() + Duration::from_secs(5);
8211        loop {
8212            if module.status().unwrap().state == ModuleState::Restarting {
8213                break;
8214            }
8215            assert!(
8216                Instant::now() < deadline,
8217                "module never entered the crash backoff"
8218            );
8219            sleep(Duration::from_millis(10)).await;
8220        }
8221
8222        let started = Instant::now();
8223        module.set_enabled(false).await.unwrap();
8224        let waited = started.elapsed();
8225
8226        assert!(
8227            waited < backoff / 2,
8228            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
8229        );
8230        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
8231
8232        // Outlast the backoff: the respawn it was counting down to must never run.
8233        sleep(backoff + Duration::from_millis(500)).await;
8234        let status = module.status().unwrap();
8235        assert_eq!(status.state, ModuleState::Disabled);
8236        assert_eq!(
8237            status.spawn_generation, 1,
8238            "module respawned after the operator disabled it"
8239        );
8240    }
8241
8242    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
8243    /// the shape of nats-server, the program this rule exists for.
8244    #[cfg(unix)]
8245    fn protocol_none_sigterm_exits_clean_spec(
8246        module_id: &str,
8247        dir: &std::path::Path,
8248    ) -> (ModuleSpec, PathBuf, PathBuf) {
8249        let ready = dir.join("ready");
8250        let marker = dir.join("sigterm");
8251        let spec = ModuleSpec {
8252            launch_nonce_env: true,
8253            module_id: module_id.to_string(),
8254            program: fake_aft_stub_path(),
8255            args: Vec::new(),
8256            env: vec![
8257                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
8258                (
8259                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
8260                    marker.display().to_string(),
8261                ),
8262                (
8263                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
8264                    ready.display().to_string(),
8265                ),
8266            ],
8267            reserved: false,
8268            reserved_prefixes: Vec::new(),
8269            protocol: ModuleProtocol::None,
8270            overlap: Default::default(),
8271        };
8272        (spec, ready, marker)
8273    }
8274
8275    /// Wait for a file the child writes, so a signal is never sent before the
8276    /// child's SIGTERM handler is installed (the default disposition would
8277    /// kill it by signal and the exit would not be clean).
8278    #[cfg(unix)]
8279    async fn wait_for_file(path: &std::path::Path) {
8280        let deadline = Instant::now() + Duration::from_secs(10);
8281        while !path.exists() {
8282            assert!(
8283                Instant::now() < deadline,
8284                "{} never appeared",
8285                path.display()
8286            );
8287            sleep(Duration::from_millis(10)).await;
8288        }
8289    }
8290
8291    /// A protocol-none module that exits 0 because something OUTSIDE the
8292    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
8293    /// the crash-path disposition rather than `stopped`.
8294    #[cfg(unix)]
8295    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8296    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
8297        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
8298        let (spec, ready, marker) =
8299            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
8300        let supervisor = Supervisor::new(
8301            Arc::new(Registry::default()),
8302            RestartPolicy::new(3, Duration::ZERO),
8303        );
8304        let module = supervisor.spawn(spec).unwrap();
8305        wait_for_file(&ready).await;
8306        let first_pid = module
8307            .status()
8308            .unwrap()
8309            .pid
8310            .expect("a running module reports its pid");
8311
8312        rustix::process::kill_process(
8313            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
8314            rustix::process::Signal::TERM,
8315        )
8316        .unwrap();
8317
8318        let deadline = Instant::now() + Duration::from_secs(10);
8319        let respawned = loop {
8320            let status = module.status().unwrap();
8321            if status.state == ModuleState::Running
8322                && status.pid.is_some_and(|pid| pid != first_pid)
8323            {
8324                break status;
8325            }
8326            assert!(
8327                Instant::now() < deadline,
8328                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
8329            );
8330            sleep(Duration::from_millis(10)).await;
8331        };
8332        assert_eq!(respawned.spawn_generation, 2);
8333        assert!(
8334            marker.exists(),
8335            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
8336        );
8337
8338        let history = module.terminal_history();
8339        assert_eq!(history.entries.len(), 1, "{history:?}");
8340        let entry = &history.entries[0];
8341        assert_eq!(entry.exit_code, Some(0));
8342        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
8343        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
8344
8345        module.stop().await.unwrap();
8346    }
8347
8348    /// Repeated unrequested clean exits of a protocol-none module spend the
8349    /// restart budget exactly as crashes do, and the module ends `failed` with
8350    /// the budget named.
8351    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8352    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
8353        let supervisor = Supervisor::new(
8354            Arc::new(Registry::default()),
8355            RestartPolicy::new(1, Duration::ZERO),
8356        );
8357        let module = supervisor
8358            .spawn(ModuleSpec {
8359                launch_nonce_env: true,
8360                module_id: "none-clean-exit-budget".to_string(),
8361                program: fake_aft_stub_path(),
8362                args: Vec::new(),
8363                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8364                reserved: false,
8365                reserved_prefixes: Vec::new(),
8366                protocol: ModuleProtocol::None,
8367                overlap: Default::default(),
8368            })
8369            .unwrap();
8370
8371        let deadline = Instant::now() + Duration::from_secs(10);
8372        loop {
8373            let status = module.status().unwrap();
8374            if status.state == ModuleState::Failed {
8375                break;
8376            }
8377            assert!(
8378                Instant::now() < deadline,
8379                "module never exhausted its budget: {status:?} {:?}",
8380                module.terminal_history()
8381            );
8382            sleep(Duration::from_millis(10)).await;
8383        }
8384        let history = module.terminal_history();
8385        assert_eq!(
8386            history
8387                .entries
8388                .iter()
8389                .map(|entry| (entry.exit_code, entry.disposition.clone()))
8390                .collect::<Vec<_>>(),
8391            vec![
8392                (Some(0), TerminalDisposition::Restarting),
8393                (Some(0), TerminalDisposition::Failed),
8394            ]
8395        );
8396        let detail = history.entries[1]
8397            .disposition_detail
8398            .as_deref()
8399            .expect("a budget failure names the budget");
8400        assert!(detail.contains("max_restarts=1"), "{detail}");
8401        assert_eq!(module.status().unwrap().spawn_generation, 2);
8402    }
8403
8404    /// A stop the supervisor itself requests still stops a protocol-none
8405    /// module, even though the child answers the SIGTERM with exit 0.
8406    #[cfg(unix)]
8407    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8408    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
8409        for disable in [false, true] {
8410            let label = if disable {
8411                "none-requested-disable"
8412            } else {
8413                "none-requested-stop"
8414            };
8415            let dir = subc_test_support::TestTempDir::new(label);
8416            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
8417            let supervisor = Supervisor::new(
8418                Arc::new(Registry::default()),
8419                RestartPolicy::new(3, Duration::ZERO),
8420            );
8421            let module = supervisor.spawn(spec).unwrap();
8422            wait_for_file(&ready).await;
8423
8424            if disable {
8425                module.set_enabled(false).await.unwrap();
8426            } else {
8427                module.stop().await.unwrap();
8428            }
8429            assert!(
8430                marker.exists(),
8431                "{label}: the child must have left through its SIGTERM handler with exit 0"
8432            );
8433
8434            // Long enough for a zero-backoff respawn to have happened if the
8435            // exit had been treated as a crash.
8436            sleep(Duration::from_millis(500)).await;
8437            let status = module.status().unwrap();
8438            let expected = if disable {
8439                ModuleState::Disabled
8440            } else {
8441                ModuleState::Stopped
8442            };
8443            assert_eq!(status.state, expected, "{label}");
8444            assert_eq!(
8445                status.spawn_generation, 1,
8446                "{label}: respawned after a requested stop"
8447            );
8448            let history = module.terminal_history();
8449            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
8450            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
8451            assert_ne!(
8452                history.entries[0].disposition,
8453                TerminalDisposition::Restarting,
8454                "{label}"
8455            );
8456        }
8457    }
8458
8459    /// A subc-wire module that exits 0 on its own is still a stop: the
8460    /// protocol-none rule must not reach it.
8461    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8462    async fn subc_wire_clean_exit_is_still_a_stop() {
8463        let supervisor = Supervisor::new(
8464            Arc::new(Registry::default()),
8465            RestartPolicy::new(3, Duration::ZERO),
8466        );
8467        let module = supervisor
8468            .spawn(ModuleSpec {
8469                launch_nonce_env: true,
8470                module_id: "wire-clean-exit".to_string(),
8471                program: fake_aft_stub_path(),
8472                args: Vec::new(),
8473                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8474                reserved: false,
8475                reserved_prefixes: Vec::new(),
8476                protocol: ModuleProtocol::Subc,
8477                overlap: Default::default(),
8478            })
8479            .unwrap();
8480
8481        let deadline = Instant::now() + Duration::from_secs(10);
8482        while module.terminal_history().entries.is_empty() {
8483            assert!(Instant::now() < deadline, "module never exited");
8484            sleep(Duration::from_millis(10)).await;
8485        }
8486        // Long enough for a zero-backoff respawn to have happened.
8487        sleep(Duration::from_millis(500)).await;
8488        let status = module.status().unwrap();
8489        assert_eq!(status.state, ModuleState::Stopped);
8490        assert_eq!(status.spawn_generation, 1);
8491        let history = module.terminal_history();
8492        assert_eq!(history.entries.len(), 1, "{history:?}");
8493        assert_eq!(history.entries[0].exit_code, Some(0));
8494        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
8495    }
8496
8497    /// Each restart-producing arm has its own state transition. Keeping their
8498    /// lifetime count assertions adjacent prevents a later new arm from silently
8499    /// spending budget without recording the historical restart.
8500    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8501    async fn every_restart_increment_path_advances_lifetime_count() {
8502        let supervisor = Supervisor::new(
8503            Arc::new(Registry::default()),
8504            RestartPolicy::new(1, Duration::ZERO),
8505        );
8506        let runtime = supervisor.runtime_config();
8507        let spec = ModuleSpec {
8508            launch_nonce_env: true,
8509            module_id: "lifetime-increment-path".to_string(),
8510            program: PathBuf::from("/unused/lifetime-increment-path"),
8511            args: Vec::new(),
8512            env: Vec::new(),
8513            reserved: false,
8514            reserved_prefixes: Vec::new(),
8515            protocol: ModuleProtocol::Subc,
8516            overlap: Default::default(),
8517        };
8518
8519        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8520        assert!(matches!(
8521            on_child_exit(
8522                &spec,
8523                runtime.restart_policy,
8524                &supervisor.registry,
8525                &crash_snapshot,
8526                &runtime.terminal_ring,
8527                &runtime.spawn_events,
8528                &runtime.child_roster,
8529                ExitReport {
8530                    kind: ExitKind::Crash,
8531                    code: Some(1),
8532                    signal: None,
8533                    at_ms: 1,
8534                },
8535            )
8536            .await,
8537            NextAction::Restart { schedule: _ }
8538        ));
8539        let (crash_restarts, crash_lifetime) = {
8540            let state = lock_snapshot(&crash_snapshot).unwrap();
8541            (state.crash_restarts.len(), state.lifetime_restarts)
8542        };
8543        assert_eq!(crash_restarts, 1);
8544        assert_eq!(crash_lifetime, 1);
8545
8546        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8547        let mut health_child = None;
8548        assert!(matches!(
8549            health_restart_child(
8550                &spec,
8551                &runtime,
8552                &supervisor.registry,
8553                &supervisor.process_liveness,
8554                &health_snapshot,
8555                &mut health_child,
8556                SupervisorHealthStatus::Failing,
8557                None,
8558                2,
8559            )
8560            .await,
8561            Err(SuperviseError::Spawn { .. })
8562        ));
8563        let (health_restarts, health_lifetime) = {
8564            let state = lock_snapshot(&health_snapshot).unwrap();
8565            (state.crash_restarts.len(), state.lifetime_restarts)
8566        };
8567        assert_eq!(health_restarts, 1);
8568        assert_eq!(health_lifetime, 1);
8569
8570        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8571        let mut reload_child = None;
8572        assert!(matches!(
8573            handle_reload_spawn_failure(
8574                &spec,
8575                &runtime,
8576                &supervisor.process_liveness,
8577                &reload_snapshot,
8578                &mut reload_child,
8579                "forced reload spawn failure".to_string(),
8580            )
8581            .await,
8582            Err(SuperviseError::ReloadFailed { .. })
8583        ));
8584        let (reload_restarts, reload_lifetime) = {
8585            let state = lock_snapshot(&reload_snapshot).unwrap();
8586            (state.crash_restarts.len(), state.lifetime_restarts)
8587        };
8588        assert_eq!(reload_restarts, 1);
8589        assert_eq!(reload_lifetime, 1);
8590    }
8591
8592    #[tokio::test]
8593    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
8594        let supervisor = Supervisor::new(
8595            Arc::new(Registry::default()),
8596            RestartPolicy::new(3, Duration::ZERO),
8597        );
8598        let runtime = supervisor.runtime_config();
8599        let spec = ModuleSpec {
8600            launch_nonce_env: true,
8601            module_id: "deliberately-severed".to_string(),
8602            program: PathBuf::from("/unused/deliberately-severed"),
8603            args: Vec::new(),
8604            env: Vec::new(),
8605            reserved: false,
8606            reserved_prefixes: Vec::new(),
8607            protocol: ModuleProtocol::Subc,
8608            overlap: Default::default(),
8609        };
8610        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8611        let process = ProcessIdentity {
8612            pid: 41,
8613            start_time: 101,
8614        };
8615        record_deliberate_severance(&snapshot, process).unwrap();
8616        let exit_report = apply_deliberate_severance_marker(
8617            &snapshot,
8618            Some(process),
8619            ExitReport {
8620                kind: ExitKind::Crash,
8621                code: Some(1),
8622                signal: None,
8623                at_ms: 1,
8624            },
8625        );
8626        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
8627
8628        assert!(matches!(
8629            on_child_exit(
8630                &spec,
8631                runtime.restart_policy,
8632                &supervisor.registry,
8633                &snapshot,
8634                &runtime.terminal_ring,
8635                &runtime.spawn_events,
8636                &runtime.child_roster,
8637                exit_report,
8638            )
8639            .await,
8640            NextAction::Restart { schedule: _ }
8641        ));
8642        let state = lock_snapshot(&snapshot).unwrap();
8643        assert_eq!(state.lifetime_restarts, 1);
8644        assert_eq!(state.crash_restarts.len(), 0);
8645    }
8646
8647    #[tokio::test]
8648    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
8649        let supervisor = Supervisor::new(
8650            Arc::new(Registry::default()),
8651            RestartPolicy::new(3, Duration::ZERO),
8652        );
8653        let runtime = supervisor.runtime_config();
8654        let spec = ModuleSpec {
8655            launch_nonce_env: true,
8656            module_id: "genuine-crash".to_string(),
8657            program: PathBuf::from("/unused/genuine-crash"),
8658            args: Vec::new(),
8659            env: Vec::new(),
8660            reserved: false,
8661            reserved_prefixes: Vec::new(),
8662            protocol: ModuleProtocol::Subc,
8663            overlap: Default::default(),
8664        };
8665        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8666
8667        assert!(matches!(
8668            on_child_exit(
8669                &spec,
8670                runtime.restart_policy,
8671                &supervisor.registry,
8672                &snapshot,
8673                &runtime.terminal_ring,
8674                &runtime.spawn_events,
8675                &runtime.child_roster,
8676                ExitReport {
8677                    kind: ExitKind::Crash,
8678                    code: Some(1),
8679                    signal: None,
8680                    at_ms: 1,
8681                },
8682            )
8683            .await,
8684            NextAction::Restart { schedule: _ }
8685        ));
8686        let state = lock_snapshot(&snapshot).unwrap();
8687        assert_eq!(state.lifetime_restarts, 1);
8688        assert_eq!(state.crash_restarts.len(), 1);
8689    }
8690
8691    fn crash_exit_report(at_ms: u64) -> ExitReport {
8692        ExitReport {
8693            kind: ExitKind::Crash,
8694            code: Some(1),
8695            signal: None,
8696            at_ms,
8697        }
8698    }
8699
8700    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
8701        ModuleSpec {
8702            launch_nonce_env: true,
8703            module_id: module_id.to_string(),
8704            program: PathBuf::from("/unused").join(module_id),
8705            args: Vec::new(),
8706            env: Vec::new(),
8707            reserved: false,
8708            reserved_prefixes: Vec::new(),
8709            protocol: ModuleProtocol::Subc,
8710            overlap: Default::default(),
8711        }
8712    }
8713
8714    /// A real crash loop still stops. Three crashes with nothing aging out spend
8715    /// a budget of two and the third respawn is refused, and both surfaces an
8716    /// operator has -- the log line and the retained terminal record -- name the
8717    /// window rather than only the cap, because `max_restarts=2` alone is what
8718    /// this budget used to mean.
8719    #[tokio::test]
8720    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
8721        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
8722        let supervisor = Supervisor::new(
8723            Arc::new(Registry::default()),
8724            RestartPolicy::new(2, Duration::ZERO),
8725        );
8726        let runtime = supervisor.runtime_config();
8727        let spec = windowed_crash_spec("crash-loop-in-window");
8728        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8729
8730        for attempt in 1..=2 {
8731            assert!(
8732                matches!(
8733                    on_child_exit(
8734                        &spec,
8735                        runtime.restart_policy,
8736                        &supervisor.registry,
8737                        &snapshot,
8738                        &runtime.terminal_ring,
8739                        &runtime.spawn_events,
8740                        &runtime.child_roster,
8741                        crash_exit_report(attempt),
8742                    )
8743                    .await,
8744                    NextAction::Restart { schedule: _ }
8745                ),
8746                "crash {attempt} is inside the budget and must respawn"
8747            );
8748        }
8749
8750        assert!(matches!(
8751            on_child_exit(
8752                &spec,
8753                runtime.restart_policy,
8754                &supervisor.registry,
8755                &snapshot,
8756                &runtime.terminal_ring,
8757                &runtime.spawn_events,
8758                &runtime.child_roster,
8759                crash_exit_report(3),
8760            )
8761            .await,
8762            NextAction::Stop { .. }
8763        ));
8764
8765        {
8766            let state = lock_snapshot(&snapshot).unwrap();
8767            assert_eq!(state.state, ModuleState::Failed);
8768            assert_eq!(state.crash_restarts.len(), 2);
8769            assert_eq!(state.lifetime_restarts, 2);
8770        }
8771
8772        let history = runtime
8773            .terminal_ring
8774            .lock()
8775            .expect("terminal ring is not poisoned")
8776            .snapshot();
8777        let last = history
8778            .entries
8779            .last()
8780            .expect("the refused crash is retained");
8781        assert_eq!(last.disposition, TerminalDisposition::Failed);
8782        assert_eq!(
8783            last.disposition_detail.as_deref(),
8784            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
8785        );
8786
8787        let captured = crate::router::test_log::captured_logs(&logs);
8788        assert!(
8789            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
8790            "the stop must be logged with its window: {captured}"
8791        );
8792    }
8793
8794    /// The rate, stated as a test: three crashes where the first has aged past
8795    /// the window are two crashes as far as the budget is concerned, so the
8796    /// third respawn is allowed and the ring holds only the two recent ones.
8797    ///
8798    /// This is the case a lifetime counter got wrong -- and the case the daemon
8799    /// now hits routinely, since a module exits non-zero every time its
8800    /// connection to the daemon drops.
8801    #[tokio::test]
8802    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
8803        let supervisor = Supervisor::new(
8804            Arc::new(Registry::default()),
8805            RestartPolicy::new(2, Duration::ZERO),
8806        );
8807        let runtime = supervisor.runtime_config();
8808        let spec = windowed_crash_spec("crash-across-windows");
8809        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8810
8811        for attempt in 1..=2 {
8812            assert!(matches!(
8813                on_child_exit(
8814                    &spec,
8815                    runtime.restart_policy,
8816                    &supervisor.registry,
8817                    &snapshot,
8818                    &runtime.terminal_ring,
8819                    &runtime.spawn_events,
8820                    &runtime.child_roster,
8821                    crash_exit_report(attempt),
8822                )
8823                .await,
8824                NextAction::Restart { schedule: _ }
8825            ));
8826        }
8827
8828        // The oldest crash moves out of the window; nothing else about the
8829        // module changes.
8830        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
8831            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
8832        })
8833        .unwrap();
8834
8835        assert!(
8836            matches!(
8837                on_child_exit(
8838                    &spec,
8839                    runtime.restart_policy,
8840                    &supervisor.registry,
8841                    &snapshot,
8842                    &runtime.terminal_ring,
8843                    &runtime.spawn_events,
8844                    &runtime.child_roster,
8845                    crash_exit_report(3),
8846                )
8847                .await,
8848                NextAction::Restart { schedule: _ }
8849            ),
8850            "a crash older than the window must not hold a budget slot"
8851        );
8852
8853        let state = lock_snapshot(&snapshot).unwrap();
8854        assert_eq!(state.state, ModuleState::Restarting);
8855        assert_eq!(
8856            state.crash_restarts.len(),
8857            2,
8858            "the aged instant is dropped and the new one takes its place"
8859        );
8860        assert_eq!(
8861            state.lifetime_restarts, 3,
8862            "the ledger counts every restart, including the ones the window forgot"
8863        );
8864    }
8865
8866    /// An operator restart hands the budget back whole, and the ledger keeps
8867    /// counting. Those are different questions -- "how close is this module to
8868    /// being stopped" and "how many times has it been replaced" -- and the
8869    /// operator action answers only the first.
8870    #[tokio::test]
8871    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
8872        let supervisor = Supervisor::new(
8873            Arc::new(Registry::default()),
8874            RestartPolicy::new(2, Duration::ZERO),
8875        );
8876        let runtime = supervisor.runtime_config();
8877        let spec = windowed_crash_spec("operator-cleared-budget");
8878        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8879
8880        for attempt in 1..=2 {
8881            assert!(matches!(
8882                on_child_exit(
8883                    &spec,
8884                    runtime.restart_policy,
8885                    &supervisor.registry,
8886                    &snapshot,
8887                    &runtime.terminal_ring,
8888                    &runtime.spawn_events,
8889                    &runtime.child_roster,
8890                    crash_exit_report(attempt),
8891                )
8892                .await,
8893                NextAction::Restart { schedule: _ }
8894            ));
8895        }
8896
8897        reset_restart_count(&snapshot, &spec.module_id).unwrap();
8898        {
8899            let state = lock_snapshot(&snapshot).unwrap();
8900            assert!(
8901                state.crash_restarts.is_empty(),
8902                "an operator restart returns the full budget"
8903            );
8904            assert_eq!(
8905                state.lifetime_restarts, 2,
8906                "clearing the budget must not unmake the crashes"
8907            );
8908        }
8909
8910        assert!(
8911            matches!(
8912                on_child_exit(
8913                    &spec,
8914                    runtime.restart_policy,
8915                    &supervisor.registry,
8916                    &snapshot,
8917                    &runtime.terminal_ring,
8918                    &runtime.spawn_events,
8919                    &runtime.child_roster,
8920                    crash_exit_report(3),
8921                )
8922                .await,
8923                NextAction::Restart { schedule: _ }
8924            ),
8925            "the cleared budget must be spendable again"
8926        );
8927        let state = lock_snapshot(&snapshot).unwrap();
8928        assert_eq!(state.crash_restarts.len(), 1);
8929        assert_eq!(state.lifetime_restarts, 3);
8930    }
8931
8932    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8933    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
8934        let severed = ProcessIdentity {
8935            pid: 41,
8936            start_time: 101,
8937        };
8938        let successor = ProcessIdentity {
8939            pid: 41,
8940            start_time: 202,
8941        };
8942        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
8943        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
8944            state.pid = Some(successor.pid);
8945            state.process_start_time = Some(successor.start_time);
8946        })
8947        .unwrap();
8948        assert!(!module.record_deliberate_severance(severed).unwrap());
8949
8950        let exit_report = apply_deliberate_severance_marker(
8951            &module.inner.snapshot,
8952            Some(successor),
8953            ExitReport {
8954                kind: ExitKind::Crash,
8955                code: Some(1),
8956                signal: None,
8957                at_ms: 1,
8958            },
8959        );
8960
8961        assert_eq!(exit_report.kind, ExitKind::Crash);
8962    }
8963
8964    #[tokio::test]
8965    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
8966        let registry = Registry::default();
8967        let supervisor = Supervisor::new(
8968            Arc::new(Registry::default()),
8969            RestartPolicy::new(3, Duration::ZERO),
8970        );
8971        let runtime = supervisor.runtime_config();
8972        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8973        let spec = ModuleSpec {
8974            launch_nonce_env: true,
8975            module_id: "drain-deliberate-severance".to_string(),
8976            program: fake_aft_stub_path(),
8977            args: Vec::new(),
8978            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8979            reserved: false,
8980            reserved_prefixes: Vec::new(),
8981            protocol: ModuleProtocol::Subc,
8982            overlap: Default::default(),
8983        };
8984        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
8985        let process = ProcessIdentity {
8986            pid: 41,
8987            start_time: 101,
8988        };
8989        child.process_identity = Some(process);
8990        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
8991            state.pid = Some(process.pid);
8992            state.process_start_time = Some(process.start_time);
8993        })
8994        .unwrap();
8995        record_deliberate_severance(&snapshot, process).unwrap();
8996
8997        drain_child_to_state(
8998            &spec.module_id,
8999            spec.protocol,
9000            // The child exits on its own; no signal may change the exit this
9001            // test classifies.
9002            StopNotice::SentOverConnection,
9003            &registry,
9004            &snapshot,
9005            &runtime.terminal_ring,
9006            &runtime.spawn_events,
9007            child,
9008            Duration::from_secs(1),
9009            ModuleState::Stopped,
9010            Some(false),
9011        )
9012        .await
9013        .unwrap();
9014
9015        let state = lock_snapshot(&snapshot).unwrap();
9016        assert_eq!(
9017            state.last_exit.as_ref().map(|exit| exit.kind),
9018            Some(ExitKind::DeliberateSeverance)
9019        );
9020        assert_eq!(state.lifetime_restarts, 1);
9021        assert_eq!(state.crash_restarts.len(), 0);
9022        drop(state);
9023        let history = runtime.terminal_ring.lock().unwrap().snapshot();
9024        assert_eq!(
9025            history.entries[0].exit_kind,
9026            subc_control::TerminalExitKind::DeliberateSeverance
9027        );
9028    }
9029
9030    #[tokio::test]
9031    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
9032        let registry = Registry::default();
9033        let supervisor = Supervisor::new(
9034            Arc::new(Registry::default()),
9035            RestartPolicy::new(3, Duration::ZERO),
9036        );
9037        let runtime = supervisor.runtime_config();
9038        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9039        let spec = ModuleSpec {
9040            launch_nonce_env: true,
9041            module_id: "ordinary-drain".to_string(),
9042            program: fake_aft_stub_path(),
9043            args: Vec::new(),
9044            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9045            reserved: false,
9046            reserved_prefixes: Vec::new(),
9047            protocol: ModuleProtocol::Subc,
9048            overlap: Default::default(),
9049        };
9050        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9051
9052        drain_child_to_state(
9053            &spec.module_id,
9054            spec.protocol,
9055            // The child exits on its own; no signal may change the exit this
9056            // test classifies.
9057            StopNotice::SentOverConnection,
9058            &registry,
9059            &snapshot,
9060            &runtime.terminal_ring,
9061            &runtime.spawn_events,
9062            child,
9063            Duration::from_secs(1),
9064            ModuleState::Stopped,
9065            Some(false),
9066        )
9067        .await
9068        .unwrap();
9069
9070        let state = lock_snapshot(&snapshot).unwrap();
9071        assert_eq!(
9072            state.last_exit.as_ref().map(|exit| exit.kind),
9073            Some(ExitKind::Crash)
9074        );
9075        assert_eq!(state.lifetime_restarts, 0);
9076        assert_eq!(state.crash_restarts.len(), 0);
9077    }
9078
9079    #[test]
9080    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
9081        // The server's generic fatal-routing branch only knows that the
9082        // connection failed; it does not know that the daemon deliberately
9083        // initiated a process-killing severance. Keep this seam explicit so a
9084        // future connection error path cannot silently reintroduce the stale
9085        // exemption that mislabels a later genuine crash.
9086        assert!(!include_str!("server.rs")
9087            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
9088    }
9089
9090    /// The `route.closed` `drained` value must be the quiescence wait's own
9091    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
9092    /// measurement at all and `false` is the one honest constant. This is the exact
9093    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
9094    /// on every return path, including the one that used to return early via `?`
9095    /// with `route.closing` already sent and no `route.closed` ever following.
9096    #[test]
9097    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
9098        assert!(drained_after_quiescence_wait(&Ok(true)));
9099        assert!(!drained_after_quiescence_wait(&Ok(false)));
9100        assert!(!drained_after_quiescence_wait(&Err(
9101            SuperviseError::StatePoisoned { module_id: None }
9102        )));
9103    }
9104
9105    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
9106    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
9107    /// already reaped out-of-band) still leaves a terminal record rather than none
9108    /// at all. Triggering the real `wait()` I/O error from an integration test would
9109    /// need a genuine already-reaped-child race, which is OS-specific and not
9110    /// something this suite attempts elsewhere; this test instead verifies the
9111    /// record produced for that arm end-to-end through the real `TerminalRing`, and
9112    /// the call site itself is verified by inspection to sit in that exact arm.
9113    #[test]
9114    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
9115        let ring = Arc::new(Mutex::new(TerminalRing::new(
9116            TerminalRingConfig::default(),
9117            0,
9118        )));
9119        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
9120
9121        let snapshot = ring.lock().unwrap().snapshot();
9122        assert_eq!(snapshot.entries.len(), 1);
9123        let entry = &snapshot.entries[0];
9124        assert_eq!(entry.exit_code, None);
9125        assert_eq!(entry.exit_signal, None);
9126        assert_eq!(entry.disposition, TerminalDisposition::Failed);
9127    }
9128
9129    #[test]
9130    fn wait_error_exit_path_preserves_spawn_event_density() {
9131        let feed = super::SpawnEventFeed::default();
9132        feed.configure_incarnation("wait-error-density".to_string());
9133        feed.emit_spawned("wait-error", 41, 1);
9134        let ring = Arc::new(Mutex::new(TerminalRing::new(
9135            TerminalRingConfig::default(),
9136            0,
9137        )));
9138
9139        record_wait_error_terminal("wait-error", &ring, &feed);
9140        feed.emit_spawned("after-wait-error", 42, 2);
9141
9142        let state = feed.0.lock().unwrap();
9143        let sequences = state
9144            .events
9145            .iter()
9146            .map(|event| event.cursor.seq)
9147            .collect::<Vec<_>>();
9148        assert_eq!(sequences, vec![1, 2, 3]);
9149        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
9150        assert_eq!(state.events[1].exit_code, None);
9151        assert_eq!(state.events[1].exit_signal, None);
9152    }
9153
9154    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
9155    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
9156    /// not a clean exit it never actually observed.
9157    #[test]
9158    fn wait_error_exit_report_is_classified_as_a_crash() {
9159        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
9160    }
9161}
9162
9163#[cfg(test)]
9164mod health_evidence_tests {
9165    use super::{HealthProbeError, HealthProbeEvidence};
9166    use std::collections::HashSet;
9167
9168    /// The evidential asymmetry, asserted rather than described.
9169    ///
9170    /// Exactly ONE observation is proof a module cannot serve, and the one that
9171    /// fires under CPU starvation is not it. Before the split, all fifteen
9172    /// construction sites collapsed into a single String, so a timeout carried the
9173    /// same weight as a dead lane -- which is how a healthy module was restarted
9174    /// three times in one day.
9175    #[test]
9176    fn only_a_dead_lane_is_proof_of_death() {
9177        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
9178        // Three non-proof classes, each for a different reason: silence is
9179        // consistent with health, a bad answer proves the module ALIVE, and a
9180        // daemon-side fault never reached the module at all.
9181        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
9182        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
9183        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
9184    }
9185
9186    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
9187    ///
9188    /// A shared label renders two different observations identically in the line an
9189    /// operator reads after an unexplained restart -- the exact confusion this
9190    /// change removes.
9191    #[test]
9192    fn every_evidence_class_has_a_distinct_label() {
9193        let labels = [
9194            HealthProbeError::lane_dead("").label(),
9195            HealthProbeError::no_answer("").label(),
9196            HealthProbeError::bad_answer("").label(),
9197            HealthProbeError::misconfigured("").label(),
9198        ];
9199        let unique: HashSet<_> = labels.iter().collect();
9200        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
9201    }
9202
9203    /// The class is additional information, not a replacement.
9204    ///
9205    /// An operator needs both "this was silence" and the specific text saying how
9206    /// long we waited; a classification that swallowed the message would trade one
9207    /// missing distinction for another.
9208    #[test]
9209    fn classification_preserves_the_original_message() {
9210        let err = HealthProbeError::no_answer("module did not answer within 5s");
9211        assert_eq!(err.to_string(), "module did not answer within 5s");
9212        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9213    }
9214}
9215
9216#[cfg(test)]
9217mod health_tombstone_tests {
9218    use std::{path::PathBuf, sync::Arc, time::Duration};
9219
9220    use subc_protocol::{
9221        manifest::Concurrency,
9222        session::{HealthStatus, ModuleControlResponse},
9223    };
9224    use tokio::sync::mpsc;
9225
9226    use super::{
9227        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
9228        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
9229    };
9230    use crate::{
9231        control::ControlHandler,
9232        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
9233        registry::{ConnectionId, Registry},
9234        router::FrameSink,
9235    };
9236
9237    struct ProbeHarness {
9238        spec: ModuleSpec,
9239        runtime: SupervisorRuntimeConfig,
9240        forwarding: Arc<ForwardingTable>,
9241        module_connection: ConnectionId,
9242        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
9243        handler: ControlHandler,
9244        module: super::SupervisedModule,
9245    }
9246
9247    fn probe_harness() -> ProbeHarness {
9248        let registry = Arc::new(Registry::default());
9249        let forwarding = Arc::new(ForwardingTable::default());
9250        let supervisor_handle = super::SupervisorHandle::new();
9251        let health = HealthConfig {
9252            cadence: Duration::from_secs(30),
9253            deadline: Duration::from_secs(5),
9254            failure_threshold: 3,
9255            on_degraded: HealthAction::Report,
9256            on_failing: HealthAction::Report,
9257            critical: false,
9258        };
9259        let supervisor = Supervisor::new(Arc::clone(&registry), RestartPolicy::default())
9260            .with_forwarding(Arc::clone(&forwarding))
9261            .with_handle(supervisor_handle.clone())
9262            .with_health_config(health);
9263        let spec = ModuleSpec {
9264            launch_nonce_env: true,
9265            module_id: "late-health-module".to_string(),
9266            program: PathBuf::from("disabled-module"),
9267            args: Vec::new(),
9268            env: Vec::new(),
9269            reserved: false,
9270            reserved_prefixes: Vec::new(),
9271            protocol: ModuleProtocol::Subc,
9272            overlap: Default::default(),
9273        };
9274        let module = supervisor
9275            .supervise_configured(spec.clone(), false)
9276            .unwrap();
9277        let runtime = supervisor.runtime_config();
9278        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
9279            .with_supervisor(supervisor_handle);
9280        let module_connection = ConnectionId::new(700);
9281        let (module_tx, module_rx) = mpsc::channel(8);
9282        forwarding
9283            .register_module_connection(
9284                module_connection,
9285                spec.module_id.clone(),
9286                subc_protocol::PROTOCOL_VERSION,
9287                Concurrency::ModuleManaged,
9288                FrameSink::new(module_tx),
9289            )
9290            .unwrap();
9291
9292        ProbeHarness {
9293            spec,
9294            runtime,
9295            forwarding,
9296            module_connection,
9297            module_rx,
9298            handler,
9299            module,
9300        }
9301    }
9302
9303    async fn finish_after(
9304        harness: &mut ProbeHarness,
9305        stall: Duration,
9306    ) -> ModuleControlRpcCompletion {
9307        assert!(stall > harness.runtime.health.deadline);
9308        let deadline = harness.runtime.health.deadline;
9309        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9310        let answer = async {
9311            let frame = harness.module_rx.recv().await.expect("health.check frame");
9312            tokio::time::advance(deadline).await;
9313            tokio::task::yield_now().await;
9314            tokio::time::advance(stall - deadline).await;
9315            harness
9316                .forwarding
9317                .complete_module_control_rpc(
9318                    harness.module_connection,
9319                    frame.header.corr,
9320                    Some("health.check"),
9321                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
9322                        status: HealthStatus::Ok,
9323                        detail: None,
9324                        metrics: None,
9325                    }),
9326                )
9327                .unwrap()
9328        };
9329        let (probe_result, completion) = tokio::join!(probe, answer);
9330        let err = probe_result.expect_err("probe must miss its deadline");
9331        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9332        completion
9333    }
9334
9335    async fn time_out_without_answer(harness: &mut ProbeHarness) {
9336        let deadline = harness.runtime.health.deadline;
9337        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9338        let exhaust_deadline = async {
9339            let _frame = harness.module_rx.recv().await.expect("health.check frame");
9340            tokio::time::advance(deadline).await;
9341            tokio::task::yield_now().await;
9342        };
9343        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
9344        let err = probe_result.expect_err("probe must miss its deadline");
9345        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9346    }
9347
9348    #[tokio::test(start_paused = true)]
9349    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
9350        let mut harness = probe_harness();
9351
9352        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
9353        let first_latency = match &first {
9354            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9355            other => panic!("late answer was not retained: {other:?}"),
9356        };
9357        assert!(harness.handler.observe_module_control_completion(first));
9358
9359        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
9360        let second_latency = match &second {
9361            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9362            other => panic!("late answer was not retained: {other:?}"),
9363        };
9364        assert!(harness.handler.observe_module_control_completion(second));
9365
9366        assert_eq!(first_latency, Duration::from_secs(8));
9367        assert_eq!(
9368            second_latency - first_latency,
9369            Duration::from_secs(3),
9370            "latency must grow linearly with the additional stall"
9371        );
9372        let health = harness.module.status().unwrap().health;
9373        assert_eq!(health.late_answer_count, 2);
9374        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
9375    }
9376
9377    /// A module that answers every probe late must never march to the kill
9378    /// threshold: the late answer proves it is alive, so it must clear the miss
9379    /// streak the timeout recorded. Without the reset, a CPU-starved module
9380    /// that serves every probe seconds past the deadline accumulates
9381    /// `consecutive_failures` to the threshold and is killed — the exact
9382    /// sequence from the 2026-08-14 aft disable, where the daemon logged
9383    /// "proves the module is alive" five times while counting five misses.
9384    #[tokio::test(start_paused = true)]
9385    async fn late_answer_clears_the_consecutive_failure_streak() {
9386        let mut harness = probe_harness();
9387
9388        // Timeout recorded first: the probe path saw no answer in time.
9389        time_out_without_answer(&mut harness).await;
9390        harness
9391            .module
9392            .record_health_probe_failure_for_test("[no-answer] test miss")
9393            .unwrap();
9394        assert_eq!(
9395            harness.module.status().unwrap().health.consecutive_failures,
9396            1,
9397            "precondition: the miss must be on the streak before the late answer"
9398        );
9399
9400        // The stalled reply then lands: proof of life.
9401        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
9402        assert!(matches!(
9403            late,
9404            ModuleControlRpcCompletion::LateHealthAnswer { .. }
9405        ));
9406        assert!(harness.handler.observe_module_control_completion(late));
9407
9408        let health = harness.module.status().unwrap().health;
9409        assert_eq!(
9410            health.consecutive_failures, 0,
9411            "a late answer is an answer: the streak must reset"
9412        );
9413        assert_eq!(health.late_answer_count, 1);
9414    }
9415
9416    #[tokio::test(start_paused = true)]
9417    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
9418        let mut harness = probe_harness();
9419
9420        for _ in 0..20 {
9421            time_out_without_answer(&mut harness).await;
9422            assert_eq!(
9423                harness.forwarding.health_probe_tombstone_count().unwrap(),
9424                1
9425            );
9426        }
9427    }
9428}
9429
9430#[cfg(test)]
9431mod child_env_tests {
9432    use super::{
9433        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
9434        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
9435        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
9436    };
9437    use std::{ffi::OsStr, path::PathBuf};
9438    use tokio::process::Command;
9439
9440    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
9441        ModuleSpec {
9442            launch_nonce_env: true,
9443            module_id: "env-plan".to_string(),
9444            program: PathBuf::from("/nonexistent"),
9445            args: Vec::new(),
9446            env,
9447            reserved: false,
9448            reserved_prefixes: Vec::new(),
9449            protocol: ModuleProtocol::Subc,
9450            overlap: Default::default(),
9451        }
9452    }
9453
9454    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
9455    /// one still gets its own.
9456    ///
9457    /// This is the narrow goal `env_clear()` was reached for, and the reason the
9458    /// fix is `env_remove` rather than deleting the line: an operator's ambient
9459    /// filter silently becoming an unconfigured module's log level is a real
9460    /// defect, just a much smaller one than clearing the environment.
9461    ///
9462    /// Asserted on the command plan rather than a spawned child because proving
9463    /// the ABSENCE of an inherited variable needs the parent's environment
9464    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
9465    /// removal as `(key, None)`, which is exactly the distinction wanted: not
9466    /// "absent because nobody set it" but "explicitly unset for the child".
9467    #[test]
9468    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
9469        let mut command = Command::new("/nonexistent");
9470        apply_child_env(&mut command, &spec(Vec::new()));
9471        let removed = command
9472            .as_std()
9473            .get_envs()
9474            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
9475        assert!(
9476            removed,
9477            "ambient CK_LOG must be explicitly removed for an unconfigured module"
9478        );
9479
9480        let mut configured = Command::new("/nonexistent");
9481        apply_child_env(
9482            &mut configured,
9483            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
9484        );
9485        let effective = configured
9486            .as_std()
9487            .get_envs()
9488            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
9489            .last()
9490            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9491        assert_eq!(
9492            effective,
9493            Some(Some("debug".to_string())),
9494            "a module's configured CK_LOG must survive the ambient removal"
9495        );
9496    }
9497
9498    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
9499    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
9500    /// the same reason as the CK_LOG test above.
9501    ///
9502    /// The argument is the load-bearing half: a stock binary exits on an
9503    /// unknown flag before it listens, so with `--subc` appended the mode
9504    /// could not supervise the one process it exists for. Found by the first
9505    /// conformance run (nats-server: `flag provided but not defined: -subc`).
9506    #[test]
9507    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
9508        let connection_file = std::path::Path::new("/run/subc-connection.json");
9509        let handle = SupervisorHandle::new();
9510
9511        let mut none_spec = spec(Vec::new());
9512        none_spec.protocol = ModuleProtocol::None;
9513        let mut none = Command::new("/nonexistent");
9514        let none_handoff =
9515            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
9516                .expect("protocol-none spawn args apply");
9517        assert!(
9518            none_handoff.is_none(),
9519            "protocol:none spawn must not receive a nonce descriptor"
9520        );
9521        assert!(
9522            !none.as_std().get_envs().any(|(key, value)| key
9523                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
9524                && value.is_some()),
9525            "protocol:none spawn must not name a nonce descriptor"
9526        );
9527        let none_args: Vec<String> = none
9528            .as_std()
9529            .get_args()
9530            .map(|a| a.to_string_lossy().into_owned())
9531            .collect();
9532        assert!(
9533            !none_args.iter().any(|a| a == SUBC_ARG),
9534            "protocol:none argv must not carry --subc; got {none_args:?}"
9535        );
9536        let none_has_nonce = none
9537            .as_std()
9538            .get_envs()
9539            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
9540        assert!(
9541            !none_has_nonce,
9542            "protocol:none spawn must not receive a launch nonce"
9543        );
9544        let none_has_module_id = none
9545            .as_std()
9546            .get_envs()
9547            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
9548        assert!(
9549            none_has_module_id,
9550            "SUBC_MODULE_ID is inert and stays on every path"
9551        );
9552        assert!(
9553            handle.spawn_nonce(&none_spec.module_id).is_none(),
9554            "no nonce record for a process that will never present one"
9555        );
9556
9557        // Control: the subc-wire path is unchanged by the branch above.
9558        let wire_spec = spec(Vec::new());
9559        let mut wire = Command::new("/nonexistent");
9560        let wire_handoff =
9561            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
9562                .expect("subc-wire spawn args apply");
9563        let wire_fd_env = wire
9564            .as_std()
9565            .get_envs()
9566            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
9567            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9568        #[cfg(unix)]
9569        assert_eq!(
9570            wire_fd_env,
9571            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
9572            "a subc-wire spawn names the pipe it will receive at descriptor 3"
9573        );
9574        #[cfg(not(unix))]
9575        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
9576        let wire_args: Vec<String> = wire
9577            .as_std()
9578            .get_args()
9579            .map(|a| a.to_string_lossy().into_owned())
9580            .collect();
9581        assert_eq!(
9582            wire_args,
9583            vec![
9584                SUBC_ARG.to_string(),
9585                connection_file.to_string_lossy().into_owned()
9586            ],
9587            "a subc-wire spawn still carries --subc <path>"
9588        );
9589        assert!(wire
9590            .as_std()
9591            .get_envs()
9592            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()));
9593        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
9594    }
9595
9596    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
9597    /// spec tries to set it; only a swap candidate carries it.
9598    ///
9599    /// "Set it only on candidates" is not enough, because spawn applies the
9600    /// spec's env verbatim and the daemon's own environment is inherited: either
9601    /// could hand a plain restart the swap role, and a module reading it would
9602    /// warm on its long swap budget while callers wait. Asserted as an explicit
9603    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
9604    /// test above gives.
9605    #[test]
9606    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
9607        let role = |command: &Command| {
9608            command
9609                .as_std()
9610                .get_envs()
9611                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
9612                .last()
9613                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
9614        };
9615        let forged = spec(vec![(
9616            SUBC_SPAWN_ROLE_ENV.to_string(),
9617            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
9618        )]);
9619
9620        let mut plain = Command::new("/nonexistent");
9621        apply_child_env(&mut plain, &forged);
9622        apply_spawn_role(&mut plain, SpawnRole::Plain);
9623        assert_eq!(
9624            role(&plain),
9625            Some(None),
9626            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
9627        );
9628
9629        let mut candidate = Command::new("/nonexistent");
9630        apply_child_env(&mut candidate, &spec(Vec::new()));
9631        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
9632        assert_eq!(
9633            role(&candidate),
9634            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
9635        );
9636    }
9637
9638    /// Daemon-private capture retention keys never reach the child.
9639    ///
9640    /// cortexkit-log exposes retention as a Rust struct with no environment
9641    /// names, so these entries are supervisor metadata. Passing them through
9642    /// would invent a public child-process contract by accident.
9643    #[test]
9644    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
9645        let mut command = Command::new("/nonexistent");
9646        apply_child_env(
9647            &mut command,
9648            &spec(vec![
9649                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
9650                ("KEPT".to_string(), "yes".to_string()),
9651            ]),
9652        );
9653        let keys: Vec<String> = command
9654            .as_std()
9655            .get_envs()
9656            .filter(|(_, value)| value.is_some())
9657            .map(|(key, _)| key.to_string_lossy().into_owned())
9658            .collect();
9659        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
9660        assert!(
9661            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
9662            "daemon-private capture key leaked to the child: {keys:?}"
9663        );
9664    }
9665}
9666
9667#[cfg(test)]
9668mod jitter_tests {
9669    use super::jittered_health_delay;
9670    use std::{collections::HashSet, time::Duration};
9671
9672    /// Module ids drawn from a real fleet, so the dispersal claim is about names
9673    /// that actually occur rather than invented ones.
9674    ///
9675    /// This is a SAMPLE, not a registry: the property under test is that distinct
9676    /// ids disperse, which holds for any set of distinct strings. Several entries
9677    /// are already historical (modules get renamed), and that costs nothing here --
9678    /// but it means a reader must not mistake this for the live module set, and a
9679    /// rename sweep will match it without there being anything to change.
9680    const FLEET: [&str; 14] = [
9681        "aft",
9682        "alfonso-core",
9683        "magic-context",
9684        "broca",
9685        "thalamus",
9686        "quota",
9687        "engram",
9688        "plexus",
9689        "cerebellum",
9690        "astrocyte",
9691        "synapse",
9692        "subc-mcp",
9693        "cortexkit-credentials",
9694        "subc-federation",
9695    ];
9696
9697    /// Probes must not converge after a fleet-wide restart.
9698    ///
9699    /// This is the property the jitter exists for: every module reconnects at
9700    /// once, and without dispersal all fourteen would then probe on the same
9701    /// tick forever. Nothing failed visibly when this went untested -- a
9702    /// convergent fleet still probes correctly, just in a burst, so the symptom
9703    /// is a periodic load spike that looks like whatever else is running.
9704    #[test]
9705    fn probe_delays_disperse_across_the_fleet() {
9706        let cadence = Duration::from_secs(30);
9707        let delays: HashSet<Duration> = FLEET
9708            .iter()
9709            .map(|id| jittered_health_delay(id, 0, cadence))
9710            .collect();
9711        assert_eq!(
9712            delays.len(),
9713            FLEET.len(),
9714            "every supervised module must land on its own probe offset"
9715        );
9716    }
9717
9718    /// The offset may only ever DELAY a probe, never bring it forward.
9719    ///
9720    /// A delay below the cadence would probe a module more often than
9721    /// configured, which is the opposite of what an operator asked for and
9722    /// would tighten the failure budget without anyone changing it.
9723    #[test]
9724    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
9725        let cadence = Duration::from_secs(30);
9726        let span = cadence / 10;
9727        for id in FLEET {
9728            for probe_index in 0..8 {
9729                let delay = jittered_health_delay(id, probe_index, cadence);
9730                assert!(
9731                    delay >= cadence,
9732                    "{id}#{probe_index}: jitter must not shorten the cadence"
9733                );
9734                assert!(
9735                    delay < cadence + span,
9736                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
9737                );
9738            }
9739        }
9740    }
9741
9742    /// A module keeps its offset across daemon restarts.
9743    ///
9744    /// The delay is derived rather than randomised precisely so a restart does
9745    /// not re-roll every module into a fresh chance of collision. A random
9746    /// source would satisfy the dispersal test above and quietly lose this.
9747    #[test]
9748    fn a_module_offset_is_stable_across_restarts() {
9749        let cadence = Duration::from_secs(30);
9750        for id in FLEET {
9751            assert_eq!(
9752                jittered_health_delay(id, 0, cadence),
9753                jittered_health_delay(id, 0, cadence),
9754                "{id}: the same module and probe index must produce the same offset"
9755            );
9756        }
9757    }
9758
9759    /// A zero cadence disables probing rather than producing a busy loop.
9760    #[test]
9761    fn zero_cadence_yields_zero_delay() {
9762        assert_eq!(
9763            jittered_health_delay("aft", 0, Duration::ZERO),
9764            Duration::ZERO
9765        );
9766    }
9767}
9768
9769#[cfg(all(test, target_os = "linux"))]
9770mod cgroup_placement_tests {
9771    use super::{
9772        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
9773        SupervisedChild,
9774    };
9775    use crate::stderr_tail::{StderrRing, StderrTailConfig};
9776    use std::{
9777        fs, io,
9778        path::{Path, PathBuf},
9779        sync::{Arc, Mutex},
9780    };
9781    use subc_test_support::TestTempDir;
9782    use tokio::process::Command;
9783
9784    #[test]
9785    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
9786        let path = Path::new("/definitely-missing-subc-cgroup");
9787        let mut command = Command::new("true");
9788        let error = apply_cgroup_placement(
9789            &mut command,
9790            &ModuleSpec {
9791                launch_nonce_env: true,
9792                module_id: "broken-cgroup".to_string(),
9793                program: PathBuf::from("true"),
9794                args: Vec::new(),
9795                env: Vec::new(),
9796                reserved: false,
9797                reserved_prefixes: Vec::new(),
9798                protocol: ModuleProtocol::Subc,
9799                overlap: Default::default(),
9800            },
9801            path,
9802        )
9803        .expect_err("a parent cgroup open failure must reject the supervised spawn");
9804        let reason = error.to_string();
9805
9806        assert!(
9807            matches!(error, SuperviseError::Cgroup { .. }),
9808            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
9809        );
9810        assert!(
9811            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
9812            "parent cgroup open failure must name cgroup.procs: {reason}"
9813        );
9814    }
9815
9816    #[tokio::test]
9817    async fn reaping_a_child_removes_its_empty_module_cgroup() {
9818        let root = TestTempDir::new("supervisor-reap-cgroup");
9819        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9820        let placement = subc_cgroup::prepare_at(&root)
9821            .expect("prepare scratch cgroup root")
9822            .expect("scratch root has a cgroup.procs marker");
9823        let module_id = "reaped-module";
9824        let module = placement
9825            .module_path(module_id)
9826            .expect("create scratch module cgroup");
9827        let child = Command::new("true")
9828            .spawn()
9829            .expect("spawn short-lived child");
9830        let pid = child.id().expect("spawned child has pid");
9831        let mut child = SupervisedChild {
9832            child,
9833            module_id: module_id.to_string(),
9834            cgroup_placement: Some(placement),
9835            stdout_pump: None,
9836            stderr_pump: None,
9837            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
9838            spawned_at_ms: 0,
9839            spawned_from: PathBuf::from("true"),
9840            spawned_file_identity: None,
9841            process_start_time: None,
9842            process_identity: None,
9843            pid,
9844            roster_guard: None,
9845        };
9846
9847        child.wait().await.expect("reap short-lived child");
9848
9849        assert!(
9850            !module.exists(),
9851            "reaping the supervised child must remove its empty cgroup"
9852        );
9853    }
9854
9855    #[test]
9856    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
9857        let root = TestTempDir::new("supervisor-non-empty-cgroup");
9858        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9859        let placement = subc_cgroup::prepare_at(&root)
9860            .expect("prepare scratch cgroup root")
9861            .expect("scratch root has a cgroup.procs marker");
9862        let module = placement
9863            .module_path("surviving-module")
9864            .expect("create scratch module cgroup");
9865        fs::write(module.join("surviving-process"), b"still present")
9866            .expect("make scratch cgroup non-empty");
9867        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
9868
9869        remove_module_cgroup(&placement, "surviving-module");
9870
9871        let logs = crate::router::test_log::captured_logs(&logs);
9872        assert!(
9873            module.exists(),
9874            "failed removal must leave the cgroup intact"
9875        );
9876        assert!(
9877            logs.contains("could not remove module cgroup after process exit; continuing teardown")
9878                && logs.contains("surviving-module"),
9879            "best-effort removal must report the failure without returning it: {logs}"
9880        );
9881    }
9882
9883    #[test]
9884    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
9885        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
9886        let reason = SuperviseError::Spawn {
9887            program: PathBuf::from("/bin/true"),
9888            source: io::Error::from_raw_os_error(13),
9889            cgroup_path: Some(cgroup_path.clone()),
9890        }
9891        .to_string();
9892
9893        assert!(
9894            reason.contains(&cgroup_path.display().to_string()),
9895            "a pre_exec spawn failure must name the cgroup path: {reason}"
9896        );
9897    }
9898}
9899
9900#[cfg(test)]
9901mod spawn_subscriber_lag_tests {
9902    use super::*;
9903
9904    /// A subscriber whose connection stops draining is dropped once its frame
9905    /// channel fills. The client must learn that from a terminal Error frame
9906    /// after the frames already queued for it, not from a stream that simply
9907    /// goes quiet.
9908    #[tokio::test]
9909    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
9910        let feed = SpawnEventFeed::default();
9911        feed.configure_incarnation("lag-incarnation".to_string());
9912        // A one-slot connection queue that nobody reads until the emits are
9913        // done: the forwarder parks on it and the subscriber channel fills.
9914        let (tx, mut rx) = mpsc::channel(1);
9915        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
9916            .expect("subscribe");
9917        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
9918        for index in 0..emitted {
9919            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
9920            // Let the forwarder take what it can so the fill point is the
9921            // subscriber channel, not a scheduling accident.
9922            tokio::task::yield_now().await;
9923        }
9924        assert_eq!(
9925            feed.subscriber_count(),
9926            0,
9927            "the lagged subscriber must be removed"
9928        );
9929
9930        let mut data = Vec::new();
9931        let mut last = None;
9932        loop {
9933            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
9934                .await
9935                .expect("the forwarder must finish once the subscriber is dropped");
9936            let Some(outbound) = next else { break };
9937            let frame = outbound.frame;
9938            if frame.header.ty == FrameType::StreamData {
9939                assert!(last.is_none(), "no data may follow the terminal frame");
9940                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
9941                data.push(event.cursor.seq);
9942            } else {
9943                assert!(last.is_none(), "exactly one terminal frame");
9944                last = Some(frame);
9945            }
9946        }
9947        assert!(!data.is_empty(), "queued frames drain before the terminal");
9948        for pair in data.windows(2) {
9949            assert_eq!(
9950                pair[1],
9951                pair[0] + 1,
9952                "queued frames arrive dense and in order"
9953            );
9954        }
9955        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
9956        assert_eq!(terminal.header.ty, FrameType::Error);
9957        assert_eq!(terminal.header.corr, 7);
9958        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
9959        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
9960        let detail = body.detail.expect("lagged error carries detail");
9961        assert_eq!(
9962            detail["first_undelivered_cursor"]["seq"],
9963            data.last().unwrap() + 1,
9964            "the named cursor is the first event the subscriber did not receive"
9965        );
9966        assert_eq!(
9967            detail["first_undelivered_cursor"]["daemon_incarnation"],
9968            "lag-incarnation"
9969        );
9970    }
9971}
9972
9973#[cfg(test)]
9974mod terminal_history_read_concurrency_tests {
9975    use super::*;
9976    use crate::terminal_journal::read_pause;
9977    use std::sync::mpsc as std_mpsc;
9978    use subc_test_support::TestTempDir;
9979
9980    fn journaled_ring(
9981        journal: &Arc<crate::terminal_journal::TerminalJournal>,
9982    ) -> Arc<Mutex<TerminalRing>> {
9983        Arc::new(Mutex::new(
9984            TerminalRing::new(TerminalRingConfig::default(), 1)
9985                .with_journal(Some(Arc::clone(journal))),
9986        ))
9987    }
9988
9989    fn crash(at_ms: u64) -> ExitReport {
9990        ExitReport {
9991            kind: ExitKind::Crash,
9992            code: Some(1),
9993            signal: None,
9994            at_ms,
9995        }
9996    }
9997
9998    /// Record an exit on another thread and report whether it finished within
9999    /// `bound`. The recorder thread is left running if it did not.
10000    fn record_within(
10001        module_id: &'static str,
10002        ring: &Arc<Mutex<TerminalRing>>,
10003        at_ms: u64,
10004        bound: Duration,
10005    ) -> bool {
10006        let ring = Arc::clone(ring);
10007        let (done, done_rx) = std_mpsc::channel();
10008        std::thread::spawn(move || {
10009            record_terminal(
10010                module_id,
10011                &ring,
10012                &SpawnEventFeed::default(),
10013                &crash(at_ms),
10014                TerminalDisposition::Restarting,
10015            );
10016            let _ = done.send(());
10017        });
10018        done_rx.recv_timeout(bound).is_ok()
10019    }
10020
10021    /// A history read in progress must not hold the journal writer (which every
10022    /// module's exit recording needs) or the module's own ring. Exits recorded
10023    /// while the read is paused complete promptly; the paused read answers as of
10024    /// the moment it started, and the next read has each exit exactly once.
10025    #[test]
10026    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
10027        let dir = TestTempDir::new("terminal-history-concurrent-read");
10028        let path = dir.join("terminals.jsonl");
10029        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
10030            path.clone(),
10031            "daemon".into(),
10032        ));
10033        let reader_ring = journaled_ring(&journal);
10034        let other_ring = journaled_ring(&journal);
10035        assert!(record_within(
10036            "reader-module",
10037            &reader_ring,
10038            10,
10039            Duration::from_secs(5)
10040        ));
10041
10042        let (started, release) = read_pause::install(&path);
10043        let reading = {
10044            let ring = Arc::clone(&reader_ring);
10045            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
10046        };
10047        started
10048            .recv_timeout(Duration::from_secs(5))
10049            .expect("the history read reached its pause");
10050
10051        let bound = Duration::from_secs(1);
10052        assert!(
10053            record_within("other-module", &other_ring, 20, bound),
10054            "another module's exit waited on a history read (journal writer held)"
10055        );
10056        assert!(
10057            record_within("reader-module", &reader_ring, 30, bound),
10058            "the read module's own exit waited on its history read (ring held)"
10059        );
10060
10061        drop(release);
10062        let paused = reading.join().unwrap();
10063        assert_eq!(
10064            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10065            vec![10],
10066            "an exit recorded after the read began lands in neither half of it"
10067        );
10068        assert_eq!(paused.journal_skipped_lines, 0);
10069        assert_eq!(paused.journal_read_errors, 0);
10070
10071        let after = durable_terminal_history_of(&reader_ring, "reader-module");
10072        assert_eq!(
10073            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10074            vec![10, 30],
10075            "the next read merges ring and journal with no duplicate"
10076        );
10077        assert_eq!(after.journal_skipped_lines, 0);
10078    }
10079}
10080
10081/// What a restart does with the exited process's stderr reader. These drive
10082/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
10083/// holds, so a reader that has not been scheduled by the bound is a controlled
10084/// input rather than something only a loaded machine produces.
10085#[cfg(test)]
10086mod stderr_settle_tests {
10087    use std::{
10088        future::Future,
10089        io,
10090        pin::Pin,
10091        sync::{Arc, Mutex},
10092        task::{Context, Poll},
10093        time::Duration,
10094    };
10095
10096    use tokio::{
10097        io::{AsyncRead, ReadBuf},
10098        sync::oneshot,
10099        time::Instant,
10100    };
10101
10102    use super::{settle_stderr_pump, StderrPump};
10103    use crate::stderr_tail::{
10104        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
10105    };
10106
10107    const BOUND: Duration = Duration::from_millis(250);
10108
10109    /// Yields `before`, then stays pending until the gate is released, then
10110    /// yields `after` and reaches EOF. The bytes after the gate were written
10111    /// by a process that has already exited; only the reader is behind.
10112    struct HeldReader {
10113        before: Option<Vec<u8>>,
10114        gate: Option<oneshot::Receiver<()>>,
10115        after: io::Cursor<Vec<u8>>,
10116    }
10117
10118    impl AsyncRead for HeldReader {
10119        fn poll_read(
10120            mut self: Pin<&mut Self>,
10121            cx: &mut Context<'_>,
10122            buf: &mut ReadBuf<'_>,
10123        ) -> Poll<io::Result<()>> {
10124            if let Some(bytes) = self.before.take() {
10125                buf.put_slice(&bytes);
10126                return Poll::Ready(Ok(()));
10127            }
10128            if let Some(gate) = self.gate.as_mut() {
10129                match Pin::new(gate).poll(cx) {
10130                    Poll::Pending => return Poll::Pending,
10131                    Poll::Ready(_) => self.gate = None,
10132                }
10133            }
10134            Pin::new(&mut self.after).poll_read(cx, buf)
10135        }
10136    }
10137
10138    struct DiscardSink;
10139
10140    impl OutputSink for DiscardSink {
10141        fn write_line(&mut self, _line: &[u8]) {}
10142    }
10143
10144    fn line(text: &str) -> TailEntry {
10145        TailEntry::Line {
10146            text: text.to_string(),
10147            truncated: false,
10148            at_ms: None,
10149        }
10150    }
10151
10152    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
10153        ring.lock().unwrap()
10154    }
10155
10156    /// Start a reader for a new process generation that delivers `before`
10157    /// immediately and `after` only once the returned sender fires (or is
10158    /// dropped).
10159    fn held_pump(
10160        ring: &Arc<Mutex<StderrRing>>,
10161        before: &str,
10162        after: &str,
10163    ) -> (StderrPump, oneshot::Sender<()>) {
10164        let generation = lock(ring).begin_process();
10165        let (release, gate) = oneshot::channel();
10166        let reader = HeldReader {
10167            before: Some(before.as_bytes().to_vec()),
10168            gate: Some(gate),
10169            after: io::Cursor::new(after.as_bytes().to_vec()),
10170        };
10171        let task = tokio::spawn(pump_stderr_to(
10172            reader,
10173            Arc::clone(ring),
10174            generation,
10175            DiscardSink,
10176        ));
10177        (StderrPump { task, generation }, release)
10178    }
10179
10180    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
10181        for _ in 0..1000 {
10182            if done(&lock(ring)) {
10183                return;
10184            }
10185            tokio::time::sleep(Duration::from_millis(1)).await;
10186        }
10187        panic!(
10188            "ring never reached the expected state: {:?}",
10189            lock(ring).snapshot(None, None)
10190        );
10191    }
10192
10193    #[tokio::test(start_paused = true)]
10194    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
10195        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10196        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
10197
10198        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
10199        let before_release = lock(&ring).snapshot(None, None);
10200        assert!(
10201            matches!(before_release.capture, CaptureState::Incomplete { .. }),
10202            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
10203        );
10204
10205        // The restart: the next process starts and writes before the old
10206        // reader catches up.
10207        let next = lock(&ring).begin_process();
10208        lock(&ring).push_line_from(next, "next process booting");
10209        release.send(()).unwrap();
10210        wait_until(&ring, |ring| {
10211            ring.snapshot(None, None).capture == CaptureState::Captured
10212        })
10213        .await;
10214
10215        assert_eq!(
10216            untimed(lock(&ring).snapshot(None, None).entries),
10217            vec![
10218                line("booting"),
10219                line("config error: missing storage"),
10220                TailEntry::ProcessStart,
10221                line("next process booting"),
10222            ],
10223            "the crash's last line must survive a slow reader and stay in the crashed process's section"
10224        );
10225    }
10226
10227    #[tokio::test(start_paused = true)]
10228    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
10229    ) {
10230        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10231        // `_held` is never fired: a descendant keeps the pipe open for the
10232        // whole test.
10233        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
10234
10235        let started = Instant::now();
10236        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
10237        assert_eq!(
10238            started.elapsed(),
10239            BOUND,
10240            "the restart must wait exactly the bound for a pipe that stays open, no longer"
10241        );
10242
10243        let next = lock(&ring).begin_process();
10244        lock(&ring).push_line_from(next, "next process booting");
10245        tokio::time::sleep(Duration::from_secs(60)).await;
10246
10247        let snapshot = lock(&ring).snapshot(None, None);
10248        match &snapshot.capture {
10249            CaptureState::Incomplete { reason } => assert!(
10250                reason.contains("had not reached EOF") && reason.contains("250ms"),
10251                "the reason must say what is missing and after how long: {reason}"
10252            ),
10253            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
10254        }
10255        assert_eq!(
10256            untimed(snapshot.entries),
10257            vec![
10258                line("parent exiting"),
10259                TailEntry::ProcessStart,
10260                line("next process booting"),
10261            ]
10262        );
10263    }
10264
10265    #[tokio::test(start_paused = true)]
10266    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
10267        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10268        let (pump, release) = held_pump(&ring, "one\n", "two\n");
10269        release.send(()).unwrap();
10270
10271        settle_stderr_pump("clean", &ring, pump, BOUND).await;
10272
10273        let snapshot = lock(&ring).snapshot(None, None);
10274        assert_eq!(snapshot.capture, CaptureState::Captured);
10275        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
10276    }
10277}
10278
10279/// Containment of a module's process tree (issue #109).
10280///
10281/// The behaviour these defend against is a module helper surviving its module:
10282/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
10283/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
10284/// compounds it.
10285///
10286/// They run against the SUPERVISOR rather than the job-object crate because the
10287/// claim is about teardown: a crate-level test proves a job can reap a tree, not
10288/// that the daemon's drain path reaches it.
10289///
10290/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
10291/// lane there is a separate containment path with its own tests.
10292#[cfg(all(test, windows))]
10293mod job_containment_tests {
10294    use super::*;
10295    use std::{
10296        path::{Path, PathBuf},
10297        sync::{Arc, Mutex},
10298        time::{Duration, Instant},
10299    };
10300    use subc_test_support::TestTempDir;
10301
10302    /// The stub, expected beside this test executable.
10303    ///
10304    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
10305    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
10306    /// failure then reads as a broken test rather than an unbuilt dependency.
10307    fn stub_path() -> PathBuf {
10308        let mut path = std::env::current_exe().expect("current_exe available in tests");
10309        path.pop();
10310        path.pop();
10311        path.push("fake-aft-stub.exe");
10312        assert!(
10313            path.exists(),
10314            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
10315             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
10316            path.display()
10317        );
10318        path
10319    }
10320
10321    /// Poll for the grandchild pid the stub records, and parse it.
10322    fn read_grandchild_pid(path: &Path) -> u32 {
10323        let deadline = Instant::now() + Duration::from_secs(10);
10324        loop {
10325            if let Ok(contents) = std::fs::read_to_string(path) {
10326                if let Ok(pid) = contents.trim().parse() {
10327                    return pid;
10328                }
10329            }
10330            assert!(
10331                Instant::now() < deadline,
10332                "the stub never recorded a grandchild pid at {}",
10333                path.display()
10334            );
10335            std::thread::sleep(Duration::from_millis(10));
10336        }
10337    }
10338
10339    /// Everything one fixture run needs, so the two tests below differ in exactly
10340    /// one place: whether the child is contained.
10341    struct Fixture {
10342        _dir: TestTempDir,
10343        module_id: String,
10344        grandchild: u32,
10345        child: Option<SupervisedChild>,
10346        registry: Arc<Registry>,
10347        snapshot: Arc<Mutex<SupervisorSnapshot>>,
10348        terminal_ring: Arc<Mutex<TerminalRing>>,
10349        spawn_events: SpawnEventFeed,
10350    }
10351
10352    fn fixture(label: &str, module_id: &str) -> Fixture {
10353        let dir = TestTempDir::new(label);
10354        let pid_file = dir.join("grandchild.pid");
10355        let supervisor = Supervisor::new(
10356            Arc::new(Registry::default()),
10357            RestartPolicy::new(3, Duration::ZERO),
10358        );
10359        let runtime = supervisor.runtime_config();
10360        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10361        let spec = ModuleSpec {
10362            launch_nonce_env: true,
10363            module_id: module_id.to_string(),
10364            program: stub_path(),
10365            // Zero args deliberately: a `--subc` argument would make the stub dial
10366            // a daemon that is not there, and the failure would land in the same
10367            // stderr ring this fixture exists to keep quiet.
10368            args: Vec::new(),
10369            env: vec![
10370                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10371                (
10372                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
10373                    pid_file.display().to_string(),
10374                ),
10375            ],
10376            reserved: false,
10377            reserved_prefixes: Vec::new(),
10378            protocol: ModuleProtocol::Subc,
10379            overlap: Default::default(),
10380        };
10381        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
10382            .expect("spawn the supervised fixture");
10383        let grandchild = read_grandchild_pid(&pid_file);
10384        Fixture {
10385            _dir: dir,
10386            module_id: module_id.to_string(),
10387            grandchild,
10388            child: Some(child),
10389            registry: Arc::new(Registry::default()),
10390            snapshot,
10391            terminal_ring: Arc::clone(&runtime.terminal_ring),
10392            spawn_events: SpawnEventFeed::default(),
10393        }
10394    }
10395
10396    impl Fixture {
10397        /// Drain through the supervisor's own teardown path.
10398        async fn drain(&mut self) {
10399            let child = self
10400                .child
10401                .take()
10402                .expect("the fixture child is still present");
10403            drain_child_to_state(
10404                &self.module_id,
10405                ModuleProtocol::Subc,
10406                // No forwarding table in this fixture, so nothing reaches the
10407                // child over a connection.
10408                StopNotice::NotSent,
10409                &self.registry,
10410                &self.snapshot,
10411                &self.terminal_ring,
10412                &self.spawn_events,
10413                child,
10414                Duration::from_millis(500),
10415                ModuleState::Stopped,
10416                Some(false),
10417            )
10418            .await
10419            .expect("drain the supervised fixture");
10420        }
10421    }
10422
10423    /// Teardown reaps the grandchild, not merely the direct child.
10424    ///
10425    /// This is the assertion the change exists for. Before containment the
10426    /// grandchild survived: it is a separate process, and `start_kill` is
10427    /// `TerminateProcess` scoped to one pid.
10428    #[tokio::test]
10429    async fn teardown_reaps_the_grandchild() {
10430        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
10431        let grandchild = fixture.grandchild;
10432
10433        assert!(
10434            subc_jobobject::process_exists(grandchild),
10435            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
10436        );
10437
10438        fixture.drain().await;
10439
10440        assert!(
10441            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10442            "grandchild {grandchild} outlived module teardown: the tree was not contained"
10443        );
10444    }
10445
10446    /// The mutation control: with containment withheld, the grandchild survives
10447    /// the same kill.
10448    ///
10449    /// This is the defect reproduction from #109 — a direct-child kill reaches
10450    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
10451    /// supervisor because `spawn_and_mark_running` now always contains on
10452    /// Windows, which is the point: there is no longer a path that spawns
10453    /// uncontained, so the control has to construct one.
10454    ///
10455    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
10456    /// grandchild ever dies here, that test is passing for a reason unrelated to
10457    /// the job object and the containment claim is unproven.
10458    #[test]
10459    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
10460        let dir = TestTempDir::new("teardown-uncontained");
10461        let pid_file = dir.join("grandchild.pid");
10462        let mut child = std::process::Command::new(stub_path())
10463            .env("FAKE_AFT_NEVER_CONNECT", "1")
10464            .env(
10465                "FAKE_AFT_GRANDCHILD_PID_FILE",
10466                pid_file.display().to_string(),
10467            )
10468            .stdin(std::process::Stdio::null())
10469            .stdout(std::process::Stdio::null())
10470            .stderr(std::process::Stdio::null())
10471            .spawn()
10472            .expect("spawn the uncontained fixture");
10473        let grandchild = read_grandchild_pid(&pid_file);
10474
10475        // Exactly what the pre-fix teardown did: kill the direct child.
10476        child.kill().expect("kill the direct child");
10477        let _ = child.wait();
10478
10479        assert!(
10480            subc_jobobject::process_exists(grandchild),
10481            "grandchild {grandchild} died with the direct child, so this control no longer \
10482             distinguishes contained from uncontained teardown and the regression test is \
10483             passing vacuously"
10484        );
10485
10486        // The orphan this control demonstrates is the leak the fix prevents, so
10487        // the control must not leave one behind.
10488        kill_tree(grandchild);
10489    }
10490
10491    /// Crash durability: closing the containment handle reaps the tree with no
10492    /// teardown code running at all.
10493    ///
10494    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
10495    /// call anything — and it is why containment is a kernel property of the
10496    /// handle rather than a step in the drain. Discovered by getting the
10497    /// mutation control wrong: clearing `job` to "disable" containment instead
10498    /// killed the tree, which is the guarantee, not a mistake.
10499    #[tokio::test]
10500    async fn dropping_containment_reaps_the_grandchild() {
10501        let mut fixture = fixture("drop-containment", "tree-drop");
10502        let grandchild = fixture.grandchild;
10503
10504        assert!(subc_jobobject::process_exists(grandchild));
10505
10506        // No `drain` call, no kill: dropping the handle is the entire mechanism.
10507        fixture.child.as_mut().expect("child present").job = None;
10508
10509        assert!(
10510            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10511            "grandchild {grandchild} survived the containment handle closing, so a daemon \
10512             crash would leave the tree behind"
10513        );
10514    }
10515
10516    /// Kill a pid and its tree, then confirm it is gone.
10517    fn kill_tree(pid: u32) {
10518        let _ = std::process::Command::new("taskkill.exe")
10519            .args(["/PID", &pid.to_string(), "/T", "/F"])
10520            .stdin(std::process::Stdio::null())
10521            .stdout(std::process::Stdio::null())
10522            .stderr(std::process::Stdio::null())
10523            .status();
10524        assert!(
10525            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
10526            "could not clean up grandchild {pid}"
10527        );
10528    }
10529}
10530
10531/// The daemon's real spawn path hands a subc-wire child its launch nonce on
10532/// descriptor 3, with an optional environment copy. The shell records the nonce
10533/// and its environment after exec so these tests observe the real handover.
10534#[cfg(all(test, unix))]
10535mod launch_nonce_descriptor_tests {
10536    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
10537    use crate::stderr_tail::{StderrRing, StderrTailConfig};
10538    use std::{
10539        path::PathBuf,
10540        sync::{Arc, Mutex},
10541        time::{Duration, Instant},
10542    };
10543    use subc_test_support::TestTempDir;
10544
10545    async fn probe(launch_nonce_env: bool, role: super::SpawnRole) {
10546        let scratch = TestTempDir::new("launch-nonce-descriptor");
10547        let fd_copy = scratch.join("from-descriptor");
10548        let env_copy = scratch.join("environment");
10549        let script = format!(
10550            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
10551            fd = fd_copy.display(), env = env_copy.display(),
10552        );
10553        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
10554        let spec = ModuleSpec {
10555            launch_nonce_env,
10556            module_id: "nonce-descriptor-probe".to_string(),
10557            program: PathBuf::from("/bin/sh"),
10558            args: vec!["-c".to_string(), script],
10559            env: vec![
10560                xdg("XDG_DATA_HOME"),
10561                xdg("XDG_RUNTIME_DIR"),
10562                xdg("XDG_CONFIG_HOME"),
10563                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
10564            ],
10565            reserved: true,
10566            reserved_prefixes: Vec::new(),
10567            protocol: ModuleProtocol::Subc,
10568            overlap: Default::default(),
10569        };
10570        let handle = SupervisorHandle::new();
10571        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10572        let roster = ChildRoster::default();
10573        let child = super::spawn_child_in_slot(
10574            &spec,
10575            None,
10576            Some(&handle),
10577            &ring,
10578            None,
10579            &roster,
10580            #[cfg(target_os = "linux")]
10581            None,
10582            role,
10583            matches!(role, super::SpawnRole::SwapCandidate),
10584        )
10585        .expect("spawn probe");
10586        let deadline = Instant::now() + Duration::from_secs(10);
10587        while !(fd_copy.exists() && env_copy.exists()) {
10588            assert!(Instant::now() < deadline, "probe never wrote its copies");
10589            tokio::time::sleep(Duration::from_millis(20)).await;
10590        }
10591        let nonce = std::fs::read_to_string(fd_copy).unwrap();
10592        assert!(!nonce.is_empty());
10593        let environment = std::fs::read_to_string(env_copy).unwrap();
10594        assert!(environment
10595            .lines()
10596            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
10597        let copy = environment
10598            .lines()
10599            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
10600        assert_eq!(
10601            copy,
10602            launch_nonce_env.then_some(nonce.as_str()),
10603            "child environment must follow launch_nonce_env"
10604        );
10605        if matches!(role, super::SpawnRole::Plain) {
10606            assert_eq!(
10607                handle.spawn_nonce(&spec.module_id).as_deref(),
10608                Some(nonce.as_str())
10609            );
10610        }
10611        drop(child);
10612    }
10613
10614    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10615    async fn a_spawned_module_receives_its_nonce_on_descriptor_3_and_in_the_environment() {
10616        probe(true, super::SpawnRole::Plain).await;
10617    }
10618
10619    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10620    async fn launch_nonce_env_false_withholds_environment_from_real_child() {
10621        probe(false, super::SpawnRole::Plain).await;
10622    }
10623
10624    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10625    async fn launch_nonce_env_false_withholds_environment_from_swap_candidate() {
10626        probe(false, super::SpawnRole::SwapCandidate).await;
10627    }
10628}