Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113struct SupervisedChild {
114    child: Child,
115    /// The protocol this process was launched with. A reload can store a new
116    /// launch spec with a different protocol, but that takes effect only at the
117    /// next spawn, so this process keeps being handled by the protocol it
118    /// actually speaks.
119    protocol: ModuleProtocol,
120    /// This process's cgroup name: a bounded module/slot label followed by a
121    /// spawn suffix unique to this process (when cgroup placement is on). A
122    /// retired process in a slot may still be draining when a later one is
123    /// spawned into that slot, so the suffix keeps the later process out of
124    /// the retired one's cgroup, which is the domain a kill applies to.
125    #[cfg(target_os = "linux")]
126    module_id: String,
127    #[cfg(target_os = "linux")]
128    cgroup_placement: Option<subc_cgroup::Placement>,
129    /// The job that contains this child and every process it spawns (issue #109).
130    ///
131    /// Dropping this handle is what reaps a surviving tree when no supervisor
132    /// code runs — a daemon crash — because the job carries
133    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
134    ///
135    /// That limit is not crash-only, and the difference is worth knowing: a
136    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
137    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
138    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
139    /// module at once. Before this change they survived that, saw EOF on the
140    /// control socket, and ran their own teardown; Unix keeps that path
141    /// deliberately, so a module can seal a WAL or close a capture rather than
142    /// be killed mid-write. So this trades graceful teardown on every Windows
143    /// daemon stop for containment on a crash, which is the right way round
144    /// today: orphaned GPU workers are a reported, recurring problem, and the
145    /// modules that write most heavily do not run on Windows.
146    ///
147    /// The fix is a real Windows stop path — the daemon draining before it
148    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
149    /// reaches only what the drain left behind, which is what it should reach.
150    #[cfg(windows)]
151    job: Option<subc_jobobject::JobObject>,
152    stdout_pump: Option<JoinHandle<()>>,
153    stderr_pump: Option<StderrPump>,
154    stderr_ring: Arc<Mutex<StderrRing>>,
155    spawned_at_ms: u64,
156    spawned_from: PathBuf,
157    spawned_file_identity: Option<SpawnedFileIdentity>,
158    process_start_time: Option<u64>,
159    process_identity: Option<ProcessIdentity>,
160    pid: u32,
161    /// This process's entry in the daemon's child roster, released when the
162    /// process is reaped or this handle is dropped.
163    roster_guard: Option<crate::child_roster::RosterGuard>,
164}
165
166impl SupervisedChild {
167    fn id(&self) -> Option<u32> {
168        Some(self.pid)
169    }
170
171    fn process_identity(&self) -> Option<ProcessIdentity> {
172        self.process_identity
173    }
174
175    async fn wait(&mut self) -> io::Result<ExitStatus> {
176        // The roster entry is NOT released here. A daemon shutdown waits for the
177        // roster to empty and then exits the process, so releasing at the reap
178        // let it exit before the exit handler wrote this child's terminal record
179        // (the stderr drain and snapshot update sit in between), and the
180        // shutdown's own `daemon_shutdown` record was intermittently lost. The
181        // caller releases it after recording the exit (`release_roster`), and
182        // dropping the handle releases it too.
183        let result = self.child.wait().await;
184        #[cfg(target_os = "linux")]
185        if result.is_ok() {
186            if let Some(placement) = self.cgroup_placement.as_ref() {
187                cleanup_reaped_cgroup(placement, &self.module_id).await;
188                // Keep ownership while awaiting kernel population changes: a
189                // drain timeout may cancel this wait and then escalate/reap.
190                self.cgroup_placement = None;
191            }
192        }
193        result
194    }
195
196    /// Releases this child's daemon-shutdown roster entry once its exit has
197    /// been recorded. The pid is already reaped and free for reuse, so the
198    /// entry must not outlive the record any longer than that.
199    fn release_roster(&mut self) {
200        self.roster_guard = None;
201    }
202
203    /// Kill the child and, where containment is available, its process tree.
204    ///
205    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
206    /// helper process leaked the helper — the Synapse embedding module's CUDA
207    /// worker holds the GPU allocation, so the leak cost VRAM until the next
208    /// restart of something else. Terminating the job reaches grandchildren that
209    /// a tree walk cannot, including one whose parent has already exited and
210    /// been reparented away.
211    ///
212    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
213    /// direct-child kill still decides the outcome, so containment can never
214    /// change whether a module is reported as stopped.
215    fn start_kill(&mut self) -> io::Result<()> {
216        #[cfg(windows)]
217        if let Some(job) = &self.job {
218            if let Err(error) = job.terminate() {
219                debug!(
220                    error = %error,
221                    "job termination failed; the direct-child kill still owns the outcome"
222                );
223            }
224        }
225        #[cfg(target_os = "linux")]
226        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
227        self.child.start_kill()
228    }
229
230    async fn drain_stderr(&mut self, module_id: &str) {
231        if let Some(mut pump) = self.stdout_pump.take() {
232            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
233                Ok(Ok(())) => {}
234                Ok(Err(error)) => {
235                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
236                }
237                Err(_) => {
238                    pump.abort();
239                    warn!(
240                        module_id,
241                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
242                        "stdout pump did not drain before restart; stopped it before the next process"
243                    );
244                }
245            }
246        }
247
248        let Some(pump) = self.stderr_pump.take() else {
249            return;
250        };
251        settle_stderr_pump(
252            module_id,
253            &self.stderr_ring,
254            pump,
255            STDERR_PUMP_DRAIN_TIMEOUT,
256        )
257        .await;
258    }
259}
260
261/// The reader task for one process's stderr, with the ring generation its
262/// lines are attributed to.
263struct StderrPump {
264    task: JoinHandle<()>,
265    generation: u64,
266}
267
268/// Retire an exited process's stderr reader and wait up to `bound` for it to
269/// reach EOF. A reader still running at the bound is detached, not stopped: it
270/// keeps filling the exited process's section of the ring until its pipe
271/// closes, and the tail reads `Incomplete` until then. See
272/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
273async fn settle_stderr_pump(
274    module_id: &str,
275    ring: &Arc<Mutex<StderrRing>>,
276    pump: StderrPump,
277    bound: Duration,
278) {
279    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
280    let StderrPump {
281        mut task,
282        generation,
283    } = pump;
284    lock().retire_pump(generation);
285    match timeout(bound, &mut task).await {
286        Ok(Ok(())) => {}
287        Ok(Err(err)) => {
288            let mut ring = lock();
289            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
290            ring.finish_pump(generation);
291            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
292        }
293        Err(_) => {
294            // Dropping the handle detaches the task; it ends at EOF on its pipe.
295            drop(task);
296            lock().mark_pump_late(
297                generation,
298                format!(
299                    "stderr of the exited process had not reached EOF {bound:?} after it was \
300                     retired (a descendant may still hold the pipe open); lines it still \
301                     writes are kept in that process's section"
302                ),
303            );
304            warn!(
305                module_id,
306                waited = ?bound,
307                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
308            );
309        }
310    }
311}
312
313fn registration_release_events() -> &'static watch::Sender<u64> {
314    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
315    EVENTS.get_or_init(|| {
316        let (sender, _receiver) = watch::channel(0);
317        sender
318    })
319}
320
321pub(crate) fn notify_registration_release() {
322    let events = registration_release_events();
323    let next_generation = (*events.borrow()).wrapping_add(1);
324    events.send_replace(next_generation);
325}
326
327/// How to launch one singleton module process.
328#[derive(Debug, Clone, PartialEq, Eq)]
329pub struct ModuleSpec {
330    pub module_id: String,
331    pub program: PathBuf,
332    pub args: Vec<String>,
333    pub env: Vec<(String, String)>,
334    /// When true this is a reserved module: each spawn gets a fresh one-time launch
335    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
336    /// process can register this module_id (a security-boundary module like the
337    /// credential vault must not be impersonable while it is down/restarting).
338    pub reserved: bool,
339    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
340    /// Prefixes come from daemon config and must end in `:` before they reach the
341    /// supervisor; the owner module's current spawn nonce authorizes claims under
342    /// each prefix.
343    pub reserved_prefixes: Vec<String>,
344    /// The wire protocol this module speaks, as DECLARED in daemon config.
345    ///
346    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
347    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
348    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
349    /// and NO launch nonce, and a clean exit the daemon did not request is
350    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
351    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
352    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
353    /// because a process ignores an environment variable it does not read.
354    ///
355    /// The argument is the part that cannot be "harmless to a process that
356    /// ignores it": a stock binary exits on an unknown flag before it listens
357    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
358    /// first conformance run against this mode found it. The nonce is withheld
359    /// because a process that will never present it gains nothing from holding
360    /// it, and a secret in the environment of a process that does not need it is
361    /// a leak surface for no benefit.
362    pub protocol: ModuleProtocol,
363    /// Whether two processes of this module may run at once, which is what a
364    /// blue/green swap does for the length of its overlap. Declared in daemon
365    /// config because the daemon must be able to answer it while the module is
366    /// down, and so a module cannot talk itself into it after registering.
367    pub overlap: ModuleOverlap,
368}
369
370/// Whether a module tolerates a second process of itself running alongside.
371///
372/// Most modules are single-writer on their store (a WAL, a capture log, a
373/// resident index behind a writer barrier), and two processes on one store
374/// corrupt it. So a swap, which overlaps the old and new process by design,
375/// is refused unless the module's config opts in with `overlap: "safe"`.
376#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
377pub enum ModuleOverlap {
378    /// Never run two processes of this module at once. The default.
379    #[default]
380    Exclusive,
381    /// The module has said a second process of itself is harmless for the
382    /// length of a swap.
383    ///
384    /// Declare it only if a second instance can run for a few seconds without
385    /// touching ANY single-writer store: every database, WAL, index, projector
386    /// and scheduled job the module owns. A lease on part of that state is not
387    /// enough. broca's session lease guards WAL appends while its run index, its
388    /// store projector and its archive fold timer (which unlinks live WAL files)
389    /// stay single-writer, so broca is exclusive despite holding a lease. The
390    /// refusal only fires after this has been decided, so the decision is the
391    /// check.
392    Safe,
393}
394
395impl ModuleOverlap {
396    pub fn as_str(self) -> &'static str {
397        match self {
398            Self::Exclusive => "exclusive",
399            Self::Safe => "safe",
400        }
401    }
402}
403
404/// Environment variable telling a spawned module which case it was started
405/// for, before it sends HELLO. Only a swap candidate carries it, as
406/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
407///
408/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
409/// longer because nobody waits on it, while a plain restart must flip ready
410/// quickly because callers see `module_warming` until it does. Absence means
411/// plain restart, the safe reading. The daemon trusts nothing about it; the
412/// candidate is proven by its launch nonce at HELLO.
413pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
414/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
415pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
416/// How long a swap waits for its candidate to register and declare itself
417/// ready when the operator does not say. A module warming as a swap candidate
418/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
419/// daemon allows that plus time to start the process and send HELLO.
420pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
421
422/// Bounded restart policy for crash exits.
423///
424/// `max_restarts` is the number of replacement processes allowed after the
425/// initial spawn WITHIN `window`. After that many crash restarts inside one
426/// window the module enters [`ModuleState::Failed`] and the supervisor stops
427/// the crash loop.
428///
429/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
430/// and that only survived because crashes were rare: a module that crashed
431/// three times across a week was disabled forever by crashes that had nothing
432/// to do with each other. That stopped being survivable once modules began
433/// exiting non-zero whenever the daemon's connection to them drops, because
434/// then every daemon-side connection drop spends a unit of the same budget and
435/// one flappy hour permanently stops a healthy module. Restarts older than
436/// `window` release their slot, so a module that crashed twice yesterday has a
437/// full budget today, while a genuine crash loop -- which is fast by
438/// definition -- still reaches the cap and stops.
439#[derive(Debug, Clone, Copy, PartialEq, Eq)]
440pub struct RestartPolicy {
441    pub max_restarts: u32,
442    /// Base delay before a crash replacement. The actual delay escalates with
443    /// the number of recent crash replacements and is capped by `max_backoff`.
444    pub backoff: Duration,
445    /// Maximum delay before a crash replacement.
446    pub max_backoff: Duration,
447    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
448    /// budget effectively infinite (nothing is ever in-window), which is why
449    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
450    pub window: Duration,
451}
452
453impl RestartPolicy {
454    /// A policy with the default crash window. Callers that care about the
455    /// window say so with [`Self::with_window`]; the ones that do not are
456    /// asking for the standard rate limit, not for no limit.
457    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
458        Self {
459            max_restarts,
460            backoff,
461            max_backoff: DEFAULT_MAX_BACKOFF,
462            window: DEFAULT_RESTART_WINDOW,
463        }
464    }
465
466    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
467        self.max_backoff = max_backoff;
468        self
469    }
470
471    pub fn with_window(mut self, window: Duration) -> Self {
472        self.window = window;
473        self
474    }
475
476    /// Calculate the capped exponential delay for the next crash replacement.
477    /// `restart_in_window` is zero for the first replacement after an operator
478    /// action (restart, reload, re-enable) cleared the crash ring, or after all
479    /// older crash replacements have aged out of the window.
480    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
481        if self.backoff.is_zero() || self.max_backoff.is_zero() {
482            return Duration::ZERO;
483        }
484
485        let mut delay = self.backoff;
486        for _ in 0..restart_in_window {
487            if delay >= self.max_backoff {
488                return self.max_backoff;
489            }
490            delay = delay
491                .checked_mul(10)
492                .unwrap_or(self.max_backoff)
493                .min(self.max_backoff);
494        }
495        delay.min(self.max_backoff)
496    }
497
498    /// The one sentence that explains a budget-exhausted stop, used for both the
499    /// log line and the terminal record so the two cannot drift. It names the
500    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
501    /// exactly what this budget is not.
502    fn budget_exhausted_detail(&self) -> String {
503        format!(
504            "crash budget exhausted: max_restarts={} within window_secs={}",
505            self.max_restarts,
506            self.window.as_secs()
507        )
508    }
509}
510
511impl Default for RestartPolicy {
512    fn default() -> Self {
513        Self {
514            max_restarts: DEFAULT_MAX_RESTARTS,
515            backoff: DEFAULT_BACKOFF,
516            max_backoff: DEFAULT_MAX_BACKOFF,
517            window: DEFAULT_RESTART_WINDOW,
518        }
519    }
520}
521
522#[derive(Debug, Clone, Copy, PartialEq, Eq)]
523struct CrashRestartSchedule {
524    restart_in_window: u32,
525    delay: Duration,
526}
527
528/// Whether the daemon itself will bring this module back after the exit being
529/// handled: it is enabled AND its in-window crash restarts are below the cap.
530///
531/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
532/// the window are dropped here rather than by a timer, so the count is right
533/// the moment somebody asks and no bookkeeping runs for idle modules.
534fn daemon_will_restart(
535    state: &mut SupervisorSnapshot,
536    policy: &RestartPolicy,
537    now: Instant,
538) -> bool {
539    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
540}
541
542const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
543const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
544const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
545const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
546
547#[derive(Debug, Clone, Copy, PartialEq, Eq)]
548pub enum HealthAction {
549    Report,
550    Restart,
551    Alert,
552}
553
554impl fmt::Display for HealthAction {
555    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
556        f.write_str(match self {
557            Self::Report => "report",
558            Self::Restart => "restart",
559            Self::Alert => "alert",
560        })
561    }
562}
563
564#[derive(Debug, Clone, PartialEq, Eq)]
565pub struct HealthConfig {
566    /// Optional loopback HTTP endpoint for a managed non-wire process.
567    /// Changing it applies live on rescan; the process protocol changes only
568    /// at its next spawn.
569    pub http: Option<String>,
570    pub cadence: Duration,
571    pub deadline: Duration,
572    pub failure_threshold: u32,
573    pub on_degraded: HealthAction,
574    pub on_failing: HealthAction,
575    pub critical: bool,
576}
577
578impl Default for HealthConfig {
579    fn default() -> Self {
580        Self {
581            http: None,
582            cadence: DEFAULT_HEALTH_CADENCE,
583            deadline: DEFAULT_HEALTH_DEADLINE,
584            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
585            on_degraded: HealthAction::Report,
586            on_failing: HealthAction::Report,
587            critical: false,
588        }
589    }
590}
591
592/// The supervisor's view of one module's health, relayed to clients over
593/// channel-0 and rendered by `ck health`.
594///
595/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
596/// stated here rather than only at the wire type a consumer reads. A reader can
597/// look up what `None` means; only a writer can silently change it, and the
598/// writer has no reason to go looking at a downstream contract before editing.
599///
600/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
601/// back to `None` on re-registration precisely so a respawned module does not
602/// carry its predecessor's timestamp — so an old value and an absent one call for
603/// opposite readings, and anything that defaulted this to a number would make a
604/// never-probed module indistinguishable from one probed at the epoch.
605///
606/// `detail` and `metrics` are `None` when the module published none on this
607/// probe, which does not mean it reported nothing wrong — it is also the shape
608/// when the probe never reached it. `last_probe_ms` is what separates those.
609#[derive(Debug, Clone, PartialEq)]
610pub struct ModuleHealthStatus {
611    pub status: SupervisorHealthStatus,
612    pub last_probe_ms: Option<u64>,
613    pub detail: Option<String>,
614    pub metrics: Option<Value>,
615    pub consecutive_failures: u32,
616    /// Number of replies received after a recurring health probe's deadline.
617    /// Unlike a timeout, every increment proves the module was alive.
618    pub late_answer_count: u64,
619    /// End-to-end latency of the newest late reply, measured from probe start.
620    pub last_late_answer_latency_ms: Option<u64>,
621    pub last_action: Option<String>,
622    /// Set together with `last_action`; the pair moves as one, and both being
623    /// absent means no escalation has ever been taken rather than that the last
624    /// one succeeded.
625    pub last_action_ms: Option<u64>,
626}
627
628impl Default for ModuleHealthStatus {
629    fn default() -> Self {
630        Self {
631            status: SupervisorHealthStatus::Unknown,
632            last_probe_ms: None,
633            detail: None,
634            metrics: None,
635            consecutive_failures: 0,
636            late_answer_count: 0,
637            last_late_answer_latency_ms: None,
638            last_action: None,
639            last_action_ms: None,
640        }
641    }
642}
643
644/// Typed lifecycle state for a supervised module.
645#[derive(Debug, Clone, Copy, PartialEq, Eq)]
646pub enum ModuleState {
647    Starting,
648    Running,
649    Unresponsive,
650    Restarting,
651    Draining,
652    Stopped,
653    Failed,
654    Disabled,
655}
656
657impl fmt::Display for ModuleState {
658    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
659        f.write_str(match self {
660            Self::Starting => "starting",
661            Self::Running => "running",
662            Self::Unresponsive => "unresponsive",
663            Self::Restarting => "restarting",
664            Self::Draining => "draining",
665            Self::Stopped => "stopped",
666            Self::Failed => "failed",
667            Self::Disabled => "disabled",
668        })
669    }
670}
671
672/// Supervisor classification of a child-process exit.
673#[derive(Debug, Clone, Copy, PartialEq, Eq)]
674pub enum ExitKind {
675    Clean,
676    Crash,
677    DeliberateSeverance,
678}
679
680impl From<ExitKind> for TerminalExitKind {
681    fn from(kind: ExitKind) -> Self {
682        match kind {
683            ExitKind::Clean => Self::Clean,
684            ExitKind::Crash => Self::Crash,
685            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
686        }
687    }
688}
689
690/// Exact process identity retained when a supervised module registers its
691/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
692#[derive(Debug, Clone, Copy, PartialEq, Eq)]
693pub(crate) struct ProcessIdentity {
694    pub(crate) pid: u32,
695    pub(crate) start_time: u64,
696}
697
698/// Last observed child exit, if any.
699#[derive(Debug, Clone, PartialEq, Eq)]
700pub struct ExitReport {
701    pub kind: ExitKind,
702    pub code: Option<i32>,
703    pub signal: Option<i32>,
704    pub at_ms: u64,
705}
706
707/// Point-in-time module status answerable by subc without forwarding to the
708/// module process.
709#[derive(Debug, Clone, PartialEq)]
710pub struct ModuleStatus {
711    pub module_id: String,
712    pub state: ModuleState,
713    pub enabled: bool,
714    pub process_alive: bool,
715    pub registration_active: bool,
716    /// The module's declared wire protocol, carried beside `live` because it is
717    /// what makes `live` readable: the two fields answer one question together.
718    /// While a process is alive this is its launch declaration, not a later
719    /// pending-reload edit. When down it is the configured next launch protocol.
720    pub protocol: ModuleProtocol,
721    /// Whether the module is serving, under the strongest definition the daemon
722    /// can assert for its protocol.
723    ///
724    /// A subc module must also be REGISTERED: its process being alive says
725    /// nothing about whether it can take a request. A `protocol: "none"` module
726    /// never registers, so that term is dropped and this falls back to "enabled,
727    /// running, and the process the daemon launched is alive" -- which is all
728    /// the daemon observes about a process that speaks no subc wire. It stays a
729    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
730    /// rather than printing it bare.
731    pub live: bool,
732    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
733    /// restarts have already released their slot, so this count can go down
734    /// without anybody touching the module.
735    pub restart_count: u32,
736    /// Replacement processes spawned over this module's entire supervisor lifetime;
737    /// unlike `restart_count`, this value is never reset by an operator action
738    /// and never falls out of a window.
739    pub lifetime_restarts: u32,
740    pub spawn_generation: u64,
741    /// The budget `restart_count` is spent against. Carried alongside the count
742    /// because the count alone does not say how close the module is to being
743    /// disabled, and reporting one without the other is what makes an
744    /// about-to-be-retired module look ordinary.
745    pub max_restarts: u32,
746    /// The span `restart_count` is counted over. Carried with the pair above for
747    /// the same reason they are carried together: "2 of 3" means one thing for a
748    /// ten-minute window and something else entirely for a lifetime.
749    pub restart_window: Duration,
750    /// Effective drain and restart timing policy used by this running module.
751    /// These values are carried together with the restart budget so status
752    /// readers can compare configured intent with what the supervisor applied.
753    pub drain_timeout: Duration,
754    pub restart_backoff: Duration,
755    pub restart_max_backoff: Duration,
756    pub pid: Option<u32>,
757    pub spawned_at_ms: Option<u64>,
758    pub spawned_from: Option<PathBuf>,
759    pub process_start_time: Option<u64>,
760    pub last_exit: Option<ExitReport>,
761    pub health: ModuleHealthStatus,
762}
763
764#[derive(Debug, Clone, PartialEq)]
765struct SupervisorSnapshot {
766    state: ModuleState,
767    enabled: bool,
768    process_alive: bool,
769    spawned_protocol: Option<ModuleProtocol>,
770    /// When each crash restart was spent, oldest first. This IS the crash
771    /// budget: its in-window length is the count an operator sees and the count
772    /// the restart decision is made against, so there is no second counter that
773    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
774    /// operator actions that used to zero the old lifetime counter.
775    crash_restarts: VecDeque<Instant>,
776    lifetime_restarts: u32,
777    /// Successful child spawns in this daemon incarnation.
778    ///
779    /// `lifetime_restarts` was considered and rejected: it starts at zero
780    /// (line 640), successful initial/operator spawns in `set_running` do not
781    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
782    /// increments before a successful replacement exists (lines 604, 3846,
783    /// and 3921), so a failed spawn can consume it. This counter moves only
784    /// when a live PID is accepted below.
785    spawn_generation: u64,
786    pid: Option<u32>,
787    /// Last reaped child, retained after current process facts are cleared.
788    reaped_pid: Option<u32>,
789    /// Whether the command-serving supervision loop has a scheduled respawn.
790    respawn_pending: bool,
791    /// A second restart is waiting for the replacement already scheduled.
792    coalesced_restart_pending: bool,
793    spawned_at_ms: Option<u64>,
794    spawned_from: Option<PathBuf>,
795    spawned_file_identity: Option<SpawnedFileIdentity>,
796    process_start_time: Option<u64>,
797    deliberate_severance: Option<ProcessIdentity>,
798    last_exit: Option<ExitReport>,
799    /// Diagnostic attached to the next drain's terminal record, if any.
800    drain_disposition_detail: Option<String>,
801    health: ModuleHealthStatus,
802    /// Whether the current process was started as a swap candidate and so
803    /// lives in the module's alternate cgroup. The next swap's candidate takes
804    /// the other one, so the two processes of a swap never share a cgroup. A
805    /// plain spawn always uses the primary cgroup.
806    in_alternate_slot: bool,
807    /// Whether the current `Draining` state ends in a replacement process
808    /// (restart, reload, health restart) rather than a stop. Only meaningful
809    /// while `state` is `Draining`; every entry into that state rewrites it.
810    /// It is what lets route.open answer the retryable `module_reloading` to a
811    /// consumer that reaches a still-registered process mid-restart, instead of
812    /// the `supervisor_not_live` a stop or disable deserves.
813    draining_to_replace: bool,
814    /// Whether a configuration update has been applied since the current
815    /// process was spawned, so that process runs an older spec than the one
816    /// the supervisor now holds. A queued restart is only coalesced into a
817    /// fresher process when this is false: a restart requested to pick up a
818    /// new configuration must not be satisfied by a process that predates it.
819    configuration_updated_since_spawn: bool,
820}
821
822impl SupervisorSnapshot {
823    fn starting() -> Self {
824        Self::new(ModuleState::Starting, true)
825    }
826
827    fn disabled() -> Self {
828        Self::new(ModuleState::Disabled, false)
829    }
830
831    fn failed() -> Self {
832        Self::new(ModuleState::Failed, true)
833    }
834
835    /// Crash restarts still inside `window`, having dropped the ones that are
836    /// not. Pruning on read is what makes the budget a rate: an instant older
837    /// than the window stops holding a slot the moment anybody counts.
838    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
839        while let Some(oldest) = self.crash_restarts.front() {
840            if now.duration_since(*oldest) > window {
841                self.crash_restarts.pop_front();
842            } else {
843                break;
844            }
845        }
846        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
847    }
848
849    /// Spend one unit of the crash budget and record the restart in the ledger.
850    ///
851    /// The ring is bounded by the cap because more than `max_restarts` in-window
852    /// instants can never be reached (the caller refuses the restart first), so
853    /// anything beyond that is an unbounded queue waiting to happen.
854    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
855        self.crash_restarts.push_back(now);
856        while self.crash_restarts.len() > policy.max_restarts as usize {
857            self.crash_restarts.pop_front();
858        }
859        self.lifetime_restarts += 1;
860    }
861
862    /// Reserve one crash-restart slot and calculate the delay before respawning.
863    /// The count is captured before recording this restart, so the first retry
864    /// uses the base delay and each later in-window retry escalates once.
865    fn next_crash_restart(
866        &mut self,
867        policy: &RestartPolicy,
868        now: Instant,
869    ) -> Option<CrashRestartSchedule> {
870        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
871        if restart_in_window >= policy.max_restarts {
872            return None;
873        }
874        self.record_crash_restart(policy, now);
875        Some(CrashRestartSchedule {
876            restart_in_window,
877            delay: policy.delay_for_restart(restart_in_window),
878        })
879    }
880
881    /// Give the module its full budget back, as an operator restart, reload, or
882    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
883    /// ledger of what actually happened, and an operator action does not unmake
884    /// the crashes.
885    fn clear_crash_restarts(&mut self) {
886        self.crash_restarts.clear();
887    }
888
889    fn new(state: ModuleState, enabled: bool) -> Self {
890        Self {
891            state,
892            enabled,
893            process_alive: false,
894            spawned_protocol: None,
895            crash_restarts: VecDeque::new(),
896            lifetime_restarts: 0,
897            spawn_generation: 0,
898            pid: None,
899            reaped_pid: None,
900            respawn_pending: false,
901            coalesced_restart_pending: false,
902            spawned_at_ms: None,
903            spawned_from: None,
904            spawned_file_identity: None,
905            process_start_time: None,
906            deliberate_severance: None,
907            last_exit: None,
908            drain_disposition_detail: None,
909            health: ModuleHealthStatus::default(),
910            in_alternate_slot: false,
911            draining_to_replace: false,
912            configuration_updated_since_spawn: false,
913        }
914    }
915}
916
917type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
918
919type SpawnSubscriberKey = (ConnectionId, u64);
920
921#[derive(Debug)]
922struct SpawnSubscriber {
923    version: u8,
924    frames: mpsc::Sender<Frame>,
925    /// Tells this subscriber's forwarder that it was dropped for lagging, and
926    /// from which event. The full frame channel cannot carry that news, so it
927    /// travels beside it; see `SpawnEventFeed::subscribe`.
928    lagged: Option<oneshot::Sender<SpawnCursor>>,
929}
930
931#[derive(Debug)]
932struct SpawnEventState {
933    daemon_incarnation: String,
934    seq: u64,
935    capacity: usize,
936    live: HashMap<String, LiveSpawn>,
937    generations: HashMap<String, u64>,
938    events: VecDeque<SpawnEvent>,
939    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
940}
941
942impl Default for SpawnEventState {
943    fn default() -> Self {
944        Self {
945            daemon_incarnation: "unconfigured".to_string(),
946            seq: 0,
947            capacity: SPAWN_EVENT_RING_CAPACITY,
948            live: HashMap::new(),
949            generations: HashMap::new(),
950            events: VecDeque::new(),
951            subscribers: HashMap::new(),
952        }
953    }
954}
955
956#[derive(Debug, Clone, Default)]
957struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
958
959#[derive(Debug, Clone, PartialEq, Eq)]
960pub(crate) enum SpawnSubscribeRefusal {
961    ForeignIncarnation { current: String },
962    TooOld { oldest: SpawnCursor },
963    Frame(String),
964}
965
966impl SpawnEventFeed {
967    fn configure_incarnation(&self, daemon_incarnation: String) {
968        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
969        state.daemon_incarnation = daemon_incarnation;
970        state.seq = 0;
971        state.live.clear();
972        state.generations.clear();
973        state.events.clear();
974        state.subscribers.clear();
975    }
976
977    fn cursor(state: &SpawnEventState) -> SpawnCursor {
978        SpawnCursor {
979            daemon_incarnation: state.daemon_incarnation.clone(),
980            seq: state.seq,
981        }
982    }
983
984    fn snapshot(&self) -> SpawnSnapshot {
985        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
986        let mut live = state.live.values().cloned().collect::<Vec<_>>();
987        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
988        SpawnSnapshot {
989            cursor: Self::cursor(&state),
990            ring_bound: state.capacity as u64,
991            live,
992        }
993    }
994
995    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
996        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
997        let generation = state
998            .generations
999            .get(module_id)
1000            .copied()
1001            .unwrap_or(0)
1002            .checked_add(1)
1003            .expect("spawn generation exhausted");
1004        state.generations.insert(module_id.to_string(), generation);
1005        let live = LiveSpawn {
1006            module_id: module_id.to_string(),
1007            spawn_generation: generation,
1008            pid,
1009            spawned_at_ms,
1010        };
1011        state.live.insert(module_id.to_string(), live);
1012        Self::emit_locked(
1013            &mut state,
1014            SpawnEventKind::Spawned,
1015            module_id.to_string(),
1016            generation,
1017            pid,
1018            None,
1019            None,
1020        );
1021        generation
1022    }
1023
1024    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1025        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1026        let Some(live) = state.live.remove(module_id) else {
1027            warn!(
1028                module_id,
1029                "terminal record had no live spawn event identity"
1030            );
1031            return;
1032        };
1033        Self::emit_locked(
1034            &mut state,
1035            SpawnEventKind::Exited,
1036            module_id.to_string(),
1037            live.spawn_generation,
1038            live.pid,
1039            exit_code,
1040            exit_signal,
1041        );
1042    }
1043
1044    /// Report the exit of a process that a swap has already replaced.
1045    ///
1046    /// `emit_exited` removes the module's live entry, which after a swap's
1047    /// cutover describes the promoted candidate, not the old process now
1048    /// exiting. This emits the old generation's exit and leaves the live entry
1049    /// alone unless it still names that generation.
1050    fn emit_superseded_exited(
1051        &self,
1052        module_id: &str,
1053        spawn_generation: u64,
1054        pid: u32,
1055        exit_code: Option<i32>,
1056        exit_signal: Option<i32>,
1057    ) {
1058        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1059        if state
1060            .live
1061            .get(module_id)
1062            .is_some_and(|live| live.spawn_generation == spawn_generation)
1063        {
1064            state.live.remove(module_id);
1065        }
1066        Self::emit_locked(
1067            &mut state,
1068            SpawnEventKind::Exited,
1069            module_id.to_string(),
1070            spawn_generation,
1071            pid,
1072            exit_code,
1073            exit_signal,
1074        );
1075    }
1076
1077    #[allow(clippy::too_many_arguments)]
1078    fn emit_locked(
1079        state: &mut SpawnEventState,
1080        kind: SpawnEventKind,
1081        module_id: String,
1082        spawn_generation: u64,
1083        pid: u32,
1084        exit_code: Option<i32>,
1085        exit_signal: Option<i32>,
1086    ) {
1087        state.seq = state
1088            .seq
1089            .checked_add(1)
1090            .expect("spawn event sequence exhausted");
1091        let event = SpawnEvent {
1092            cursor: Self::cursor(state),
1093            kind,
1094            module_id,
1095            spawn_generation,
1096            pid,
1097            exit_code,
1098            exit_signal,
1099        };
1100        state.events.push_back(event.clone());
1101        while state.events.len() > state.capacity {
1102            state.events.pop_front();
1103        }
1104        let body = match serde_json::to_vec(&event) {
1105            Ok(body) => body,
1106            Err(error) => {
1107                error!(%error, "failed to serialize supervisor spawn event");
1108                return;
1109            }
1110        };
1111        state.subscribers.retain(|(connection_id, corr), subscriber| {
1112            let frame = Frame::build_with_version(
1113                subscriber.version,
1114                FrameType::StreamData,
1115                control_flags(),
1116                0,
1117                0,
1118                *corr,
1119                body.clone(),
1120            );
1121            match frame {
1122                Ok(frame) => {
1123                    if subscriber.frames.try_send(frame).is_ok() {
1124                        true
1125                    } else {
1126                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1127                        if let Some(lagged) = subscriber.lagged.take() {
1128                            let _ = lagged.send(event.cursor.clone());
1129                        }
1130                        false
1131                    }
1132                }
1133                Err(error) => {
1134                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1135                    false
1136                }
1137            }
1138        });
1139    }
1140
1141    fn subscribe(
1142        &self,
1143        connection_id: ConnectionId,
1144        corr: u64,
1145        version: u8,
1146        since: Option<SpawnCursor>,
1147        sink: FrameSink,
1148    ) -> Result<(), SpawnSubscribeRefusal> {
1149        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1150        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1151        {
1152            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1153            let replay = if let Some(since) = since {
1154                if since.daemon_incarnation != state.daemon_incarnation {
1155                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1156                        current: state.daemon_incarnation.clone(),
1157                    });
1158                }
1159                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1160                    if since.seq < oldest.seq.saturating_sub(1) {
1161                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1162                    }
1163                }
1164                state
1165                    .events
1166                    .iter()
1167                    .filter(|event| event.cursor.seq > since.seq)
1168                    .cloned()
1169                    .collect::<Vec<_>>()
1170            } else {
1171                Vec::new()
1172            };
1173            for event in replay {
1174                let body = serde_json::to_vec(&event)
1175                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1176                let frame = Frame::build_with_version(
1177                    version,
1178                    FrameType::StreamData,
1179                    control_flags(),
1180                    0,
1181                    0,
1182                    corr,
1183                    body,
1184                )
1185                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1186                frames
1187                    .try_send(frame)
1188                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1189            }
1190            state.subscribers.insert(
1191                (connection_id, corr),
1192                SpawnSubscriber {
1193                    version,
1194                    frames,
1195                    lagged: Some(lagged),
1196                },
1197            );
1198        }
1199        // The lagged terminal is sent here, by the forwarder, rather than by
1200        // the emitter: at the moment of the drop the subscriber's own channel
1201        // is full, and writing to the connection sink directly from the emitter
1202        // would put the Error AHEAD of the events still queued in that channel
1203        // (and the emitter holds the feed lock, so it cannot await the sink).
1204        // Dropping the subscriber drops the only sender, so `recv` drains every
1205        // queued event and then returns `None`; only then is the Error sent, so
1206        // the client sees each event it can keep, then the reason it was cut.
1207        // Cancel and connection removal drop the oneshot unsent, so they end
1208        // the stream with no Error.
1209        tokio::spawn(async move {
1210            while let Some(frame) = receiver.recv().await {
1211                if sink.send(frame).await.is_err() {
1212                    return;
1213                }
1214            }
1215            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1216                return;
1217            };
1218            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1219                Ok(frame) => {
1220                    let _ = sink.send(frame).await;
1221                }
1222                Err(error) => {
1223                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1224                }
1225            }
1226        });
1227        Ok(())
1228    }
1229
1230    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1231        let Some(subscriber) = self
1232            .0
1233            .lock()
1234            .unwrap_or_else(|p| p.into_inner())
1235            .subscribers
1236            .remove(&(connection_id, corr))
1237        else {
1238            return false;
1239        };
1240        if let Ok(frame) = Frame::build_with_version(
1241            subscriber.version,
1242            FrameType::StreamEnd,
1243            control_flags(),
1244            0,
1245            0,
1246            corr,
1247            Vec::new(),
1248        ) {
1249            tokio::spawn(async move {
1250                let _ = subscriber.frames.send(frame).await;
1251            });
1252        }
1253        true
1254    }
1255
1256    fn remove_connection(&self, connection_id: ConnectionId) {
1257        self.0
1258            .lock()
1259            .unwrap_or_else(|p| p.into_inner())
1260            .subscribers
1261            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1262    }
1263
1264    #[cfg(any(test, feature = "test-support"))]
1265    fn set_capacity(&self, capacity: usize) {
1266        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1267    }
1268
1269    #[cfg(any(test, feature = "test-support"))]
1270    fn subscriber_count(&self) -> usize {
1271        self.0
1272            .lock()
1273            .unwrap_or_else(|p| p.into_inner())
1274            .subscribers
1275            .len()
1276    }
1277}
1278
1279/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1280/// The terminal Error a lagged spawn subscriber receives after its queued events.
1281fn spawn_subscriber_lagged_frame(
1282    version: u8,
1283    corr: u64,
1284    first_undelivered: SpawnCursor,
1285) -> Result<Frame, String> {
1286    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1287        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1288        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1289            .to_string(),
1290        detail: Some(serde_json::json!({
1291            "first_undelivered_cursor": first_undelivered
1292        })),
1293    })
1294    .map_err(|error| error.to_string())?;
1295    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1296        .map_err(|error| error.to_string())
1297}
1298
1299pub trait ModuleProcessLiveness: Send + Sync {
1300    fn process_live(&self, module_id: &str) -> Option<bool>;
1301
1302    /// Whether the supervisor is replacing this module's process right now: an
1303    /// operator restart or reload, a health restart, or a crash respawn whose
1304    /// backoff is running. A module in that state is not live, but a consumer
1305    /// refused now should retry shortly rather than treat the target as gone.
1306    /// Stopped, failed, and disabled modules are not replacing.
1307    fn process_replacing(&self, _module_id: &str) -> bool {
1308        false
1309    }
1310}
1311
1312/// Shared process-liveness registry keyed by supervised `module_id`.
1313#[derive(Debug, Clone, Default)]
1314pub struct SupervisorProcessLiveness {
1315    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1316}
1317
1318impl SupervisorProcessLiveness {
1319    pub fn new() -> Self {
1320        Self::default()
1321    }
1322
1323    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1324        let mut snapshots = self
1325            .snapshots
1326            .lock()
1327            .unwrap_or_else(|poisoned| poisoned.into_inner());
1328        snapshots.insert(module_id, snapshot);
1329    }
1330
1331    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1332        let mut snapshots = self
1333            .snapshots
1334            .lock()
1335            .unwrap_or_else(|poisoned| poisoned.into_inner());
1336        let is_current = snapshots
1337            .get(module_id)
1338            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1339            .unwrap_or(false);
1340        if is_current {
1341            snapshots.remove(module_id);
1342        }
1343    }
1344}
1345
1346impl ModuleProcessLiveness for SupervisorProcessLiveness {
1347    fn process_live(&self, module_id: &str) -> Option<bool> {
1348        let snapshot = {
1349            let snapshots = self
1350                .snapshots
1351                .lock()
1352                .unwrap_or_else(|poisoned| poisoned.into_inner());
1353            snapshots.get(module_id).cloned()
1354        }?;
1355        let snapshot = snapshot
1356            .lock()
1357            .unwrap_or_else(|poisoned| poisoned.into_inner());
1358        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1359    }
1360
1361    fn process_replacing(&self, module_id: &str) -> bool {
1362        let Some(snapshot) = self
1363            .snapshots
1364            .lock()
1365            .unwrap_or_else(|poisoned| poisoned.into_inner())
1366            .get(module_id)
1367            .cloned()
1368        else {
1369            return false;
1370        };
1371        let snapshot = snapshot
1372            .lock()
1373            .unwrap_or_else(|poisoned| poisoned.into_inner());
1374        snapshot.enabled
1375            && match snapshot.state {
1376                ModuleState::Restarting => true,
1377                ModuleState::Draining => snapshot.draining_to_replace,
1378                ModuleState::Starting
1379                | ModuleState::Running
1380                | ModuleState::Unresponsive
1381                | ModuleState::Stopped
1382                | ModuleState::Failed
1383                | ModuleState::Disabled => false,
1384            }
1385    }
1386}
1387
1388#[cfg(test)]
1389#[derive(Debug, Default)]
1390struct ReloadExitRecordGate {
1391    reached: tokio::sync::Notify,
1392    resume: tokio::sync::Notify,
1393}
1394
1395#[derive(Debug, Clone, Copy)]
1396enum RespawnKind {
1397    Spawn,
1398    Reload,
1399}
1400
1401#[derive(Debug, Clone, Copy)]
1402struct PendingRespawn {
1403    deadline: Instant,
1404    kind: RespawnKind,
1405}
1406
1407type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1408
1409#[derive(Debug, Clone)]
1410struct SupervisorRuntimeConfig {
1411    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1412    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1413    /// A reload acknowledges completion only after its replacement registers.
1414    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1415    restart_policy: RestartPolicy,
1416    /// This module's RESOLVED drain budget: per-module config when present,
1417    /// else `default_drain_timeout`.
1418    drain_timeout: Duration,
1419    /// Shared with the status handle so the attested value changes atomically
1420    /// when a rescan updates the running drain policy.
1421    effective_drain_timeout: Arc<Mutex<Duration>>,
1422    /// The supervisor-wide fallback, kept so a configuration update that
1423    /// REMOVES the per-module override can re-resolve to it.
1424    default_drain_timeout: Duration,
1425    health: HealthConfig,
1426    connection_file_path: Option<PathBuf>,
1427    capture_logs_dir: Option<PathBuf>,
1428    forwarding: Option<Arc<ForwardingTable>>,
1429    /// The shared handle, so every spawn path (initial, restart, reload) records the
1430    /// reserved-module launch nonce the HELLO verifier checks against.
1431    supervisor_handle: Option<SupervisorHandle>,
1432    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1433    /// status queries.
1434    ///
1435    /// One ring per module, held across every respawn. The lines explaining an exit
1436    /// are written BEFORE that exit, so a ring recreated per process would be empty
1437    /// exactly when it is asked for.
1438    stderr_ring: Arc<Mutex<StderrRing>>,
1439    terminal_ring: Arc<Mutex<TerminalRing>>,
1440    spawn_events: SpawnEventFeed,
1441    child_roster: ChildRoster,
1442    #[cfg(target_os = "linux")]
1443    cgroup_placement: Option<subc_cgroup::Placement>,
1444    #[cfg(test)]
1445    test_seed_stale_facts_before_enable_spawn: bool,
1446    #[cfg(test)]
1447    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1448}
1449
1450#[derive(Debug, Clone, PartialEq, Eq)]
1451struct SupervisedConfiguration {
1452    spec: ModuleSpec,
1453    health: HealthConfig,
1454}
1455
1456/// Shared daemon lookup table for supervised module handles.
1457///
1458/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1459/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1460/// launch nonces recorded at spawn are checked by the same daemon instance.
1461#[derive(Debug, Clone, Default)]
1462pub struct SupervisorHandle {
1463    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1464    /// Module ids the supervisor has taken on. An id is added BEFORE the
1465    /// module's first process is spawned and removed only when the module
1466    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1467    /// the keys of `modules`.
1468    ///
1469    /// `modules` cannot answer "is this module configured?" on its own: a
1470    /// [`SupervisedModule`] only exists once its process has been spawned, and
1471    /// a fast child can connect, register, sync its scopes and ask about them
1472    /// before the supervisor has inserted it. Answering "not configured" in that
1473    /// gap makes scope admission refuse with the terminal "will never sync"
1474    /// instead of the retryable "has not synced yet".
1475    configured_ids: Arc<Mutex<HashSet<String>>>,
1476    spawn_events: SpawnEventFeed,
1477    /// The current expected launch nonce for each reserved module_id. Set when the
1478    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1479    /// non-reserved module never has an entry here and is never nonce-checked.
1480    /// Reserved module ids and the nonce that authorizes their next HELLO.
1481    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1482    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1483    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1484    /// had NO entry and admitted anyone: the reservation protected the nonce
1485    /// holder, not the NAME (found live by CKCRED's canary probe registering
1486    /// against a reserved scratch id).
1487    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1488    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1489    ///
1490    /// This is deliberately in-memory only: subc is state-free across daemon
1491    /// restarts, and the tombstone only explains the hours-after-removal window
1492    /// while this executing daemon is still alive. Do not persist it in a store.
1493    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1494    /// The current launch nonce for every supervised spawn. This is separate from
1495    /// reserved_nonces because consumer route.open attestation applies to all spawned
1496    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1497    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1498    /// Reserved namespace prefixes mapped to the supervised owner module whose
1499    /// current spawn nonce authorizes HELLO claims below the prefix.
1500    ///
1501    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1502    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1503    /// accidental collisions and lower-trust processes from squatting protected
1504    /// namespaces.
1505    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1506    /// Blue/green swaps in progress, by module id. An entry exists from just
1507    /// before the candidate process is spawned until the swap has failed, or
1508    /// has cut over and the old process is gone. While it exists, HELLO for the
1509    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1510    /// consumer attestation accepts both processes' nonces.
1511    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1512    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1513    promotion_observer: PromotionObserverSlot,
1514    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1515    /// this daemon-wide ordering, a rescan could retire or update a module while a
1516    /// concurrent reload still held its old handle and launch specification.
1517    operation_lock: Arc<AsyncMutex<()>>,
1518}
1519
1520/// Told when a swap has promoted its candidate to be the module's active
1521/// registration.
1522///
1523/// An ordinary HELLO runs the control plane's registration side effects (the
1524/// capability cache, the deny census, the requirement recompute) as it
1525/// registers. A swap candidate's HELLO does not, because it is not routable;
1526/// promotion is when those must run instead, and promotion happens in the
1527/// supervisor, which has no other way into the control handler.
1528pub(crate) trait SwapPromotionObserver: Send + Sync {
1529    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1530}
1531
1532/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1533/// control handler) owns this handle, so a strong reference back would be a
1534/// cycle that keeps both alive.
1535#[derive(Clone, Default)]
1536struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1537
1538impl fmt::Debug for PromotionObserverSlot {
1539    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1540        f.write_str("PromotionObserverSlot")
1541    }
1542}
1543
1544/// The nonces of one open swap.
1545#[derive(Debug, Clone)]
1546struct OpenSwap {
1547    /// The launch nonce minted for the candidate process. It is the swap
1548    /// token: the only thing that admits a HELLO into the candidate slot.
1549    candidate_nonce: String,
1550    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1551    /// here because cutover moves the module's recorded spawn nonce to the
1552    /// candidate while the incumbent is still draining and its consumers are
1553    /// still attesting with this one.
1554    incumbent_nonce: Option<String>,
1555    /// Set once a HELLO has been admitted with the swap token, so the token
1556    /// admits one registration and cannot be replayed after cutover empties
1557    /// the candidate slot.
1558    candidate_admitted: bool,
1559}
1560
1561/// What the swap gate says about a HELLO. See
1562/// [`SupervisorHandle::swap_hello_admission`].
1563#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1564pub(crate) enum SwapHelloAdmission {
1565    /// No swap is open for the id (or the HELLO carries the incumbent's own
1566    /// nonce); the ordinary gates decide.
1567    NotSwapping,
1568    /// The HELLO carries the swap token: register it into the candidate slot.
1569    Candidate,
1570    /// A swap is open and the HELLO carries a nonce the supervisor did not
1571    /// mint for this id, no nonce, or a token already used.
1572    Refused,
1573}
1574
1575#[derive(Debug, Clone, PartialEq, Eq)]
1576pub(crate) enum ReservedHelloRejection {
1577    Exact {
1578        module_id: String,
1579    },
1580    Prefix {
1581        prefix: String,
1582        owner_module_id: String,
1583    },
1584}
1585
1586impl SupervisorHandle {
1587    pub fn new() -> Self {
1588        Self::default()
1589    }
1590
1591    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1592        self.spawn_events.snapshot()
1593    }
1594
1595    pub(crate) fn subscribe_spawns(
1596        &self,
1597        connection_id: ConnectionId,
1598        corr: u64,
1599        version: u8,
1600        since: Option<SpawnCursor>,
1601        sink: FrameSink,
1602    ) -> Result<(), SpawnSubscribeRefusal> {
1603        self.spawn_events
1604            .subscribe(connection_id, corr, version, since, sink)
1605    }
1606
1607    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1608        self.spawn_events.cancel(connection_id, corr)
1609    }
1610
1611    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1612        self.spawn_events.remove_connection(connection_id);
1613    }
1614
1615    #[cfg(any(test, feature = "test-support"))]
1616    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1617        assert!(capacity > 0, "spawn event capacity must be non-zero");
1618        self.spawn_events.set_capacity(capacity);
1619    }
1620
1621    #[cfg(any(test, feature = "test-support"))]
1622    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1623        self.spawn_events.subscriber_count()
1624    }
1625
1626    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1627    /// a respawn invalidates stale consumer identities.
1628    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1629        self.spawn_nonces
1630            .lock()
1631            .unwrap_or_else(|poisoned| poisoned.into_inner())
1632            .insert(module_id.to_string(), nonce);
1633    }
1634
1635    /// Record the launch nonce expected from the next HELLO for a reserved module,
1636    /// replacing any prior nonce (a respawn invalidates the previous one).
1637    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1638        self.reserved_nonces
1639            .lock()
1640            .unwrap_or_else(|poisoned| poisoned.into_inner())
1641            .insert(module_id.to_string(), Some(nonce));
1642    }
1643
1644    /// Record namespace prefixes owned by a supervised module.
1645    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1646        let mut owners = self
1647            .reserved_prefix_owners
1648            .lock()
1649            .unwrap_or_else(|poisoned| poisoned.into_inner());
1650        owners.retain(|_, owner| owner != owner_module_id);
1651        for prefix in prefixes {
1652            owners.insert(prefix.clone(), owner_module_id.to_string());
1653        }
1654    }
1655
1656    /// The launch nonce most recently minted for a module's spawn, if any.
1657    #[cfg(test)]
1658    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1659        self.spawn_nonces
1660            .lock()
1661            .unwrap_or_else(|poisoned| poisoned.into_inner())
1662            .get(module_id)
1663            .cloned()
1664    }
1665
1666    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1667        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1668        let spawn_nonce = self
1669            .spawn_nonces
1670            .lock()
1671            .unwrap_or_else(|poisoned| poisoned.into_inner())
1672            .get(&spec.module_id)
1673            .cloned();
1674        let mut reserved_nonces = self
1675            .reserved_nonces
1676            .lock()
1677            .unwrap_or_else(|poisoned| poisoned.into_inner());
1678        if spec.reserved {
1679            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1680            // reserved name whose module has never spawned has no legitimate
1681            // holder, and the entry's absence is what used to leave the name
1682            // open to the first claimant.
1683            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1684        }
1685        drop(reserved_nonces);
1686        // A later unreserved declaration must not silently unreserve an id that
1687        // was retained after its reserved configuration was removed. The explicit
1688        // release ceremony is the only operation that retires that gate.
1689        self.removal_tombstones
1690            .lock()
1691            .unwrap_or_else(|poisoned| poisoned.into_inner())
1692            .remove(&spec.module_id);
1693    }
1694
1695    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
1696    /// authorized only by its expected nonce; otherwise a matching reserved prefix
1697    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
1698    /// with no matching prefix are always authorized.
1699    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1700        self.reserved_hello_rejection(module_id, presented)
1701            .is_none()
1702    }
1703
1704    pub(crate) fn reserved_hello_rejection(
1705        &self,
1706        module_id: &str,
1707        presented: Option<&str>,
1708    ) -> Option<ReservedHelloRejection> {
1709        let nonces = self
1710            .reserved_nonces
1711            .lock()
1712            .unwrap_or_else(|poisoned| poisoned.into_inner());
1713        if let Some(expected) = nonces.get(module_id) {
1714            // `None` = reserved with no legitimate holder: refuse every
1715            // presentation, because no process can hold a nonce that was never
1716            // minted. Only a real minted nonce admits, in constant time.
1717            let authorized = match expected {
1718                Some(expected) => {
1719                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1720                }
1721                None => false,
1722            };
1723            if authorized {
1724                return None;
1725            }
1726            return Some(ReservedHelloRejection::Exact {
1727                module_id: module_id.to_string(),
1728            });
1729        }
1730        drop(nonces);
1731
1732        let matched_prefix = self
1733            .reserved_prefix_owners
1734            .lock()
1735            .unwrap_or_else(|poisoned| poisoned.into_inner())
1736            .iter()
1737            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
1738            .max_by_key(|(prefix, _)| prefix.len())
1739            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
1740        let (prefix, owner_module_id) = matched_prefix?;
1741
1742        let authorized = presented.is_some_and(|presented| {
1743            self.spawn_nonces
1744                .lock()
1745                .unwrap_or_else(|poisoned| poisoned.into_inner())
1746                .get(&owner_module_id)
1747                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
1748                // While the owner is being swapped, children started by
1749                // either of its two processes hold that process's nonce.
1750                || self.swap_nonce_matches(&owner_module_id, presented)
1751        });
1752        if authorized {
1753            None
1754        } else {
1755            Some(ReservedHelloRejection::Prefix {
1756                prefix,
1757                owner_module_id,
1758            })
1759        }
1760    }
1761
1762    /// Whether a consumer connection proved it came from a daemon-spawned module.
1763    ///
1764    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
1765    /// accepted only for module ids the supervisor has spawned.
1766    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
1767        if presented.is_empty() {
1768            return false;
1769        }
1770        let nonces = self
1771            .spawn_nonces
1772            .lock()
1773            .unwrap_or_else(|poisoned| poisoned.into_inner());
1774        let current = nonces
1775            .get(module_id)
1776            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
1777        drop(nonces);
1778        // During a swap two processes of the module are alive, and a consumer
1779        // started by either one presents that process's nonce. Accepting only
1780        // the recorded one would fail the incumbent's consumers for the whole
1781        // overlap once cutover moves the record to the candidate.
1782        current || self.swap_nonce_matches(module_id, presented)
1783    }
1784
1785    /// Whether `presented` is either nonce of an open swap for `module_id`.
1786    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
1787        let swaps = self
1788            .swaps
1789            .lock()
1790            .unwrap_or_else(|poisoned| poisoned.into_inner());
1791        swaps.get(module_id).is_some_and(|swap| {
1792            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
1793                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
1794                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
1795                })
1796        })
1797    }
1798
1799    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
1800    /// Called before the candidate process exists.
1801    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
1802        let incumbent_nonce = self
1803            .spawn_nonces
1804            .lock()
1805            .unwrap_or_else(|poisoned| poisoned.into_inner())
1806            .get(module_id)
1807            .cloned();
1808        self.swaps
1809            .lock()
1810            .unwrap_or_else(|poisoned| poisoned.into_inner())
1811            .insert(
1812                module_id.to_string(),
1813                OpenSwap {
1814                    candidate_nonce,
1815                    incumbent_nonce,
1816                    candidate_admitted: false,
1817                },
1818            );
1819    }
1820
1821    /// Close the swap for `module_id`, releasing whichever nonce is no longer
1822    /// the module's recorded one.
1823    pub(crate) fn close_swap(&self, module_id: &str) {
1824        self.swaps
1825            .lock()
1826            .unwrap_or_else(|poisoned| poisoned.into_inner())
1827            .remove(module_id);
1828    }
1829
1830    /// Install the observer told about swap promotions, replacing any earlier
1831    /// one.
1832    pub(crate) fn set_swap_promotion_observer(
1833        &self,
1834        observer: std::sync::Weak<dyn SwapPromotionObserver>,
1835    ) {
1836        *self
1837            .promotion_observer
1838            .0
1839            .lock()
1840            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
1841    }
1842
1843    /// Tell the installed observer, if it is still alive, that a swap promoted
1844    /// `registration`.
1845    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
1846        let observer = self
1847            .promotion_observer
1848            .0
1849            .lock()
1850            .unwrap_or_else(|poisoned| poisoned.into_inner())
1851            .as_ref()
1852            .and_then(std::sync::Weak::upgrade);
1853        if let Some(observer) = observer {
1854            observer.swap_promoted(registration);
1855        }
1856    }
1857
1858    /// Whether a swap is open for `module_id`.
1859    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
1860        self.swaps
1861            .lock()
1862            .unwrap_or_else(|poisoned| poisoned.into_inner())
1863            .contains_key(module_id)
1864    }
1865
1866    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
1867    /// respawn would, once cutover has made the candidate the module's process.
1868    /// The swap stays open so the incumbent's nonce keeps attesting until the
1869    /// incumbent has drained and exited.
1870    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
1871        let candidate_nonce = self
1872            .swaps
1873            .lock()
1874            .unwrap_or_else(|poisoned| poisoned.into_inner())
1875            .get(module_id)
1876            .map(|swap| swap.candidate_nonce.clone());
1877        let Some(nonce) = candidate_nonce else {
1878            return;
1879        };
1880        self.set_spawn_nonce(module_id, nonce.clone());
1881        if reserved {
1882            self.set_reserved_nonce(module_id, nonce);
1883        }
1884    }
1885
1886    /// The swap gate for a HELLO claiming `module_id`.
1887    ///
1888    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
1889    /// presents the candidate nonce, which the reserved gate (holding the
1890    /// incumbent's nonce) would refuse as `reserved_module` before swap
1891    /// admission was ever reached. And it applies to unreserved ids too: for an
1892    /// unreserved id the only thing that ever stopped a second process claiming
1893    /// a live id was the `duplicate_module_id` refusal, which is exactly the
1894    /// refusal a swap lifts for its candidate.
1895    ///
1896    /// The incumbent's own nonce falls through to the ordinary gates, which
1897    /// treat it as they always have (a live incumbent is refused as a
1898    /// duplicate). Anything else while a swap is open is refused, including an
1899    /// absent nonce.
1900    pub(crate) fn swap_hello_admission(
1901        &self,
1902        module_id: &str,
1903        presented: Option<&str>,
1904    ) -> SwapHelloAdmission {
1905        let swaps = self
1906            .swaps
1907            .lock()
1908            .unwrap_or_else(|poisoned| poisoned.into_inner());
1909        let Some(swap) = swaps.get(module_id) else {
1910            return SwapHelloAdmission::NotSwapping;
1911        };
1912        let Some(presented) = presented else {
1913            return SwapHelloAdmission::Refused;
1914        };
1915        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
1916            return if swap.candidate_admitted {
1917                SwapHelloAdmission::Refused
1918            } else {
1919                SwapHelloAdmission::Candidate
1920            };
1921        }
1922        if swap
1923            .incumbent_nonce
1924            .as_deref()
1925            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
1926        {
1927            return SwapHelloAdmission::NotSwapping;
1928        }
1929        SwapHelloAdmission::Refused
1930    }
1931
1932    /// Record that the swap token has registered a candidate, so it admits no
1933    /// second HELLO.
1934    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
1935        if let Some(swap) = self
1936            .swaps
1937            .lock()
1938            .unwrap_or_else(|poisoned| poisoned.into_inner())
1939            .get_mut(module_id)
1940        {
1941            swap.candidate_admitted = true;
1942        }
1943    }
1944
1945    /// Test/support lookup for the current launch nonce of a supervised spawn.
1946    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1947        self.spawn_nonces
1948            .lock()
1949            .unwrap_or_else(|poisoned| poisoned.into_inner())
1950            .get(module_id)
1951            .cloned()
1952    }
1953
1954    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
1955    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1956        self.reserved_nonces
1957            .lock()
1958            .unwrap_or_else(|poisoned| poisoned.into_inner())
1959            .get(module_id)
1960            .cloned()
1961            .flatten()
1962    }
1963
1964    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
1965        // Normally already marked before the process was spawned; marking here
1966        // too keeps `configured_ids` a superset of the roster for any caller
1967        // that inserts a module directly.
1968        self.mark_configured(module.module_id());
1969        let mut modules = self
1970            .modules
1971            .lock()
1972            .unwrap_or_else(|poisoned| poisoned.into_inner());
1973        modules.insert(module.module_id().to_string(), module)
1974    }
1975
1976    /// Record that the supervisor has taken on `module_id`. Called before the
1977    /// module's first process is spawned, so that by the time that process can
1978    /// register, [`Self::is_configured`] already answers true.
1979    fn mark_configured(&self, module_id: &str) {
1980        self.configured_ids
1981            .lock()
1982            .unwrap_or_else(|poisoned| poisoned.into_inner())
1983            .insert(module_id.to_string());
1984    }
1985
1986    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
1987    /// before it was ever put on the roster. A module already on the roster
1988    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
1989    fn unmark_configured_unless_rostered(&self, module_id: &str) {
1990        let modules = self
1991            .modules
1992            .lock()
1993            .unwrap_or_else(|poisoned| poisoned.into_inner());
1994        if !modules.contains_key(module_id) {
1995            self.configured_ids
1996                .lock()
1997                .unwrap_or_else(|poisoned| poisoned.into_inner())
1998                .remove(module_id);
1999        }
2000    }
2001
2002    /// Whether `module_id` is a module this daemon supervises: on the roster,
2003    /// or about to be (its process is being spawned right now).
2004    ///
2005    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2006    /// for scopes: a supervised module's process can register and sync before
2007    /// [`Self::get`] can return it, and in that window it is still a module
2008    /// that will sync, not one that never will.
2009    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2010        self.configured_ids
2011            .lock()
2012            .unwrap_or_else(|poisoned| poisoned.into_inner())
2013            .contains(module_id)
2014    }
2015
2016    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2017        let modules = self
2018            .modules
2019            .lock()
2020            .unwrap_or_else(|poisoned| poisoned.into_inner());
2021        modules.get(module_id).cloned()
2022    }
2023
2024    pub(crate) fn record_late_health_answer(
2025        &self,
2026        module_id: &str,
2027        latency_ms: u64,
2028    ) -> Result<bool, SuperviseError> {
2029        let Some(module) = self.get(module_id) else {
2030            return Ok(false);
2031        };
2032        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2033            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2034            state.health.last_late_answer_latency_ms = Some(latency_ms);
2035            // A late answer is an answer: the module served the probe, just past
2036            // the deadline. Leaving the miss streak in place while logging
2037            // "proves the module is alive" is how a CPU-starved module that
2038            // answers every probe a few seconds late still marches to the
2039            // threshold and gets killed — the exact kill class `NoAnswer` is
2040            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2041            // is degradation, and degradation reports; it does not restart.
2042            state.health.consecutive_failures = 0;
2043        })?;
2044        Ok(true)
2045    }
2046
2047    /// Arm the one-shot marker for the module process that this caller
2048    /// deliberately initiated severance against. Generic connection teardown
2049    /// must not call this:
2050    /// a surviving process would otherwise retain an exemption for a later
2051    /// genuine crash.
2052    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2053        let Some(module) = self.get(module_id) else {
2054            return Ok(false);
2055        };
2056        let status = module.status()?;
2057        let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
2058            return Ok(false);
2059        };
2060        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2061    }
2062
2063    pub fn list(&self) -> Vec<SupervisedModule> {
2064        let modules = self
2065            .modules
2066            .lock()
2067            .unwrap_or_else(|poisoned| poisoned.into_inner());
2068        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2069        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2070        modules
2071    }
2072
2073    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2074        self.spawn_nonces
2075            .lock()
2076            .unwrap_or_else(|poisoned| poisoned.into_inner())
2077            .remove(module_id);
2078        self.close_swap(module_id);
2079        let mut reserved_nonces = self
2080            .reserved_nonces
2081            .lock()
2082            .unwrap_or_else(|poisoned| poisoned.into_inner());
2083        if reserved_nonces.contains_key(module_id) {
2084            // The old nonce must die with the removed process, but the exact-id
2085            // gate remains until an operator explicitly releases it.
2086            reserved_nonces.insert(module_id.to_string(), None);
2087        }
2088        drop(reserved_nonces);
2089        self.reserved_prefix_owners
2090            .lock()
2091            .unwrap_or_else(|poisoned| poisoned.into_inner())
2092            .retain(|_, owner| owner != module_id);
2093        let removed = self
2094            .modules
2095            .lock()
2096            .unwrap_or_else(|poisoned| poisoned.into_inner())
2097            .remove(module_id);
2098        self.configured_ids
2099            .lock()
2100            .unwrap_or_else(|poisoned| poisoned.into_inner())
2101            .remove(module_id);
2102        removed
2103    }
2104
2105    /// Remember a module removed by a non-preview rescan so route.open can
2106    /// distinguish that intentional removal from an unknown id.
2107    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2108        self.removal_tombstones
2109            .lock()
2110            .unwrap_or_else(|poisoned| poisoned.into_inner())
2111            .insert(module_id.to_string(), unix_ms_now());
2112    }
2113
2114    /// Return how long ago a rescan removed this module in milliseconds.
2115    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2116        self.removal_tombstones
2117            .lock()
2118            .unwrap_or_else(|poisoned| poisoned.into_inner())
2119            .get(module_id)
2120            .copied()
2121            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2122    }
2123
2124    /// Retire a reserved-id gate only after its module has left supervision.
2125    ///
2126    /// A retained gate has no live nonce (`None`), so releasing any other entry
2127    /// would weaken a currently configured or otherwise active reservation.
2128    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2129        if self.get(module_id).is_some() {
2130            return false;
2131        }
2132        let mut reserved_nonces = self
2133            .reserved_nonces
2134            .lock()
2135            .unwrap_or_else(|poisoned| poisoned.into_inner());
2136        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2137            return false;
2138        }
2139        reserved_nonces.remove(module_id);
2140        true
2141    }
2142
2143    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2144        Arc::clone(&self.operation_lock)
2145    }
2146}
2147
2148/// Process supervisor for subc-owned singleton modules.
2149#[derive(Debug, Clone)]
2150pub struct Supervisor {
2151    registry: Arc<Registry>,
2152    restart_policy: RestartPolicy,
2153    drain_timeout: Duration,
2154    connection_file_path: Option<PathBuf>,
2155    capture_logs_dir: Option<PathBuf>,
2156    forwarding: Option<Arc<ForwardingTable>>,
2157    process_liveness: Arc<SupervisorProcessLiveness>,
2158    supervisor_handle: Option<SupervisorHandle>,
2159    health: HealthConfig,
2160    daemon_start_clock: crate::clock::StartClock,
2161    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2162    spawn_events: SpawnEventFeed,
2163    provenance_probe: ExecutableIdentityProbe,
2164    /// Every process spawned through this supervisor (and its clones) and not
2165    /// yet reaped, so daemon shutdown can end them.
2166    child_roster: ChildRoster,
2167    #[cfg(target_os = "linux")]
2168    cgroup_placement: Option<subc_cgroup::Placement>,
2169    #[cfg(test)]
2170    test_after_first_spawn: AfterFirstSpawnHook,
2171}
2172
2173/// Test-only hook run on the path that takes on a new module, right after its
2174/// first `spawn_child` returns (the process exists and could already be
2175/// registering) and before that process is handed to the module's supervise
2176/// loop and put on the roster. Lets a test observe what a fast child would see
2177/// in that window without racing a real one.
2178#[cfg(test)]
2179#[derive(Clone, Default)]
2180struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2181
2182#[cfg(test)]
2183type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2184
2185#[cfg(test)]
2186impl fmt::Debug for AfterFirstSpawnHook {
2187    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2188        f.write_str("AfterFirstSpawnHook")
2189    }
2190}
2191
2192#[cfg(test)]
2193impl AfterFirstSpawnHook {
2194    fn run(&self, module_id: &str) {
2195        if let Some(hook) = &self.0 {
2196            hook(module_id);
2197        }
2198    }
2199}
2200
2201impl Supervisor {
2202    /// The first step of an announced daemon shutdown, before the notice and
2203    /// before any connection is closed.
2204    ///
2205    /// Sets the daemon-shutdown flag first: from here on no module is
2206    /// respawned (crash restart, operator restart, or swap), and every child
2207    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2208    /// the module exits on the EOF this shutdown gives it or is signalled by a
2209    /// service manager that kills the whole cgroup. Then writes the journal's
2210    /// shutdown marker, which records the instant and closes this daemon
2211    /// incarnation's stretch of the journal.
2212    #[cfg(unix)]
2213    pub(crate) fn begin_daemon_shutdown(&self) {
2214        self.child_roster.close();
2215        if let Some(journal) = &self.terminal_journal {
2216            journal.stamp_shutdown();
2217        }
2218    }
2219
2220    /// Announce a cut while established connections can still carry replies.
2221    /// These budgets promise notice and a bounded wait, not child completion;
2222    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2223    #[cfg(unix)]
2224    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2225        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2226        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2227        let Some(forwarding) = &self.forwarding else {
2228            return Ok(());
2229        };
2230        let module_ids = forwarding
2231            .begin_daemon_drain()
2232            .map_err(SuperviseError::Forwarding)?;
2233        let deadline_ms =
2234            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2235        let mut notices = tokio::task::JoinSet::new();
2236        let mut drains = Vec::new();
2237        for module_id in module_ids {
2238            let Some(target) = forwarding
2239                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2240                .map_err(SuperviseError::Forwarding)?
2241            else {
2242                continue;
2243            };
2244            let routes = forwarding
2245                .endpoint_routes(target.endpoint)
2246                .map_err(SuperviseError::Forwarding)?;
2247            // Restart allows deployed consumers to reopen after the new daemon
2248            // appears. The wire reason stays `restart`; what tells a daemon cut
2249            // apart from a module restart afterwards is the terminal record
2250            // itself, whose disposition is `daemon_shutdown` for every exit
2251            // observed once `begin_daemon_shutdown` has run.
2252            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2253                reason: RouteCloseReason::Restart,
2254                deadline_ms,
2255            })
2256            .expect("module draining serializes");
2257            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2258            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2259            for route in routes {
2260                let client = route.goodbye_target;
2261                if let Some((_, channels)) = clients
2262                    .iter_mut()
2263                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2264                {
2265                    channels.push(client.channel);
2266                } else {
2267                    let channel = client.channel;
2268                    clients.push((client, vec![channel]));
2269                }
2270            }
2271            for (client, mut channels) in clients {
2272                channels.sort_unstable();
2273                channels.dedup();
2274                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2275                    module_id: module_id.clone(),
2276                    channels,
2277                    reason: RouteCloseReason::Restart,
2278                })
2279                .expect("route closing serializes");
2280                recipients.push((client.sink, client.negotiated_ver, closing));
2281            }
2282            for (sink, version, body) in recipients {
2283                notices.spawn(async move {
2284                    let frame = Frame::build_with_version(
2285                        version,
2286                        FrameType::Push,
2287                        control_flags(),
2288                        0,
2289                        0,
2290                        0,
2291                        body,
2292                    )
2293                    .expect("bounded lifecycle notice frame builds");
2294                    sink.send_flushed(frame).await
2295                });
2296            }
2297            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2298            drains.push((module_id, target.endpoint, gauges));
2299        }
2300        // A quiet forwarding table is not proof that queued notices reached the
2301        // socket. Wait for writer flush acknowledgements before testing quiescence.
2302        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2303        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2304            if !matches!(result, Ok(Ok(()))) {
2305                warn!(?result, "daemon shutdown notice delivery failed");
2306            }
2307        }
2308        notices.abort_all();
2309        let deadline = Instant::now() + DRAIN_BUDGET;
2310        let mut waits = tokio::task::JoinSet::new();
2311        for (module_id, endpoint, gauges) in drains {
2312            let forwarding = Arc::clone(forwarding);
2313            let mut runtime = self.runtime_config();
2314            runtime.health.cadence = Duration::from_millis(100);
2315            waits.spawn(async move {
2316                wait_for_forwarding_quiescence(
2317                    &forwarding,
2318                    &module_id,
2319                    &runtime,
2320                    endpoint,
2321                    deadline,
2322                    &gauges,
2323                    DrainScope::Active,
2324                )
2325                .await
2326            });
2327        }
2328        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2329            if !matches!(result, Ok(Ok(true))) {
2330                warn!(?result, "daemon shutdown drain did not reach quiescence");
2331            }
2332        }
2333        Ok(())
2334    }
2335
2336    /// The last step of an announced daemon shutdown, after the notice and the
2337    /// drain: send every registered module a module GOODBYE, the same planned
2338    /// stop signal `ck module stop` gives, then close every connection so each
2339    /// subc module sees EOF and starts its own teardown, then end every
2340    /// supervised child that has not exited
2341    /// by its own deadline (its drain budget, capped). Modules lead their own
2342    /// process groups, so a
2343    /// service manager's group kill no longer reaches them; without this a
2344    /// child that does not stop on EOF (every `protocol: "none"` child, which
2345    /// has no connection) would outlive the daemon. Every wait is bounded (see
2346    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2347    #[cfg(unix)]
2348    pub(crate) async fn end_children_for_daemon_shutdown(
2349        &self,
2350        already_escalated: bool,
2351        escalate: impl std::future::Future<Output = ()>,
2352    ) {
2353        tokio::pin!(escalate);
2354        let mut escalated = already_escalated;
2355        if let Some(forwarding) = &self.forwarding {
2356            let reason = CloseReason::new(
2357                "daemon_shutdown",
2358                "the daemon is exiting after its shutdown notice and drain",
2359            );
2360            if escalated {
2361                // The operator asked to stop waiting: queue the GOODBYEs but
2362                // do not wait for them to be written.
2363                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2364            } else {
2365                tokio::select! {
2366                    biased;
2367                    _ = escalate.as_mut() => {
2368                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2369                        escalated = true;
2370                    }
2371                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2372                }
2373            }
2374            let closed = forwarding.close_all_connections(&reason);
2375            debug!(closed, "closed established connections for daemon shutdown");
2376        }
2377        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2378        // already completed and must not be polled again; the child shutdown
2379        // wait is told it is escalated and gets a future that never fires.
2380        let escalated_here = escalated && !already_escalated;
2381        let remaining_escalate = async move {
2382            if escalated_here {
2383                std::future::pending::<()>().await;
2384            } else {
2385                escalate.await;
2386            }
2387        };
2388        crate::child_roster::end_children_for_daemon_shutdown(
2389            &self.child_roster,
2390            escalated,
2391            remaining_escalate,
2392        )
2393        .await;
2394    }
2395
2396    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2397        Self {
2398            registry,
2399            restart_policy,
2400            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2401            connection_file_path: None,
2402            capture_logs_dir: None,
2403            forwarding: None,
2404            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2405            supervisor_handle: None,
2406            health: HealthConfig::default(),
2407            daemon_start_clock: crate::clock::StartClock::capture(),
2408            terminal_journal: None,
2409            spawn_events: SpawnEventFeed::default(),
2410            provenance_probe: ExecutableIdentityProbe::default(),
2411            child_roster: ChildRoster::default(),
2412            #[cfg(target_os = "linux")]
2413            cgroup_placement: None,
2414            #[cfg(test)]
2415            test_after_first_spawn: AfterFirstSpawnHook::default(),
2416        }
2417    }
2418
2419    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2420        self.drain_timeout = drain_timeout;
2421        self
2422    }
2423
2424    pub fn with_process_liveness(
2425        mut self,
2426        process_liveness: Arc<SupervisorProcessLiveness>,
2427    ) -> Self {
2428        self.process_liveness = process_liveness;
2429        self
2430    }
2431
2432    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2433        self.connection_file_path = Some(connection_file_path.into());
2434        self
2435    }
2436
2437    /// Enables daemon-owned capture files for supervised stdout and stderr.
2438    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2439        self.capture_logs_dir = Some(logs_dir.into());
2440        self
2441    }
2442
2443    /// Names this daemon lifetime in spawn events, independently of whether a
2444    /// terminal journal is configured.
2445    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2446        // A millisecond start stamp can repeat after clock rollback or a rapid
2447        // restart. Use the connection file's random daemon_id instead: it already
2448        // identifies this daemon lifetime independently of the wall clock.
2449        self.spawn_events.configure_incarnation(daemon_incarnation);
2450        self
2451    }
2452
2453    /// Enables best-effort history shared by every supervised module. Without
2454    /// it, terminal history is kept only in each module's in-memory ring.
2455    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2456        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2457        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2458            path,
2459            daemon_incarnation,
2460        )));
2461        this
2462    }
2463
2464    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2465        self.forwarding = Some(forwarding);
2466        self
2467    }
2468
2469    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2470        self.spawn_events = supervisor_handle.spawn_events.clone();
2471        self.supervisor_handle = Some(supervisor_handle);
2472        self
2473    }
2474
2475    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2476        self.health = health;
2477        self
2478    }
2479
2480    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2481    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2482    /// record is kept.
2483    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2484        self.child_roster.record_to(path.into());
2485        self
2486    }
2487
2488    #[cfg(target_os = "linux")]
2489    pub fn with_cgroup_placement(
2490        mut self,
2491        cgroup_placement: Option<subc_cgroup::Placement>,
2492    ) -> Self {
2493        self.cgroup_placement = cgroup_placement;
2494        self
2495    }
2496
2497    /// Spawn `spec.program` and start monitoring it.
2498    ///
2499    /// The child is expected to parse `--subc <connection-file-path>`, read the
2500    /// TCP+key connection file, authenticate to the already-running listener, and
2501    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2502    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2503        validate_spec(&spec)?;
2504        self.establish_identity(&spec);
2505
2506        let runtime = self.runtime_config();
2507        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2508        let spawned = spawn_child(
2509            &spec,
2510            runtime.connection_file_path.as_deref(),
2511            self.supervisor_handle.as_ref(),
2512            &runtime.stderr_ring,
2513            runtime.capture_logs_dir.as_deref(),
2514            &runtime.child_roster,
2515            #[cfg(target_os = "linux")]
2516            runtime.cgroup_placement.as_ref(),
2517        );
2518        #[cfg(test)]
2519        self.test_after_first_spawn.run(&spec.module_id);
2520        let child = match spawned {
2521            Ok(child) => child,
2522            Err(err) => {
2523                // Unlike the configured paths, a failed `spawn` leaves nothing
2524                // on the roster, so the module must not stay marked configured.
2525                self.abandon_unrostered(&spec.module_id);
2526                return Err(err);
2527            }
2528        };
2529        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2530
2531        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2532    }
2533
2534    /// Make `spec`'s module count as configured, with its identity gates
2535    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2536    /// exists.
2537    ///
2538    /// Every path that takes on a new module calls this before `spawn_child`.
2539    /// The order is the point: the child can connect, register, sync its
2540    /// scopes and ask about them as soon as it is spawned, and the module is
2541    /// only put on the roster after `spawn_child` returns. Were the mark set
2542    /// with the roster entry, a fast child would see its own owner reported
2543    /// as not configured, and a scoped `route.open` in that window would be
2544    /// refused as terminal `scope_not_live` ("will never sync") instead of
2545    /// retryable `scope_not_synced`.
2546    fn establish_identity(&self, spec: &ModuleSpec) {
2547        if let Some(supervisor_handle) = &self.supervisor_handle {
2548            supervisor_handle.apply_identity_configuration(spec);
2549            supervisor_handle.mark_configured(&spec.module_id);
2550        }
2551    }
2552
2553    /// Take back [`Self::establish_identity`]'s configured mark when the
2554    /// module will not be put on the roster after all.
2555    fn abandon_unrostered(&self, module_id: &str) {
2556        if let Some(supervisor_handle) = &self.supervisor_handle {
2557            supervisor_handle.unmark_configured_unless_rostered(module_id);
2558        }
2559    }
2560
2561    /// Record a freshly spawned first process as running. On failure the
2562    /// module never reaches the roster, so its configured mark is taken back.
2563    fn mark_first_process_running(
2564        &self,
2565        spec: &ModuleSpec,
2566        runtime: &SupervisorRuntimeConfig,
2567        snapshot: &SharedSnapshot,
2568        child: &SupervisedChild,
2569    ) -> Result<(), SuperviseError> {
2570        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2571            self.abandon_unrostered(&spec.module_id);
2572            return Err(err);
2573        }
2574        self.process_liveness
2575            .track(spec.module_id.clone(), Arc::clone(snapshot));
2576        Ok(())
2577    }
2578
2579    /// Start supervising a module declared in daemon configuration.
2580    ///
2581    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2582    /// failures in the supervisor handle so operator-facing `supervisor.list`
2583    /// reflects every configured module while daemon startup continues.
2584    pub fn supervise_configured(
2585        &self,
2586        spec: ModuleSpec,
2587        enabled: bool,
2588    ) -> Result<SupervisedModule, SuperviseError> {
2589        validate_spec(&spec)?;
2590        self.establish_identity(&spec);
2591
2592        let runtime = self.runtime_config();
2593        if !enabled {
2594            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2595            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2596        }
2597
2598        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2599        let spawned = spawn_child(
2600            &spec,
2601            runtime.connection_file_path.as_deref(),
2602            self.supervisor_handle.as_ref(),
2603            &runtime.stderr_ring,
2604            runtime.capture_logs_dir.as_deref(),
2605            &runtime.child_roster,
2606            #[cfg(target_os = "linux")]
2607            runtime.cgroup_placement.as_ref(),
2608        );
2609        #[cfg(test)]
2610        self.test_after_first_spawn.run(&spec.module_id);
2611        match spawned {
2612            Ok(child) => {
2613                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2614                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2615            }
2616            Err(err) => {
2617                error!(
2618                    module_id = %spec.module_id,
2619                    program = %spec.program.display(),
2620                    error = %err,
2621                    "configured module failed to spawn; marking failed and continuing"
2622                );
2623                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2624                Ok(self.supervised_module(spec, runtime, snapshot, None))
2625            }
2626        }
2627    }
2628
2629    /// Supervise a configured module with its own health, drain, and crash
2630    /// budget. The restart policy is per-module because the config file is:
2631    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2632    /// module that is expensive to restart should not be forced onto the same
2633    /// budget as one that is cheap.
2634    pub fn supervise_configured_with_health(
2635        &self,
2636        spec: ModuleSpec,
2637        enabled: bool,
2638        health: HealthConfig,
2639        drain_timeout_ms: Option<u64>,
2640        restart_policy: RestartPolicy,
2641    ) -> Result<SupervisedModule, SuperviseError> {
2642        validate_spec(&spec)?;
2643        self.establish_identity(&spec);
2644
2645        let mut runtime = self.runtime_config();
2646        runtime.health = health.clone();
2647        runtime.restart_policy = restart_policy;
2648        if let Some(ms) = drain_timeout_ms {
2649            runtime.drain_timeout = Duration::from_millis(ms);
2650            *runtime
2651                .effective_drain_timeout
2652                .lock()
2653                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2654        }
2655        if !enabled {
2656            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2657            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2658        }
2659
2660        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2661        let spawned = spawn_child(
2662            &spec,
2663            runtime.connection_file_path.as_deref(),
2664            self.supervisor_handle.as_ref(),
2665            &runtime.stderr_ring,
2666            runtime.capture_logs_dir.as_deref(),
2667            &runtime.child_roster,
2668            #[cfg(target_os = "linux")]
2669            runtime.cgroup_placement.as_ref(),
2670        );
2671        #[cfg(test)]
2672        self.test_after_first_spawn.run(&spec.module_id);
2673        match spawned {
2674            Ok(child) => {
2675                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2676                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2677            }
2678            Err(err) => {
2679                if health.critical {
2680                    error!(
2681                        module_id = %spec.module_id,
2682                        program = %spec.program.display(),
2683                        error = %err,
2684                        "critical configured module failed to spawn; marking failed and alerting"
2685                    );
2686                } else {
2687                    error!(
2688                        module_id = %spec.module_id,
2689                        program = %spec.program.display(),
2690                        error = %err,
2691                        "configured module failed to spawn; marking failed and continuing"
2692                    );
2693                }
2694                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2695                Ok(self.supervised_module(spec, runtime, snapshot, None))
2696            }
2697        }
2698    }
2699
2700    fn runtime_config(&self) -> SupervisorRuntimeConfig {
2701        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2702        SupervisorRuntimeConfig {
2703            scheduled_respawn: Arc::default(),
2704            deferred_reload_reply: Arc::default(),
2705            restart_policy: self.restart_policy,
2706            drain_timeout: self.drain_timeout,
2707            // Shared with this module's roster copy: daemon shutdown waits on
2708            // each child for the module's own drain budget, as resolved now.
2709            child_roster: self
2710                .child_roster
2711                .for_module(Arc::clone(&effective_drain_timeout)),
2712            effective_drain_timeout,
2713            default_drain_timeout: self.drain_timeout,
2714            health: self.health.clone(),
2715            connection_file_path: self.connection_file_path.clone(),
2716            capture_logs_dir: self.capture_logs_dir.clone(),
2717            forwarding: self.forwarding.clone(),
2718            supervisor_handle: self.supervisor_handle.clone(),
2719            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
2720            terminal_ring: Arc::new(Mutex::new(
2721                TerminalRing::new(
2722                    TerminalRingConfig::default(),
2723                    self.daemon_start_clock.started_at_ms(),
2724                )
2725                .with_start_clock(self.daemon_start_clock)
2726                .with_journal(self.terminal_journal.clone())
2727                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
2728            )),
2729            spawn_events: self.spawn_events.clone(),
2730            #[cfg(target_os = "linux")]
2731            cgroup_placement: self.cgroup_placement.clone(),
2732            #[cfg(test)]
2733            test_seed_stale_facts_before_enable_spawn: false,
2734            #[cfg(test)]
2735            test_reload_exit_record_gate: None,
2736        }
2737    }
2738
2739    fn supervised_module(
2740        &self,
2741        spec: ModuleSpec,
2742        runtime: SupervisorRuntimeConfig,
2743        snapshot: SharedSnapshot,
2744        child: Option<SupervisedChild>,
2745    ) -> SupervisedModule {
2746        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
2747            spec: spec.clone(),
2748            health: runtime.health.clone(),
2749        }));
2750        let stderr_ring = Arc::clone(&runtime.stderr_ring);
2751        let terminal_ring = Arc::clone(&runtime.terminal_ring);
2752        // The module's OWN policy, which may be its per-module config rather than
2753        // the supervisor-wide one; status must report the budget the supervise
2754        // loop actually enforces.
2755        let restart_policy = runtime.restart_policy;
2756        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
2757        let (tx, rx) = mpsc::channel(4);
2758        let monitor = tokio::spawn(supervise_loop(
2759            spec.clone(),
2760            runtime,
2761            Arc::clone(&self.registry),
2762            Arc::clone(&self.process_liveness),
2763            Arc::clone(&snapshot),
2764            child,
2765            rx,
2766        ));
2767
2768        let module_id = spec.module_id.clone();
2769        let module = SupervisedModule {
2770            inner: Arc::new(SupervisedModuleInner {
2771                module_id: module_id.clone(),
2772                registry: Arc::clone(&self.registry),
2773                snapshot,
2774                configuration,
2775                stderr_ring,
2776                terminal_ring,
2777                commands: tx,
2778                monitor: Mutex::new(Some(monitor)),
2779                restart_policy,
2780                effective_drain_timeout,
2781                provenance_probe: self.provenance_probe.clone(),
2782            }),
2783        };
2784        // The identity gates and the configured mark were set by
2785        // `establish_identity` before any process was spawned; only the roster
2786        // entry waits for the module handle, which needs the spawned child.
2787        if let Some(supervisor_handle) = &self.supervisor_handle {
2788            supervisor_handle.insert(module.clone());
2789        }
2790        module
2791    }
2792}
2793
2794impl Default for Supervisor {
2795    fn default() -> Self {
2796        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
2797    }
2798}
2799
2800/// Handle to one supervised child process.
2801#[derive(Clone)]
2802pub struct SupervisedModule {
2803    inner: Arc<SupervisedModuleInner>,
2804}
2805
2806struct SupervisedModuleInner {
2807    module_id: String,
2808    registry: Arc<Registry>,
2809    snapshot: SharedSnapshot,
2810    configuration: Arc<Mutex<SupervisedConfiguration>>,
2811    stderr_ring: Arc<Mutex<StderrRing>>,
2812    terminal_ring: Arc<Mutex<TerminalRing>>,
2813    commands: mpsc::Sender<SupervisorCommand>,
2814    monitor: Mutex<Option<JoinHandle<()>>>,
2815    /// Copied from the supervisor's runtime config at spawn so `status()` can
2816    /// report the restart budget without reaching back into the supervisor. The
2817    /// policy is fixed for the process's lifetime, so a copy cannot drift.
2818    restart_policy: RestartPolicy,
2819    effective_drain_timeout: Arc<Mutex<Duration>>,
2820    provenance_probe: ExecutableIdentityProbe,
2821}
2822
2823impl fmt::Debug for SupervisedModule {
2824    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2825        f.debug_struct("SupervisedModule")
2826            .field("module_id", &self.inner.module_id)
2827            .field("status", &self.status())
2828            .finish_non_exhaustive()
2829    }
2830}
2831
2832impl SupervisedModule {
2833    pub fn module_id(&self) -> &str {
2834        &self.inner.module_id
2835    }
2836
2837    /// Test-only: put one probe miss on the streak, the way
2838    /// `handle_health_probe_failure` does, so tests can assert what a later
2839    /// event does to the streak without driving the whole probe loop.
2840    #[cfg(test)]
2841    pub(crate) fn record_health_probe_failure_for_test(
2842        &self,
2843        detail: &str,
2844    ) -> Result<(), SuperviseError> {
2845        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
2846            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
2847            state.health.detail = Some(detail.to_string());
2848        })
2849    }
2850
2851    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
2852        Ok(lock_snapshot(&self.inner.snapshot)?.state)
2853    }
2854
2855    /// The module's retained stderr, newest lines last.
2856    ///
2857    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
2858    /// module, `supervisor.list` renders every module, and putting it in the
2859    /// shared snapshot would make each status read carry a payload almost nobody
2860    /// asked for. Callers that want the text ask for it.
2861    pub fn stderr_tail(
2862        &self,
2863        max_lines: Option<usize>,
2864        max_bytes: Option<usize>,
2865    ) -> StderrTailSnapshot {
2866        self.inner
2867            .stderr_ring
2868            .lock()
2869            .unwrap_or_else(|poisoned| poisoned.into_inner())
2870            .snapshot(max_lines, max_bytes)
2871    }
2872
2873    /// The module's bounded terminal history, oldest retained exit first.
2874    ///
2875    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
2876    /// daemon whose in-memory history was necessarily reset.
2877    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
2878        self.inner
2879            .terminal_ring
2880            .lock()
2881            .unwrap_or_else(|poisoned| poisoned.into_inner())
2882            .snapshot()
2883    }
2884
2885    /// Retained observations from the current ring and all journal generations.
2886    ///
2887    /// Blocking: this reads the journal files. Async callers use
2888    /// [`Self::read_durable_terminal_history`].
2889    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
2890        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
2891    }
2892
2893    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
2894    /// read (up to every retained generation) never occupies a runtime worker.
2895    /// Fails only if the blocking task could not finish (runtime shutdown or a
2896    /// panic in the read).
2897    pub(crate) async fn read_durable_terminal_history(
2898        &self,
2899    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
2900        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
2901        let module_id = self.inner.module_id.clone();
2902        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
2903            .await
2904    }
2905
2906    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
2907        self.status_with_snapshot_lock(&self.inner.snapshot, None)
2908    }
2909
2910    pub(crate) fn record_deliberate_severance(
2911        &self,
2912        identity: ProcessIdentity,
2913    ) -> Result<bool, SuperviseError> {
2914        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2915        if snapshot.pid != Some(identity.pid)
2916            || snapshot.process_start_time != Some(identity.start_time)
2917        {
2918            return Ok(false);
2919        }
2920        snapshot.deliberate_severance = Some(identity);
2921        Ok(true)
2922    }
2923
2924    /// Read status for a channel-0 renderer and report a contended snapshot lock.
2925    ///
2926    /// Internal supervision callers use [`Self::status`] so writer-side machinery
2927    /// does not produce reader-observability logs.
2928    pub(crate) fn status_for_control(
2929        &self,
2930        caller: &'static str,
2931    ) -> Result<ModuleStatus, SuperviseError> {
2932        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
2933    }
2934
2935    fn status_with_snapshot_lock(
2936        &self,
2937        snapshot: &SharedSnapshot,
2938        caller: Option<&'static str>,
2939    ) -> Result<ModuleStatus, SuperviseError> {
2940        let mut guard = match caller {
2941            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
2942            None => lock_snapshot(snapshot)?,
2943        };
2944        // Read the budget through the pruning path so a reader sees the same
2945        // in-window count the restart decision would use, not a stale total.
2946        let restart_count =
2947            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
2948        let snapshot = guard.clone();
2949        drop(guard);
2950        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
2951            SuperviseError::StatePoisoned {
2952                module_id: Some(self.inner.module_id.clone()),
2953            }
2954        })?;
2955        let registration_active = self
2956            .inner
2957            .registry
2958            .get_module(&self.inner.module_id)
2959            .map_err(SuperviseError::Registry)?
2960            .is_some();
2961        let protocol = snapshot
2962            .spawned_protocol
2963            .unwrap_or(self.declared_protocol()?);
2964        let running_process =
2965            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
2966        // Registration is the difference between the two protocols and the only
2967        // one: a subc module that has not registered cannot serve a request even
2968        // though its process is up, and a `none` module never registers at all,
2969        // so requiring it there would pin `live` to false for the whole life of
2970        // a perfectly healthy process.
2971        let live = match protocol {
2972            ModuleProtocol::Subc => running_process && registration_active,
2973            ModuleProtocol::None => running_process,
2974        };
2975
2976        Ok(ModuleStatus {
2977            module_id: self.inner.module_id.clone(),
2978            state: snapshot.state,
2979            enabled: snapshot.enabled,
2980            process_alive: snapshot.process_alive,
2981            registration_active,
2982            protocol,
2983            live,
2984            restart_count,
2985            lifetime_restarts: snapshot.lifetime_restarts,
2986            spawn_generation: snapshot.spawn_generation,
2987            max_restarts: self.inner.restart_policy.max_restarts,
2988            restart_window: self.inner.restart_policy.window,
2989            drain_timeout,
2990            restart_backoff: self.inner.restart_policy.backoff,
2991            restart_max_backoff: self.inner.restart_policy.max_backoff,
2992            pid: snapshot.pid,
2993            spawned_at_ms: snapshot.spawned_at_ms,
2994            spawned_from: snapshot.spawned_from,
2995            process_start_time: snapshot.process_start_time,
2996            last_exit: snapshot.last_exit,
2997            health: snapshot.health,
2998        })
2999    }
3000
3001    #[cfg(test)]
3002    pub(crate) fn hold_snapshot_for_test(
3003        &self,
3004        acquired: std::sync::mpsc::Sender<()>,
3005        hold: Duration,
3006    ) -> std::thread::JoinHandle<()> {
3007        let snapshot = Arc::clone(&self.inner.snapshot);
3008        std::thread::spawn(move || {
3009            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3010            acquired
3011                .send(())
3012                .expect("test receiver waits for snapshot lock");
3013            std::thread::sleep(hold);
3014        })
3015    }
3016
3017    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3018        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3019            Ok(snapshot) => snapshot.clone(),
3020            Err(_) => {
3021                return subc_control::RunningImageAgreement::Unavailable {
3022                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3023                };
3024            }
3025        };
3026        self.inner
3027            .provenance_probe
3028            .observe(
3029                snapshot.pid,
3030                snapshot.spawned_from.as_deref(),
3031                snapshot.spawned_file_identity,
3032                snapshot.process_start_time,
3033            )
3034            .await
3035    }
3036
3037    /// Memory and CPU time of the module's current process, read now. Only the
3038    /// process the supervisor spawned is read, not processes it has started.
3039    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3040        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3041            Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
3042            Err(_) => {
3043                return subc_control::ChildResourceUsage::Unavailable {
3044                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3045                }
3046            }
3047        };
3048        crate::child_resources::read(pid, start_time)
3049    }
3050
3051    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3052        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3053        Ok(match snapshot.state {
3054            ModuleState::Restarting => true,
3055            ModuleState::Failed | ModuleState::Disabled => false,
3056            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3057        })
3058    }
3059
3060    #[cfg(test)]
3061    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3062        self.is_warming_with_snapshot_lock(None)
3063    }
3064
3065    pub(crate) fn is_warming_for_control(
3066        &self,
3067        caller: &'static str,
3068    ) -> Result<bool, SuperviseError> {
3069        self.is_warming_with_snapshot_lock(Some(caller))
3070    }
3071
3072    fn is_warming_with_snapshot_lock(
3073        &self,
3074        caller: Option<&'static str>,
3075    ) -> Result<bool, SuperviseError> {
3076        let snapshot = match caller {
3077            Some(caller) => {
3078                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3079            }
3080            None => lock_snapshot(&self.inner.snapshot)?,
3081        }
3082        .clone();
3083        Ok(matches!(
3084            snapshot.state,
3085            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3086        ))
3087    }
3088
3089    /// Drain the module and stop monitoring it.
3090    pub async fn drain(&self) -> Result<(), SuperviseError> {
3091        self.stop().await
3092    }
3093
3094    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3095        match self.state()? {
3096            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3097            ModuleState::Starting
3098            | ModuleState::Running
3099            | ModuleState::Unresponsive
3100            | ModuleState::Restarting
3101            | ModuleState::Draining
3102            | ModuleState::Disabled => {}
3103        }
3104
3105        let (reply_tx, reply_rx) = oneshot::channel();
3106        self.inner
3107            .commands
3108            .send(SupervisorCommand::Retire { reply: reply_tx })
3109            .await
3110            .map_err(|_| SuperviseError::CommandClosed {
3111                module_id: self.inner.module_id.clone(),
3112            })?;
3113        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3114            module_id: self.inner.module_id.clone(),
3115        })?
3116    }
3117
3118    pub async fn stop(&self) -> Result<(), SuperviseError> {
3119        match self.state()? {
3120            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3121            ModuleState::Starting
3122            | ModuleState::Running
3123            | ModuleState::Unresponsive
3124            | ModuleState::Restarting
3125            | ModuleState::Draining
3126            | ModuleState::Disabled => {}
3127        }
3128
3129        let (reply_tx, reply_rx) = oneshot::channel();
3130        self.inner
3131            .commands
3132            .send(SupervisorCommand::Drain { reply: reply_tx })
3133            .await
3134            .map_err(|_| SuperviseError::CommandClosed {
3135                module_id: self.inner.module_id.clone(),
3136            })?;
3137        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3138            module_id: self.inner.module_id.clone(),
3139        })?
3140    }
3141
3142    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3143        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3144        let (reply_tx, reply_rx) = oneshot::channel();
3145        self.inner
3146            .commands
3147            .send(SupervisorCommand::Restart {
3148                drain_timeout_ms,
3149                received_at_generation,
3150                queued_at: Instant::now(),
3151                reply: reply_tx,
3152            })
3153            .await
3154            .map_err(|_| SuperviseError::CommandClosed {
3155                module_id: self.inner.module_id.clone(),
3156            })?;
3157        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3158            module_id: self.inner.module_id.clone(),
3159        })?
3160    }
3161
3162    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3163    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3164    /// process then drains in the background of the supervise loop) or has
3165    /// failed, leaving the old process serving.
3166    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3167        let (reply_tx, reply_rx) = oneshot::channel();
3168        self.inner
3169            .commands
3170            .send(SupervisorCommand::Swap {
3171                ready_timeout,
3172                reply: reply_tx,
3173            })
3174            .await
3175            .map_err(|_| SuperviseError::CommandClosed {
3176                module_id: self.inner.module_id.clone(),
3177            })?;
3178        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3179            module_id: self.inner.module_id.clone(),
3180        })?
3181    }
3182
3183    pub async fn reload(&self) -> Result<(), SuperviseError> {
3184        let (reply_tx, reply_rx) = oneshot::channel();
3185        self.inner
3186            .commands
3187            .send(SupervisorCommand::Reload { reply: reply_tx })
3188            .await
3189            .map_err(|_| SuperviseError::CommandClosed {
3190                module_id: self.inner.module_id.clone(),
3191            })?;
3192        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3193            module_id: self.inner.module_id.clone(),
3194        })?
3195    }
3196
3197    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3198        let (reply_tx, reply_rx) = oneshot::channel();
3199        self.inner
3200            .commands
3201            .send(SupervisorCommand::SetEnabled {
3202                enabled,
3203                reply: reply_tx,
3204            })
3205            .await
3206            .map_err(|_| SuperviseError::CommandClosed {
3207                module_id: self.inner.module_id.clone(),
3208            })?;
3209        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3210            module_id: self.inner.module_id.clone(),
3211        })?
3212    }
3213
3214    /// The current process's protocol, or the configured protocol when down.
3215    /// A rescan stores the next launch spec without changing how an existing
3216    /// process registers, serves routes, is probed, or exits.
3217    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3218        let configured = self
3219            .inner
3220            .configuration
3221            .lock()
3222            .map_err(|_| SuperviseError::StatePoisoned {
3223                module_id: Some(self.inner.module_id.clone()),
3224            })?
3225            .spec
3226            .protocol;
3227        let state = lock_snapshot(&self.inner.snapshot)?;
3228        Ok(state.spawned_protocol.unwrap_or(configured))
3229    }
3230
3231    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3232        let configuration =
3233            self.inner
3234                .configuration
3235                .lock()
3236                .map_err(|_| SuperviseError::StatePoisoned {
3237                    module_id: Some(self.inner.module_id.clone()),
3238                })?;
3239        Ok((configuration.spec.clone(), configuration.health.clone()))
3240    }
3241
3242    /// Replace this module's launch spec, keeping its health and drain policy,
3243    /// the way a rescan does for a changed config entry. The running process is
3244    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3245    #[cfg(any(test, feature = "test-support"))]
3246    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3247        let (_, health) = self.configuration()?;
3248        let drain_timeout_ms = u64::try_from(
3249            self.inner
3250                .effective_drain_timeout
3251                .lock()
3252                .unwrap_or_else(|poisoned| poisoned.into_inner())
3253                .as_millis(),
3254        )
3255        .ok();
3256        self.update_configuration(spec, health, drain_timeout_ms)
3257            .await
3258    }
3259
3260    pub(crate) async fn update_configuration(
3261        &self,
3262        spec: ModuleSpec,
3263        health: HealthConfig,
3264        drain_timeout_ms: Option<u64>,
3265    ) -> Result<(), SuperviseError> {
3266        if spec.module_id != self.inner.module_id {
3267            return Err(SuperviseError::InvalidSpec {
3268                reason: "a supervised module's module_id cannot be changed".to_string(),
3269            });
3270        }
3271        validate_spec(&spec)?;
3272        let (reply_tx, reply_rx) = oneshot::channel();
3273        self.inner
3274            .commands
3275            .send(SupervisorCommand::UpdateConfiguration {
3276                spec: spec.clone(),
3277                health: health.clone(),
3278                drain_timeout_ms,
3279                reply: reply_tx,
3280            })
3281            .await
3282            .map_err(|_| SuperviseError::CommandClosed {
3283                module_id: self.inner.module_id.clone(),
3284            })?;
3285        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3286            module_id: self.inner.module_id.clone(),
3287        })?;
3288        let mut configuration =
3289            self.inner
3290                .configuration
3291                .lock()
3292                .map_err(|_| SuperviseError::StatePoisoned {
3293                    module_id: Some(self.inner.module_id.clone()),
3294                })?;
3295        configuration.spec = spec;
3296        configuration.health = health;
3297        Ok(())
3298    }
3299}
3300
3301impl Drop for SupervisedModuleInner {
3302    fn drop(&mut self) {
3303        let Ok(mut monitor) = self.monitor.lock() else {
3304            return;
3305        };
3306        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3307            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3308                state.state = ModuleState::Stopped;
3309                clear_current_process_facts(state);
3310            });
3311            monitor.abort();
3312        }
3313        let _ = monitor.take();
3314    }
3315}
3316
3317#[derive(Debug)]
3318enum SupervisorCommand {
3319    Drain {
3320        reply: oneshot::Sender<Result<(), SuperviseError>>,
3321    },
3322    Retire {
3323        reply: oneshot::Sender<Result<(), SuperviseError>>,
3324    },
3325    Restart {
3326        /// Operator override for this one restart's drain budget, in ms. `None`
3327        /// uses the module's configured/default budget; `Some(0)` cuts
3328        /// immediately (wedge bounce: a stuck request never settles, so
3329        /// waiting only delays recovery).
3330        drain_timeout_ms: Option<u64>,
3331        /// The module's `spawn_generation` when the request was received, before
3332        /// it waited in the command queue. A queued restart whose module has
3333        /// since spawned a newer process is already satisfied (see the handler).
3334        received_at_generation: u64,
3335        /// When the request entered the command queue, so the handler can log
3336        /// how long it waited behind the loop's other work.
3337        queued_at: Instant,
3338        reply: oneshot::Sender<Result<(), SuperviseError>>,
3339    },
3340    Reload {
3341        reply: oneshot::Sender<Result<(), SuperviseError>>,
3342    },
3343    SetEnabled {
3344        enabled: bool,
3345        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3346    },
3347    UpdateConfiguration {
3348        spec: ModuleSpec,
3349        health: HealthConfig,
3350        /// Per-module drain override from the new config; `None` re-resolves to
3351        /// the supervisor-wide default.
3352        drain_timeout_ms: Option<u64>,
3353        reply: oneshot::Sender<()>,
3354    },
3355    Swap {
3356        /// How long the candidate may take to register and declare itself
3357        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3358        ready_timeout: Option<Duration>,
3359        /// Answered at cutover or failure; the incumbent's drain follows.
3360        reply: oneshot::Sender<Result<(), SuperviseError>>,
3361    },
3362}
3363
3364#[derive(Debug)]
3365pub enum SuperviseError {
3366    InvalidSpec {
3367        reason: String,
3368    },
3369    Spawn {
3370        program: PathBuf,
3371        source: io::Error,
3372        cgroup_path: Option<PathBuf>,
3373    },
3374    Cgroup {
3375        module_id: String,
3376        source: io::Error,
3377    },
3378    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3379    /// than spawn a reserved module without its identity binding.
3380    LaunchNonce {
3381        reason: String,
3382    },
3383    Wait {
3384        module_id: String,
3385        source: io::Error,
3386    },
3387    Kill {
3388        module_id: String,
3389        source: io::Error,
3390    },
3391    Forwarding(ForwardingError),
3392    Registry(RegistryError),
3393    ReloadUnavailable {
3394        module_id: String,
3395        reason: String,
3396    },
3397    /// An operator restart/reload was requested for a module that is currently
3398    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3399    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3400    /// by a restart, so these commands are rejected instead of re-enabling it.
3401    Disabled {
3402        module_id: String,
3403    },
3404    ReloadFailed {
3405        module_id: String,
3406        reason: String,
3407    },
3408    RegistrationStillActive {
3409        module_id: String,
3410        waited: Duration,
3411    },
3412    StatePoisoned {
3413        module_id: Option<String>,
3414    },
3415    CommandClosed {
3416        module_id: String,
3417    },
3418    /// A restart or reload arrived while a swap's candidate was warming. The
3419    /// swap owns the module until it cuts over or fails; a stop or disable
3420    /// would have aborted it instead.
3421    SwapInProgress {
3422        module_id: String,
3423    },
3424    /// A swap was refused before anything was spawned.
3425    SwapRefused {
3426        module_id: String,
3427        reason: SwapRefusal,
3428    },
3429    /// A swap spawned a candidate and gave up on it. The candidate has been
3430    /// killed and its slot freed; the incumbent was left serving and was never
3431    /// drained, except in the one `CutoverLost` case described on that arm.
3432    SwapFailed {
3433        module_id: String,
3434        arm: SwapFailureArm,
3435        detail: String,
3436        /// How the candidate exited, when it exited on its own before the
3437        /// supervisor gave up on it.
3438        candidate_exit: Option<ExitReport>,
3439    },
3440}
3441
3442/// Why a swap was refused before a candidate was spawned.
3443#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3444pub enum SwapRefusal {
3445    /// The module's config does not declare `overlap: "safe"`.
3446    OverlapExclusive,
3447    /// The module is not registered, so there is no incumbent to keep serving
3448    /// and nothing a swap would improve on; a plain restart is the tool.
3449    NotRegistered,
3450    /// The module does not speak the subc wire, so a candidate could never
3451    /// register or declare itself ready.
3452    ProtocolNone,
3453    /// The supervisor lacks the forwarding table (to cut routes over) or the
3454    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3455    NotConfigured,
3456    /// A swap is already open for this module.
3457    AlreadySwapping,
3458}
3459
3460impl SwapRefusal {
3461    pub fn as_str(self) -> &'static str {
3462        match self {
3463            Self::OverlapExclusive => "overlap_exclusive",
3464            Self::NotRegistered => "not_registered",
3465            Self::ProtocolNone => "protocol_none",
3466            Self::NotConfigured => "not_configured",
3467            Self::AlreadySwapping => "already_swapping",
3468        }
3469    }
3470}
3471
3472/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3473/// serving and undrained; see `CutoverLost`.
3474#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3475pub enum SwapFailureArm {
3476    /// The candidate process could not be started.
3477    SpawnFailed,
3478    /// The candidate did not register within the readiness budget.
3479    NeverRegistered,
3480    /// The candidate registered but did not declare itself ready in time.
3481    NeverReady,
3482    /// The candidate exited before cutover.
3483    CandidateExited,
3484    /// The candidate declared itself ready but failed its health probe.
3485    CandidateUnhealthy,
3486    /// An operator stop, disable or retire arrived while the candidate warmed.
3487    /// The candidate was killed and the operator's command then carried out on
3488    /// the incumbent.
3489    Interrupted,
3490    /// The candidate's connection closed at the moment of cutover. If it
3491    /// closed before forwarding moved, the incumbent is untouched. If it closed
3492    /// between the forwarding and registry halves of cutover, forwarding can no
3493    /// longer route to the incumbent, so the module is restarted plainly.
3494    CutoverLost,
3495}
3496
3497impl SwapFailureArm {
3498    pub fn as_str(self) -> &'static str {
3499        match self {
3500            Self::SpawnFailed => "spawn_failed",
3501            Self::NeverRegistered => "never_registered",
3502            Self::NeverReady => "never_ready",
3503            Self::CandidateExited => "candidate_exited",
3504            Self::CandidateUnhealthy => "candidate_unhealthy",
3505            Self::Interrupted => "interrupted",
3506            Self::CutoverLost => "cutover_lost",
3507        }
3508    }
3509}
3510
3511impl fmt::Display for SuperviseError {
3512    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3513        match self {
3514            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3515            Self::Spawn {
3516                program,
3517                source,
3518                cgroup_path: Some(cgroup_path),
3519            } => write!(
3520                f,
3521                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3522                cgroup_path.display(),
3523                program.display()
3524            ),
3525            Self::Spawn {
3526                program,
3527                source,
3528                cgroup_path: None,
3529            } => write!(
3530                f,
3531                "failed to spawn module '{}': {source}",
3532                program.display()
3533            ),
3534            Self::Cgroup { module_id, source } => {
3535                write!(
3536                    f,
3537                    "failed to prepare cgroup for module '{module_id}': {source}"
3538                )
3539            }
3540            Self::LaunchNonce { reason } => {
3541                write!(
3542                    f,
3543                    "failed to generate reserved-module launch nonce: {reason}"
3544                )
3545            }
3546            Self::Wait { module_id, source } => {
3547                write!(f, "failed to wait for module '{module_id}': {source}")
3548            }
3549            Self::Kill { module_id, source } => {
3550                write!(f, "failed to kill module '{module_id}': {source}")
3551            }
3552            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3553            Self::Registry(err) => write!(f, "registry error: {err}"),
3554            Self::ReloadUnavailable { module_id, reason } => {
3555                write!(f, "reload unavailable for module '{module_id}': {reason}")
3556            }
3557            Self::Disabled { module_id } => {
3558                write!(
3559                    f,
3560                    "module '{module_id}' is disabled; enable it before restart or reload"
3561                )
3562            }
3563            Self::ReloadFailed { module_id, reason } => {
3564                write!(f, "reload failed for module '{module_id}': {reason}")
3565            }
3566            Self::RegistrationStillActive { module_id, waited } => write!(
3567                f,
3568                "module '{module_id}' registration remained active after waiting {waited:?}"
3569            ),
3570            Self::StatePoisoned { module_id } => match module_id {
3571                Some(module_id) => {
3572                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3573                }
3574                None => write!(f, "supervisor state was poisoned"),
3575            },
3576            Self::CommandClosed { module_id } => {
3577                write!(
3578                    f,
3579                    "supervisor command channel for module '{module_id}' is closed"
3580                )
3581            }
3582            Self::SwapInProgress { module_id } => write!(
3583                f,
3584                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3585            ),
3586            Self::SwapRefused { module_id, reason } => match reason {
3587                SwapRefusal::OverlapExclusive => write!(
3588                    f,
3589                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3590                ),
3591                SwapRefusal::NotRegistered => write!(
3592                    f,
3593                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3594                ),
3595                SwapRefusal::ProtocolNone => write!(
3596                    f,
3597                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3598                ),
3599                SwapRefusal::NotConfigured => write!(
3600                    f,
3601                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3602                ),
3603                SwapRefusal::AlreadySwapping => {
3604                    write!(f, "module '{module_id}' is already being swapped")
3605                }
3606            },
3607            Self::SwapFailed {
3608                module_id,
3609                arm,
3610                detail,
3611                ..
3612            } => write!(
3613                f,
3614                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3615                arm.as_str()
3616            ),
3617        }
3618    }
3619}
3620
3621impl Error for SuperviseError {
3622    fn source(&self) -> Option<&(dyn Error + 'static)> {
3623        match self {
3624            Self::Spawn { source, .. }
3625            | Self::Cgroup { source, .. }
3626            | Self::Wait { source, .. }
3627            | Self::Kill { source, .. } => Some(source),
3628            Self::Forwarding(err) => Some(err),
3629            Self::Registry(err) => Some(err),
3630            Self::LaunchNonce { .. }
3631            | Self::InvalidSpec { .. }
3632            | Self::ReloadUnavailable { .. }
3633            | Self::Disabled { .. }
3634            | Self::ReloadFailed { .. }
3635            | Self::RegistrationStillActive { .. }
3636            | Self::StatePoisoned { .. }
3637            | Self::CommandClosed { .. }
3638            | Self::SwapInProgress { .. }
3639            | Self::SwapRefused { .. }
3640            | Self::SwapFailed { .. } => None,
3641        }
3642    }
3643}
3644
3645pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3646    if spec.module_id.trim().is_empty() {
3647        return Err(SuperviseError::InvalidSpec {
3648            reason: "module_id must not be empty".to_string(),
3649        });
3650    }
3651
3652    Ok(())
3653}
3654
3655#[derive(Debug, Default)]
3656struct HealthProbeRuntime {
3657    configured_health: Option<HealthConfig>,
3658    registered_connection: Option<crate::ConnectionId>,
3659    advertised: bool,
3660    next_probe_at: Option<Instant>,
3661    probe_index: u64,
3662}
3663
3664fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
3665    lock_snapshot(snapshot)
3666        .ok()
3667        .and_then(|state| state.spawned_protocol)
3668        .unwrap_or(spec.protocol)
3669}
3670
3671impl HealthProbeRuntime {
3672    fn refresh_registration(
3673        &mut self,
3674        spec: &ModuleSpec,
3675        runtime: &SupervisorRuntimeConfig,
3676        registry: &Registry,
3677        snapshot: &SharedSnapshot,
3678    ) {
3679        if self.configured_health.as_ref() != Some(&runtime.health) {
3680            self.configured_health = Some(runtime.health.clone());
3681            self.next_probe_at = None;
3682            self.registered_connection = None;
3683            self.probe_index = 0;
3684        }
3685        // A non-wire process never registers. Only an explicitly configured
3686        // HTTP endpoint can arm its health probe; an absent HELLO is not a
3687        // health failure for that kind of process.
3688        if running_protocol(spec, snapshot) == ModuleProtocol::None {
3689            self.registered_connection = None;
3690            self.advertised = runtime.health.http.is_some();
3691            if !self.advertised {
3692                self.next_probe_at = None;
3693                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3694                    state.health = ModuleHealthStatus::default();
3695                });
3696            } else if self.next_probe_at.is_none() {
3697                self.next_probe_at = Some(
3698                    Instant::now()
3699                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3700                );
3701            }
3702            return;
3703        }
3704
3705        let registration = match registry.get_module(&spec.module_id) {
3706            Ok(registration) => registration,
3707            Err(err) => {
3708                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3709                self.advertised = false;
3710                self.next_probe_at = None;
3711                return;
3712            }
3713        };
3714
3715        let Some(registration) = registration else {
3716            self.registered_connection = None;
3717            self.advertised = false;
3718            self.next_probe_at = None;
3719            return;
3720        };
3721
3722        let advertised = registration
3723            .control_ops
3724            .iter()
3725            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
3726        if !advertised {
3727            self.registered_connection = Some(registration.connection_id);
3728            self.advertised = false;
3729            self.next_probe_at = None;
3730            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3731                state.health.status = SupervisorHealthStatus::Unknown;
3732                state.health.consecutive_failures = 0;
3733                state.health.last_probe_ms = None;
3734                state.health.detail = None;
3735                state.health.metrics = None;
3736            });
3737            return;
3738        }
3739
3740        let reregistered = self.registered_connection != Some(registration.connection_id);
3741        self.registered_connection = Some(registration.connection_id);
3742        self.advertised = true;
3743        if reregistered || self.next_probe_at.is_none() {
3744            self.probe_index = 0;
3745            self.next_probe_at = Some(
3746                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3747            );
3748            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3749                state.health.status = SupervisorHealthStatus::Unknown;
3750                state.health.consecutive_failures = 0;
3751                state.health.detail = None;
3752                state.health.metrics = None;
3753            });
3754        }
3755    }
3756
3757    fn wake_after(&self) -> Duration {
3758        if !self.advertised {
3759            return REGISTRY_RELEASE_POLL;
3760        }
3761        self.next_probe_at
3762            .map(|next| next.saturating_duration_since(Instant::now()))
3763            .unwrap_or(REGISTRY_RELEASE_POLL)
3764    }
3765
3766    fn due(&self) -> bool {
3767        self.advertised
3768            && self
3769                .next_probe_at
3770                .is_some_and(|next| Instant::now() >= next)
3771    }
3772
3773    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
3774        self.probe_index = self.probe_index.wrapping_add(1);
3775        self.next_probe_at = Some(
3776            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
3777        );
3778    }
3779}
3780
3781/// What a failed health probe actually OBSERVED, kept apart from how it reads.
3782///
3783/// This was a struct with a single `message: String`, and every one of the
3784/// fifteen construction sites collapsed into it. Each site knows exactly what it
3785/// saw -- the lane is gone, the module did not answer in time, the module
3786/// answered with the wrong thing -- and `handle_health_probe_failure` then
3787/// treated all of them identically: increment a counter, compare to a threshold,
3788/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
3789/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
3790///
3791/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
3792///
3793/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
3794///   answer on it again.
3795/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
3796///   AND with a perfectly healthy one that lost a CPU race -- which is what
3797///   happens under machine load, and is how this supervisor killed a healthy
3798///   module three times in one day.
3799/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
3800///   Restarting on it is defensible, but it is not the silence case and should
3801///   never be counted as one.
3802/// * `Misconfigured` is a daemon-side fault. The module has not been asked
3803///   anything, so it cannot be evidence about the module at all.
3804///
3805/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
3806/// one that fires most often, and while every variant collapsed into one string
3807/// it carried the same weight as the strongest.
3808///
3809/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
3810/// DESIGN and a reader stopping at it gets the build backwards: the restart
3811/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
3812/// probes still increment the failure streak and drive escalation at the
3813/// threshold (see `is_proof_of_death` below for why that is deliberate and
3814/// what gates the change). Absence of evidence restarts modules today.
3815#[derive(Debug)]
3816enum HealthProbeEvidence {
3817    /// The module's control lane is gone. Proof of death.
3818    LaneDead,
3819    /// No reply within the deadline. Proves nothing about the module's state.
3820    NoAnswer,
3821    /// The module replied, but not with a usable health report. Proves it is alive.
3822    BadAnswer,
3823    /// The daemon could not ask. Says nothing about the module.
3824    Misconfigured,
3825}
3826
3827#[derive(Debug)]
3828struct HealthProbeError {
3829    evidence: HealthProbeEvidence,
3830    message: String,
3831}
3832
3833impl HealthProbeError {
3834    fn lane_dead(message: impl Into<String>) -> Self {
3835        Self::with(HealthProbeEvidence::LaneDead, message)
3836    }
3837
3838    fn no_answer(message: impl Into<String>) -> Self {
3839        Self::with(HealthProbeEvidence::NoAnswer, message)
3840    }
3841
3842    fn bad_answer(message: impl Into<String>) -> Self {
3843        Self::with(HealthProbeEvidence::BadAnswer, message)
3844    }
3845
3846    fn misconfigured(message: impl Into<String>) -> Self {
3847        Self::with(HealthProbeEvidence::Misconfigured, message)
3848    }
3849
3850    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
3851        Self {
3852            evidence,
3853            message: message.into(),
3854        }
3855    }
3856
3857    /// Whether this observation is proof the module cannot serve.
3858    ///
3859    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
3860    /// variant that fires under CPU starvation, and treating it as proof is the
3861    /// defect this enum exists to make impossible to reintroduce silently.
3862    ///
3863    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
3864    /// to restart also needs a bound for the case it excludes -- a genuinely
3865    /// wedged module, alive but never answering -- and that bound must come from
3866    /// the distribution of real late-answer latencies, which nothing measures
3867    /// yet. Landing the classification first makes the later change a one-line
3868    /// decision against evidence that already exists, rather than two unproven
3869    /// changes at once.
3870    #[allow(dead_code)]
3871    fn is_proof_of_death(&self) -> bool {
3872        matches!(self.evidence, HealthProbeEvidence::LaneDead)
3873    }
3874
3875    /// Short stable label for logs and the health snapshot.
3876    ///
3877    /// An operator reading `ck health` currently cannot tell "the module is gone"
3878    /// from "the module did not answer in five seconds", because both render as
3879    /// prose in the same field. These labels are what make the two
3880    /// distinguishable at a glance, and they are what a later restart-policy
3881    /// change will be argued from.
3882    fn label(&self) -> &'static str {
3883        match self.evidence {
3884            HealthProbeEvidence::LaneDead => "lane-dead",
3885            HealthProbeEvidence::NoAnswer => "no-answer",
3886            HealthProbeEvidence::BadAnswer => "bad-answer",
3887            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
3888        }
3889    }
3890}
3891
3892impl fmt::Display for HealthProbeError {
3893    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3894        f.write_str(&self.message)
3895    }
3896}
3897
3898async fn run_health_probe_cycle(
3899    spec: &ModuleSpec,
3900    runtime: &SupervisorRuntimeConfig,
3901    registry: &Registry,
3902    process_liveness: &SupervisorProcessLiveness,
3903    snapshot: &SharedSnapshot,
3904    child: &mut Option<SupervisedChild>,
3905) {
3906    let now_ms = unix_ms_now();
3907    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
3908        .then_some(runtime.health.http.as_deref())
3909        .flatten();
3910    let result = match http {
3911        Some(url) => probe_http_health(url, runtime.health.deadline).await,
3912        None => probe_module_health(&spec.module_id, runtime, None).await,
3913    };
3914    match result {
3915        Ok(report) => {
3916            handle_health_report(
3917                spec,
3918                runtime,
3919                registry,
3920                process_liveness,
3921                snapshot,
3922                child,
3923                report,
3924                now_ms,
3925            )
3926            .await;
3927        }
3928        Err(err) => {
3929            if http.is_some() {
3930                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3931                    state.health.status = SupervisorHealthStatus::Failing;
3932                });
3933            }
3934            handle_health_probe_failure(
3935                spec,
3936                runtime,
3937                registry,
3938                process_liveness,
3939                snapshot,
3940                child,
3941                err,
3942                now_ms,
3943            )
3944            .await;
3945        }
3946    }
3947}
3948
3949pub(crate) struct HttpProbeTarget<'a> {
3950    address: std::net::SocketAddr,
3951    localhost: bool,
3952    authority: &'a str,
3953    path: String,
3954}
3955
3956/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
3957/// or TLS. A URL cannot turn a local health check into an outbound connection.
3958pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
3959    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
3960        return Err("must not contain whitespace, controls, or a fragment".into());
3961    }
3962    let rest = url
3963        .strip_prefix("http://")
3964        .ok_or("must use plain http://")?;
3965    let split = rest.find(['/', '?']).unwrap_or(rest.len());
3966    let (authority, suffix) = rest.split_at(split);
3967    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
3968        ("::1", rest)
3969    } else {
3970        let split = authority.find(':').unwrap_or(authority.len());
3971        authority.split_at(split)
3972    };
3973    let ip: std::net::IpAddr = match host {
3974        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
3975        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
3976        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
3977    };
3978    let port = if port.is_empty() {
3979        80
3980    } else {
3981        port.strip_prefix(':')
3982            .and_then(|p| p.parse::<u16>().ok())
3983            .filter(|p| *p > 0)
3984            .ok_or("must have a valid nonzero TCP port")?
3985    };
3986    let path = if suffix.is_empty() {
3987        "/".into()
3988    } else if suffix.starts_with('?') {
3989        format!("/{suffix}")
3990    } else {
3991        suffix.into()
3992    };
3993    Ok(HttpProbeTarget {
3994        address: std::net::SocketAddr::new(ip, port),
3995        localhost: host == "localhost",
3996        authority,
3997        path,
3998    })
3999}
4000
4001async fn probe_http_health(
4002    url: &str,
4003    deadline: Duration,
4004) -> Result<HealthReport, HealthProbeError> {
4005    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4006    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4007    // Keep partial diagnostics outside the timed future so cancellation does
4008    // not discard a status line or body bytes already received.
4009    let mut response_status = String::new();
4010    let mut body = Vec::new();
4011    let probe = async {
4012        // Resolve localhost ourselves so a hosts-file override cannot turn
4013        // this into an outbound request, while IPv6-only local servers work.
4014        let connection = match tokio::net::TcpStream::connect(target.address).await {
4015            Err(_) if target.localhost => {
4016                tokio::net::TcpStream::connect((
4017                    std::net::Ipv6Addr::LOCALHOST,
4018                    target.address.port(),
4019                ))
4020                .await
4021            }
4022            result => result,
4023        };
4024        let mut stream = connection.map_err(|error| {
4025            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4026        })?;
4027        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4028            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4029        let mut reader = BufReader::new(stream);
4030        let mut budget = 16 * 1024;
4031        let status = http_line(&mut reader, &mut budget).await?;
4032        let mut words = status.split_ascii_whitespace();
4033        let version = words.next();
4034        let code = words
4035            .next()
4036            .filter(|word| word.len() == 3)
4037            .and_then(|word| word.parse::<u16>().ok());
4038        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4039            || !code.is_some_and(|code| (100..600).contains(&code))
4040        {
4041            return Err(HealthProbeError::bad_answer(format!(
4042                "invalid HTTP status: {status}"
4043            )));
4044        }
4045        let code = code.expect("validated status code");
4046        response_status = status.clone();
4047        let mut length = None;
4048        let mut chunked = false;
4049        loop {
4050            let line = http_line(&mut reader, &mut budget).await?;
4051            if line.is_empty() {
4052                break;
4053            }
4054            if let Some((name, value)) = line.split_once(':') {
4055                if name.eq_ignore_ascii_case("content-length") {
4056                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4057                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4058                    })?);
4059                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4060                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4061                }
4062            }
4063        }
4064        if chunked {
4065            while body.len() < 200 {
4066                let line = http_line(&mut reader, &mut budget).await?;
4067                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4068                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4069                if size == 0 {
4070                    break;
4071                }
4072                let count = size.min((200 - body.len()) as u64) as usize;
4073                let start = body.len();
4074                (&mut reader)
4075                    .take(count as u64)
4076                    .read_to_end(&mut body)
4077                    .await
4078                    .map_err(|error| {
4079                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4080                    })?;
4081                if body.len() - start != count {
4082                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4083                }
4084                if size > count as u64 || body.len() == 200 {
4085                    break;
4086                }
4087                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4088                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4089                }
4090            }
4091        } else {
4092            reader
4093                .take(length.unwrap_or(200).min(200))
4094                .read_to_end(&mut body)
4095                .await
4096                .map_err(|error| {
4097                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4098                })?;
4099        }
4100        if (200..300).contains(&code) {
4101            Ok(HealthReport::ok())
4102        } else {
4103            Err(HealthProbeError::bad_answer(
4104                "HTTP health endpoint returned non-2xx",
4105            ))
4106        }
4107    };
4108    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4109        Err(HealthProbeError::no_answer(format!(
4110            "HTTP probe timed out after {deadline:?}"
4111        )))
4112    });
4113    if let Err(error) = &mut result {
4114        if !response_status.is_empty() {
4115            error.message = format!(
4116                "{}; {response_status}: {}",
4117                error.message,
4118                String::from_utf8_lossy(&body)
4119            );
4120        }
4121    }
4122    result
4123}
4124
4125async fn http_line(
4126    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4127    remaining: &mut usize,
4128) -> Result<String, HealthProbeError> {
4129    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4130    let mut line = Vec::new();
4131    (&mut *reader)
4132        .take(*remaining as u64)
4133        .read_until(b'\n', &mut line)
4134        .await
4135        .map_err(|error| {
4136            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4137        })?;
4138    *remaining -= line.len();
4139    if !line.ends_with(b"\r\n") {
4140        return Err(HealthProbeError::bad_answer(
4141            "HTTP headers are incomplete or exceed 16 KiB",
4142        ));
4143    }
4144    line.truncate(line.len() - 2);
4145    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4146}
4147
4148async fn probe_module_health(
4149    module_id: &str,
4150    runtime: &SupervisorRuntimeConfig,
4151    drain_deadline: Option<Instant>,
4152) -> Result<HealthReport, HealthProbeError> {
4153    let Some(forwarding) = runtime.forwarding.as_ref() else {
4154        return Err(HealthProbeError::misconfigured(
4155            "supervisor was not configured with a forwarding table",
4156        ));
4157    };
4158    let probe_started_at = Instant::now();
4159    let mut deadline = probe_started_at + runtime.health.deadline;
4160    if let Some(drain_deadline) = drain_deadline {
4161        deadline = deadline.min(drain_deadline);
4162    }
4163    let pending = if drain_deadline.is_some() {
4164        forwarding.begin_drain_health_probe_rpc_for(
4165            module_id,
4166            MODULE_CONTROL_OP_HEALTH_CHECK,
4167            probe_started_at,
4168            deadline,
4169        )
4170    } else {
4171        forwarding.begin_health_probe_rpc_for(
4172            module_id,
4173            MODULE_CONTROL_OP_HEALTH_CHECK,
4174            probe_started_at,
4175            deadline,
4176        )
4177    }
4178    .map_err(|err| {
4179        // The endpoint is not registered, so there is no live control lane to
4180        // ask. That is the module being absent, not slow.
4181        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4182    })?;
4183    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4184}
4185
4186/// [`probe_module_health`] for one endpoint rather than the id's active one.
4187///
4188/// A swap probes two processes that no by-id lookup reaches: its candidate
4189/// before cutover, and its superseded incumbent (for busy gauges) while the
4190/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4191/// bounds the by-id drain probe.
4192async fn probe_endpoint_health(
4193    endpoint: crate::ModuleEndpointId,
4194    runtime: &SupervisorRuntimeConfig,
4195    deadline_cap: Option<Instant>,
4196) -> Result<HealthReport, HealthProbeError> {
4197    let Some(forwarding) = runtime.forwarding.as_ref() else {
4198        return Err(HealthProbeError::misconfigured(
4199            "supervisor was not configured with a forwarding table",
4200        ));
4201    };
4202    let probe_started_at = Instant::now();
4203    let mut deadline = probe_started_at + runtime.health.deadline;
4204    if let Some(cap) = deadline_cap {
4205        deadline = deadline.min(cap);
4206    }
4207    let pending = forwarding
4208        .begin_endpoint_health_probe_rpc_for(
4209            endpoint,
4210            MODULE_CONTROL_OP_HEALTH_CHECK,
4211            probe_started_at,
4212            deadline,
4213        )
4214        .map_err(|err| {
4215            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4216        })?;
4217    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4218}
4219
4220/// Send a begun health probe and classify its answer.
4221async fn await_health_probe(
4222    forwarding: &ForwardingTable,
4223    pending: PendingModuleControlRpc,
4224    deadline: Instant,
4225    probe_budget: Duration,
4226) -> Result<HealthReport, HealthProbeError> {
4227    let PendingModuleControlRpc {
4228        endpoint,
4229        module_sink,
4230        negotiated_ver,
4231        corr,
4232        receiver,
4233    } = pending;
4234    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4235        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4236    })?;
4237    let frame = Frame::build_with_version(
4238        negotiated_ver,
4239        FrameType::Request,
4240        control_flags(),
4241        0,
4242        0,
4243        corr,
4244        body,
4245    )
4246    .map_err(|err| {
4247        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4248    })?;
4249
4250    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4251    // blocks waiting for capacity when the module's egress queue is full, and an
4252    // unbounded await here freezes the whole supervision actor (it stops polling
4253    // Child::wait and supervisor commands), making the module unrecoverable
4254    // in-band. On timeout the probe fails like any transport failure.
4255    match timeout_at(deadline, module_sink.send(frame)).await {
4256        Ok(Ok(())) => {}
4257        Ok(Err(err)) => {
4258            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4259            // A closed sink means the module's egress channel is gone -- the
4260            // receiving half is dropped when its connection tears down. Proof.
4261            return Err(HealthProbeError::lane_dead(format!(
4262                "failed to send health.check: {err}"
4263            )));
4264        }
4265        Err(_elapsed) => {
4266            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4267            // A full egress queue means the module is not draining its socket, which
4268            // is consistent with a wedged module AND with one whose reader is merely
4269            // starved. Silence, not proof.
4270            return Err(HealthProbeError::no_answer(
4271                "health.check send timed out before enqueue (module egress full)",
4272            ));
4273        }
4274    }
4275
4276    match timeout_at(deadline, receiver).await {
4277        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4278        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4279        // and those prove it is alive even though the probe failed.
4280        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4281            response.health_report().ok_or_else(|| {
4282                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4283            })
4284        }
4285        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4286            format!("health.check rejected: {}", body.message),
4287        )),
4288        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4289            Err(HealthProbeError::lane_dead(message))
4290        }
4291        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4292            Err(HealthProbeError::bad_answer(message))
4293        }
4294        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4295            Err(HealthProbeError::bad_answer(format!(
4296                "expected module-control op '{expected}', got '{actual}'"
4297            )))
4298        }
4299        // A reply that crosses the deadline before this waiter observes it is
4300        // still proof of life. The forwarding path records its end-to-end latency
4301        // before delivering this classification.
4302        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4303            "module answered health.check after its daemon deadline",
4304        )),
4305        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4306            "health.check waiter was canceled before the module responded",
4307        )),
4308        Err(_) => {
4309            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4310            Err(HealthProbeError::no_answer(format!(
4311                "module did not answer health.check within {probe_budget:?}"
4312            )))
4313        }
4314    }
4315}
4316
4317#[allow(clippy::too_many_arguments)]
4318async fn handle_health_report(
4319    spec: &ModuleSpec,
4320    runtime: &SupervisorRuntimeConfig,
4321    registry: &Registry,
4322    process_liveness: &SupervisorProcessLiveness,
4323    snapshot: &SharedSnapshot,
4324    child: &mut Option<SupervisedChild>,
4325    report: HealthReport,
4326    now_ms: u64,
4327) {
4328    let status = supervisor_health_status(report.status);
4329    let detail = report.detail.clone();
4330    let metrics = truncate_health_metrics(report.metrics);
4331    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4332        state.health.status = status;
4333        state.health.last_probe_ms = Some(now_ms);
4334        state.health.detail = detail.clone();
4335        state.health.metrics = metrics.clone();
4336        state.health.consecutive_failures = 0;
4337    });
4338
4339    let action = match report.status {
4340        HealthStatus::Ok => return,
4341        HealthStatus::Degraded => runtime.health.on_degraded,
4342        HealthStatus::Failing => runtime.health.on_failing,
4343    };
4344    apply_l3_health_action(
4345        spec,
4346        runtime,
4347        registry,
4348        process_liveness,
4349        snapshot,
4350        child,
4351        status,
4352        detail.as_deref(),
4353        action,
4354        now_ms,
4355    )
4356    .await;
4357}
4358
4359#[allow(clippy::too_many_arguments)]
4360async fn handle_health_probe_failure(
4361    spec: &ModuleSpec,
4362    runtime: &SupervisorRuntimeConfig,
4363    registry: &Registry,
4364    process_liveness: &SupervisorProcessLiveness,
4365    snapshot: &SharedSnapshot,
4366    child: &mut Option<SupervisedChild>,
4367    err: HealthProbeError,
4368    now_ms: u64,
4369) {
4370    let threshold = runtime.health.failure_threshold.max(1);
4371    let mut failures = 0;
4372    // Carry the evidence class into the operator-visible detail. Without it,
4373    // "module did not answer within 5s" and "the control lane is gone" are two
4374    // prose strings in the same field, and the reader has to know the codebase to
4375    // tell which one is proof of anything.
4376    let detail = format!("[{}] {err}", err.label());
4377    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4378        state.health.last_probe_ms = Some(now_ms);
4379        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4380        state.health.detail = Some(detail.clone());
4381        state.health.metrics = None;
4382        failures = state.health.consecutive_failures;
4383    });
4384
4385    if failures < threshold {
4386        warn!(
4387            module_id = %spec.module_id,
4388            consecutive_failures = failures,
4389            threshold,
4390            evidence = err.label(),
4391            detail = %detail,
4392            "health.check probe failed"
4393        );
4394        return;
4395    }
4396
4397    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4398        state.state = ModuleState::Unresponsive;
4399        state.health.status = SupervisorHealthStatus::Unresponsive;
4400    });
4401    // The evidence class is logged at the kill site because this is the line an
4402    // operator reads after an unexplained restart. A streak of `no-answer` under
4403    // machine load is the known false-positive shape; a `lane-dead` is not.
4404    if runtime.health.critical {
4405        error!(
4406            module_id = %spec.module_id,
4407            status = "unresponsive",
4408            evidence = err.label(),
4409            detail = %detail,
4410            "critical module health alert"
4411        );
4412    } else {
4413        warn!(
4414            module_id = %spec.module_id,
4415            status = "unresponsive",
4416            evidence = err.label(),
4417            detail = %detail,
4418            "module health threshold breached"
4419        );
4420    }
4421    if let Err(err) = health_restart_child(
4422        spec,
4423        runtime,
4424        registry,
4425        process_liveness,
4426        snapshot,
4427        child,
4428        SupervisorHealthStatus::Unresponsive,
4429        Some(&detail),
4430        now_ms,
4431    )
4432    .await
4433    {
4434        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4435    }
4436}
4437
4438#[allow(clippy::too_many_arguments)]
4439async fn apply_l3_health_action(
4440    spec: &ModuleSpec,
4441    runtime: &SupervisorRuntimeConfig,
4442    registry: &Registry,
4443    process_liveness: &SupervisorProcessLiveness,
4444    snapshot: &SharedSnapshot,
4445    child: &mut Option<SupervisedChild>,
4446    status: SupervisorHealthStatus,
4447    detail: Option<&str>,
4448    action: HealthAction,
4449    now_ms: u64,
4450) {
4451    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4452    match action {
4453        HealthAction::Report => {
4454            info!(
4455                module_id = %spec.module_id,
4456                status = ?status,
4457                detail,
4458                "module reported non-ok health"
4459            );
4460        }
4461        HealthAction::Alert => {
4462            error!(
4463                module_id = %spec.module_id,
4464                status = ?status,
4465                detail,
4466                "module health alert"
4467            );
4468        }
4469        HealthAction::Restart => {
4470            if let Err(err) = health_restart_child(
4471                spec,
4472                runtime,
4473                registry,
4474                process_liveness,
4475                snapshot,
4476                child,
4477                status,
4478                detail,
4479                now_ms,
4480            )
4481            .await
4482            {
4483                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4484            }
4485        }
4486    }
4487}
4488
4489#[allow(clippy::too_many_arguments)]
4490async fn health_restart_child(
4491    spec: &ModuleSpec,
4492    runtime: &SupervisorRuntimeConfig,
4493    registry: &Registry,
4494    process_liveness: &SupervisorProcessLiveness,
4495    snapshot: &SharedSnapshot,
4496    child: &mut Option<SupervisedChild>,
4497    status: SupervisorHealthStatus,
4498    detail: Option<&str>,
4499    now_ms: u64,
4500) -> Result<(), SuperviseError> {
4501    let (enabled, schedule) = {
4502        let mut state = lock_snapshot(snapshot)?;
4503        let enabled = state.enabled;
4504        let schedule = if enabled {
4505            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4506        } else {
4507            None
4508        };
4509        (enabled, schedule)
4510    };
4511
4512    if !enabled {
4513        return Err(SuperviseError::Disabled {
4514            module_id: spec.module_id.clone(),
4515        });
4516    }
4517
4518    if schedule.is_none() {
4519        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4520        error!(
4521            module_id = %spec.module_id,
4522            status = ?status,
4523            detail,
4524            max_restarts = runtime.restart_policy.max_restarts,
4525            window_secs = runtime.restart_policy.window.as_secs(),
4526            reason = %runtime.restart_policy.budget_exhausted_detail(),
4527            "health restart budget exhausted; marking module failed"
4528        );
4529        let stop_notice = begin_forwarding_drain_if_configured(
4530            spec,
4531            runtime,
4532            registry,
4533            snapshot,
4534            Some(true),
4535            RouteCloseReason::Disable,
4536        )
4537        .await?;
4538        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4539            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4540        })?;
4541        drain_optional_child(
4542            &spec.module_id,
4543            spec.protocol,
4544            stop_notice,
4545            registry,
4546            runtime.forwarding.as_deref(),
4547            snapshot,
4548            &runtime.terminal_ring,
4549            &runtime.spawn_events,
4550            child,
4551            runtime.drain_timeout,
4552            ModuleState::Failed,
4553            Some(true),
4554        )
4555        .await?;
4556        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4557        return Ok(());
4558    }
4559
4560    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4561    let mut restart_count = 0;
4562    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4563        restart_count = state.crash_restarts.len();
4564        state.state = ModuleState::Unresponsive;
4565        state.health.status = status;
4566        state.health.last_action = Some(HealthAction::Restart.to_string());
4567        state.health.last_action_ms = Some(now_ms);
4568    })?;
4569    warn!(
4570        module_id = %spec.module_id,
4571        status = ?status,
4572        detail,
4573        restart_count,
4574        restart_in_window = schedule.restart_in_window,
4575        delay_ms = schedule.delay.as_millis() as u64,
4576        "health-triggered module restart"
4577    );
4578
4579    let stop_notice = begin_forwarding_drain_if_configured(
4580        spec,
4581        runtime,
4582        registry,
4583        snapshot,
4584        Some(true),
4585        RouteCloseReason::Restart,
4586    )
4587    .await?;
4588    drain_optional_child(
4589        &spec.module_id,
4590        spec.protocol,
4591        stop_notice,
4592        registry,
4593        runtime.forwarding.as_deref(),
4594        snapshot,
4595        &runtime.terminal_ring,
4596        &runtime.spawn_events,
4597        child,
4598        runtime.drain_timeout,
4599        ModuleState::Restarting,
4600        Some(true),
4601    )
4602    .await?;
4603    schedule_respawn(
4604        runtime,
4605        snapshot,
4606        &spec.module_id,
4607        schedule.delay,
4608        RespawnKind::Spawn,
4609    )
4610}
4611
4612fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4613    if let Some(reply) = runtime
4614        .deferred_reload_reply
4615        .lock()
4616        .unwrap_or_else(|p| p.into_inner())
4617        .take()
4618    {
4619        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4620            module_id: module_id.to_string(),
4621            reason: reason.to_string(),
4622        }));
4623    }
4624}
4625
4626fn schedule_respawn(
4627    runtime: &SupervisorRuntimeConfig,
4628    snapshot: &SharedSnapshot,
4629    module_id: &str,
4630    delay: Duration,
4631    kind: RespawnKind,
4632) -> Result<(), SuperviseError> {
4633    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4634    update_snapshot(snapshot, Some(module_id), |state| {
4635        state.respawn_pending = true
4636    })?;
4637    *runtime
4638        .scheduled_respawn
4639        .lock()
4640        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
4641        deadline: Instant::now() + delay,
4642        kind,
4643    });
4644    Ok(())
4645}
4646
4647fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4648    let _ = update_snapshot(snapshot, Some(module_id), |state| {
4649        state.health.last_action = Some(action);
4650        state.health.last_action_ms = Some(now_ms);
4651    });
4652}
4653
4654fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4655    match status {
4656        HealthStatus::Ok => SupervisorHealthStatus::Ok,
4657        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4658        HealthStatus::Failing => SupervisorHealthStatus::Failing,
4659    }
4660}
4661
4662/// Caps the metrics blob stored in the cached supervisor snapshot, which is
4663/// returned to every `supervisor.list` and `supervisor.health` caller.
4664///
4665/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
4666/// path: that request exists to return a module's complete metrics object, and
4667/// `ck health <module-id>` documents it as the way to see what the cached view
4668/// truncates. The asymmetry is the feature.
4669///
4670/// So a new caller must decide which side it is on rather than assume the cap is
4671/// universal. Reaching for it on a fresh-probe path would silently reintroduce
4672/// the truncation that path exists to avoid.
4673fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4674    let metrics = metrics?;
4675    match serde_json::to_vec(&metrics) {
4676        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4677            "truncated": true,
4678            "original_bytes": encoded.len(),
4679        })),
4680        Ok(_) | Err(_) => Some(metrics),
4681    }
4682}
4683
4684/// Spread health probes so a fleet-wide restart does not converge them.
4685///
4686/// The delay is derived from the module id and probe index rather than a random
4687/// source, so it is deterministic per module: a module keeps its own offset
4688/// across daemon restarts instead of re-rolling into a collision.
4689fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4690    if cadence.is_zero() {
4691        return Duration::ZERO;
4692    }
4693    let cadence_ms = cadence.as_millis() as u64;
4694    // This early return is REDUNDANT, deliberately, and a mutation run will show
4695    // it surviving removal. Recording why here so the next person to notice does
4696    // not have to re-derive it:
4697    //
4698    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
4699    //   a zero cadence and builds the Duration from whole milliseconds, so a
4700    //   sub-millisecond cadence cannot come from config.
4701    // - Even if reached it changes no answer. The `.max(1)` below makes the span
4702    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
4703    //   -- exactly what this returns.
4704    //
4705    // Kept as a guard against a future widening of the config parser (accepting
4706    // microseconds, say), which would make the sub-millisecond case reachable.
4707    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
4708    // divides by zero. Remove this and nothing changes.
4709    if cadence_ms == 0 {
4710        return cadence;
4711    }
4712    // Note that this never returns less than one cadence, including for the FIRST
4713    // probe. So a freshly registered module reports health `unknown` for a full
4714    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
4715    // ready to answer.
4716    //
4717    // That is a property of the supervisor's schedule, not of any module: an
4718    // operator watching a restart sees `unknown` and cannot tell it from a module
4719    // that is slow to warm. Measured on two unrelated modules, both flipping to
4720    // `ok` between 22s and 32s after restart.
4721    //
4722    // Left as-is because spreading the first probe is what keeps a fleet-wide
4723    // restart from firing fourteen simultaneous probes into a cold machine. The
4724    // alternative -- probe at t+0 and jitter only from the second onward -- trades
4725    // that thundering herd for a faster first reading.
4726    let jitter_span = (cadence_ms / 10).max(1);
4727    let hash = module_id.as_bytes().iter().fold(
4728        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
4729        |acc, byte| {
4730            acc.wrapping_mul(1099511628211)
4731                .wrapping_add(u64::from(*byte))
4732        },
4733    );
4734    cadence + Duration::from_millis(hash % jitter_span)
4735}
4736
4737#[cfg(test)]
4738mod tests {
4739    use super::*;
4740
4741    #[test]
4742    fn readding_a_module_clears_its_rescan_removal_tombstone() {
4743        let handle = SupervisorHandle::new();
4744        let module_id = "readded-tombstone";
4745        handle.record_rescan_removal(module_id);
4746        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
4747
4748        handle.apply_identity_configuration(&ModuleSpec {
4749            module_id: module_id.to_string(),
4750            program: PathBuf::from("/test/module"),
4751            args: Vec::new(),
4752            env: Vec::new(),
4753            reserved: false,
4754            reserved_prefixes: Vec::new(),
4755            protocol: ModuleProtocol::Subc,
4756            overlap: Default::default(),
4757        });
4758
4759        assert!(
4760            handle.removal_tombstone_age_ms(module_id).is_none(),
4761            "a re-added module must not retain a stale removal tombstone"
4762        );
4763    }
4764
4765    /// What one module's owner looked like from the control plane at the
4766    /// instant after its first process was spawned.
4767    #[derive(Debug, PartialEq, Eq)]
4768    struct OwnerInSpawnWindow {
4769        module_id: String,
4770        configured: bool,
4771        on_roster: bool,
4772        admission_refusal: Option<&'static str>,
4773    }
4774
4775    /// A supervised module's process can connect, register, sync its scopes
4776    /// and describe them as soon as it is spawned, which is BEFORE the
4777    /// supervisor puts the module on the roster. In that window the owner must
4778    /// already read as configured, so a scoped `route.open` against it is
4779    /// refused as retryable `scope_not_synced` and not as terminal
4780    /// `scope_not_live` ("will never sync").
4781    ///
4782    /// The hook runs in exactly that window on every path that takes on a new
4783    /// module, so no race with a real child is needed: `on_roster: false`
4784    /// proves each observation was taken before the roster insert.
4785    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4786    async fn a_new_module_is_configured_before_its_first_process_can_register() {
4787        use crate::scopes::ScopeTable;
4788        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
4789
4790        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
4791        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
4792            module_id: module_id.to_string(),
4793            program,
4794            args: Vec::new(),
4795            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
4796                .into_iter()
4797                .map(|key| (key.to_string(), dir.path().display().to_string()))
4798                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
4799                .collect(),
4800            reserved: false,
4801            reserved_prefixes: Vec::new(),
4802            protocol: ModuleProtocol::Subc,
4803            overlap: Default::default(),
4804        };
4805        let live = super::terminal_history_tests::fake_aft_stub_path();
4806        let missing = dir.path().join("definitely-missing-module");
4807
4808        let handle = SupervisorHandle::new();
4809        let mut supervisor =
4810            Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
4811                .with_handle(handle.clone());
4812        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
4813        let hook_handle = handle.clone();
4814        let hook_observed = Arc::clone(&observed);
4815        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
4816            // Exactly what the control plane computes for a scoped route.open
4817            // naming this module as the owner of a scope it has not synced.
4818            let configured = hook_handle.is_configured(module_id);
4819            let selector = ScopeSelector {
4820                owner: Principal::Reserved {
4821                    module_id: module_id.to_string(),
4822                },
4823                scope_ref: "s".to_string(),
4824                scope_epoch: Some(1),
4825            };
4826            let carrier = Principal::Reserved {
4827                module_id: "carrier".to_string(),
4828            };
4829            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
4830                .admit(&carrier, module_id, &selector, configured)
4831            {
4832                Ok(_) => None,
4833                Err(refusal) => Some(refusal.code),
4834            };
4835            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
4836                module_id: module_id.to_string(),
4837                configured,
4838                on_roster: hook_handle.get(module_id).is_some(),
4839                admission_refusal,
4840            });
4841        })));
4842
4843        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
4844        let configured = supervisor
4845            .supervise_configured(stub("configured", live.clone()), true)
4846            .unwrap();
4847        let with_health = supervisor
4848            .supervise_configured_with_health(
4849                stub("with-health", live.clone()),
4850                true,
4851                HealthConfig::default(),
4852                None,
4853                RestartPolicy::default(),
4854            )
4855            .unwrap();
4856        // The failed-spawn path still puts the module on the roster (as
4857        // failed), so it is configured throughout.
4858        let failed = supervisor
4859            .supervise_configured_with_health(
4860                stub("failed-spawn", missing.clone()),
4861                true,
4862                HealthConfig::default(),
4863                None,
4864                RestartPolicy::default(),
4865            )
4866            .unwrap();
4867        // A failed plain `spawn` puts nothing on the roster, so its mark is
4868        // taken back once the spawn has failed.
4869        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
4870
4871        let expected = [
4872            "plain",
4873            "configured",
4874            "with-health",
4875            "failed-spawn",
4876            "spawn-error",
4877        ]
4878        .into_iter()
4879        .map(|module_id| OwnerInSpawnWindow {
4880            module_id: module_id.to_string(),
4881            configured: true,
4882            on_roster: false,
4883            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
4884        })
4885        .collect::<Vec<_>>();
4886        assert_eq!(*observed.lock().unwrap(), expected);
4887
4888        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
4889            assert!(
4890                handle.get(module_id).is_some(),
4891                "{module_id} is on the roster"
4892            );
4893            assert!(
4894                handle.is_configured(module_id),
4895                "{module_id} stays configured"
4896            );
4897        }
4898        assert!(handle.get("spawn-error").is_none());
4899        assert!(
4900            !handle.is_configured("spawn-error"),
4901            "a plain spawn that failed must not leave its module marked configured"
4902        );
4903
4904        // Leaving the roster clears the mark with it.
4905        handle.retire("failed-spawn");
4906        assert!(!handle.is_configured("failed-spawn"));
4907
4908        for module in [plain, configured, with_health] {
4909            module.stop().await.unwrap();
4910        }
4911        drop(failed);
4912    }
4913
4914    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
4915        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
4916        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
4917            snapshot.process_alive = true;
4918            snapshot.pid = Some(41);
4919            snapshot.spawned_at_ms = Some(42);
4920            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
4921            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
4922                device: 43,
4923                inode: 44,
4924            });
4925        })
4926        .unwrap();
4927        snapshot
4928    }
4929
4930    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
4931        let snapshot = lock_snapshot(snapshot).unwrap();
4932        assert!(!snapshot.process_alive);
4933        assert_eq!(snapshot.pid, None);
4934        assert_eq!(snapshot.spawned_at_ms, None);
4935        assert_eq!(snapshot.spawned_from, None);
4936        assert_eq!(snapshot.spawned_file_identity, None);
4937    }
4938
4939    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4940    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
4941        let supervisor = Supervisor::default();
4942        let mut runtime = supervisor.runtime_config();
4943        runtime.test_seed_stale_facts_before_enable_spawn = true;
4944        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
4945        let mut child = None;
4946        let spec = ModuleSpec {
4947            module_id: "failed-enable-clears-facts".to_string(),
4948            program: PathBuf::from("/definitely/missing/failed-enable-module"),
4949            args: Vec::new(),
4950            env: Vec::new(),
4951            reserved: false,
4952            reserved_prefixes: Vec::new(),
4953            protocol: ModuleProtocol::Subc,
4954            overlap: Default::default(),
4955        };
4956
4957        let result = set_child_enabled(
4958            &spec,
4959            &runtime,
4960            &supervisor.registry,
4961            &supervisor.process_liveness,
4962            &snapshot,
4963            &mut child,
4964            true,
4965        )
4966        .await;
4967
4968        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
4969        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4970        assert_snapshot_process_facts_cleared(&snapshot);
4971    }
4972
4973    #[tokio::test]
4974    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
4975        let supervisor = Supervisor::default();
4976        let runtime = supervisor.runtime_config();
4977        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
4978            ModuleState::Restarting,
4979            true,
4980        )));
4981        let spec = ModuleSpec {
4982            module_id: "start-stranded-restarting".to_string(),
4983            program: super::terminal_history_tests::fake_aft_stub_path(),
4984            args: Vec::new(),
4985            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
4986            reserved: false,
4987            reserved_prefixes: Vec::new(),
4988            protocol: ModuleProtocol::None,
4989            overlap: Default::default(),
4990        };
4991        let mut child = None;
4992        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
4993        assert!(!super::set_child_enabled(
4994            &spec,
4995            &runtime,
4996            &Registry::default(),
4997            &supervisor.process_liveness,
4998            &snapshot,
4999            &mut child,
5000            true
5001        )
5002        .await
5003        .unwrap());
5004        assert!(child.is_none());
5005        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5006        assert!(super::set_child_enabled(
5007            &spec,
5008            &runtime,
5009            &Registry::default(),
5010            &supervisor.process_liveness,
5011            &snapshot,
5012            &mut child,
5013            true
5014        )
5015        .await
5016        .unwrap());
5017        assert_eq!(
5018            lock_snapshot(&snapshot).unwrap().state,
5019            ModuleState::Running
5020        );
5021        let mut child = child.unwrap();
5022        child.start_kill().unwrap();
5023        child.wait().await.unwrap();
5024    }
5025
5026    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5027    async fn failed_reload_spawn_clears_current_process_facts() {
5028        let supervisor = Supervisor::default();
5029        let mut runtime = supervisor.runtime_config();
5030        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5031        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5032        let mut child = None;
5033        let spec = ModuleSpec {
5034            module_id: "failed-reload-clears-facts".to_string(),
5035            program: PathBuf::from("/unused/failed-reload-module"),
5036            args: Vec::new(),
5037            env: Vec::new(),
5038            reserved: false,
5039            reserved_prefixes: Vec::new(),
5040            protocol: ModuleProtocol::Subc,
5041            overlap: Default::default(),
5042        };
5043
5044        let result = handle_reload_spawn_failure(
5045            &spec,
5046            &runtime,
5047            &supervisor.process_liveness,
5048            &snapshot,
5049            &mut child,
5050            "forced reload spawn failure".to_string(),
5051        )
5052        .await;
5053
5054        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5055        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5056        assert_snapshot_process_facts_cleared(&snapshot);
5057    }
5058
5059    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5060    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5061        let supervisor = Supervisor::default();
5062        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5063        let module = supervisor.supervised_module(
5064            ModuleSpec {
5065                module_id: "drop-clears-facts".to_string(),
5066                program: PathBuf::from("/unused/drop-module"),
5067                args: Vec::new(),
5068                env: Vec::new(),
5069                reserved: false,
5070                reserved_prefixes: Vec::new(),
5071                protocol: ModuleProtocol::Subc,
5072                overlap: Default::default(),
5073            },
5074            supervisor.runtime_config(),
5075            Arc::clone(&snapshot),
5076            None,
5077        );
5078        assert!(!module
5079            .inner
5080            .monitor
5081            .lock()
5082            .unwrap()
5083            .as_ref()
5084            .unwrap()
5085            .is_finished());
5086
5087        drop(module);
5088
5089        assert_eq!(
5090            lock_snapshot(&snapshot).unwrap().state,
5091            ModuleState::Stopped
5092        );
5093        assert_snapshot_process_facts_cleared(&snapshot);
5094    }
5095
5096    #[cfg(unix)]
5097    #[tokio::test]
5098    async fn rescan_preserves_running_protocol_until_respawn() {
5099        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5100        let initial = ModuleSpec {
5101            module_id: "rescan-protocol".into(),
5102            program: PathBuf::from("/bin/sleep"),
5103            args: vec!["60".into()],
5104            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5105                .into_iter()
5106                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5107                .collect(),
5108            reserved: false,
5109            reserved_prefixes: vec![],
5110            protocol: ModuleProtocol::None,
5111            overlap: Default::default(),
5112        };
5113        let supervisor = Supervisor::default();
5114        let module = supervisor.spawn(initial.clone()).unwrap();
5115        assert!(module.status().unwrap().live);
5116        let mut next = initial;
5117        next.protocol = ModuleProtocol::Subc;
5118        module
5119            .update_configuration(next.clone(), HealthConfig::default(), None)
5120            .await
5121            .unwrap();
5122        assert!(
5123            module.status().unwrap().live,
5124            "rescan must not require HELLO from the old non-wire process"
5125        );
5126        let runtime = supervisor.runtime_config();
5127        let action = on_child_exit(
5128            &next,
5129            RestartPolicy::default(),
5130            &supervisor.registry,
5131            &module.inner.snapshot,
5132            &runtime.terminal_ring,
5133            &runtime.spawn_events,
5134            &runtime.child_roster,
5135            ExitReport {
5136                kind: ExitKind::Clean,
5137                code: Some(0),
5138                signal: None,
5139                at_ms: unix_ms_now(),
5140            },
5141        )
5142        .await;
5143        assert!(
5144            matches!(action, NextAction::Restart { .. }),
5145            "the old non-wire process's clean exit must restart"
5146        );
5147        module.drain().await.unwrap();
5148    }
5149
5150    #[cfg(unix)]
5151    #[tokio::test]
5152    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5153        use std::os::unix::fs::PermissionsExt;
5154        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5155        let script = dir.join("module.sh");
5156        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5157        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5158        let record_path = dir.join("live-children.json");
5159        let supervisor = Supervisor::default().with_live_children_record(&record_path);
5160        for (program, args) in [
5161            (PathBuf::from("sleep"), vec!["60".into()]),
5162            (script, vec![]),
5163        ] {
5164            let spec = ModuleSpec {
5165                module_id: "image-identity".into(),
5166                program,
5167                args,
5168                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5169                    .into_iter()
5170                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5171                    .collect(),
5172                reserved: false,
5173                reserved_prefixes: vec![],
5174                protocol: ModuleProtocol::None,
5175                overlap: Default::default(),
5176            };
5177            let module = supervisor.spawn(spec).unwrap();
5178            let entry = crate::live_children::read_record(&record_path)
5179                .unwrap()
5180                .pop()
5181                .unwrap();
5182            let observed = subc_os::Process::open(entry.pid)
5183                .unwrap()
5184                .unwrap()
5185                .observe()
5186                .unwrap();
5187            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5188            module.drain().await.unwrap();
5189            assert_eq!(verdict, crate::live_children::IdentityVerdict::Matches);
5190        }
5191    }
5192
5193    #[cfg(unix)]
5194    fn http_fixture(
5195        dir: &std::path::Path,
5196        url: &str,
5197        threshold: u32,
5198    ) -> crate::daemon_config::ConfiguredModule {
5199        let path = dir.join("subc.jsonc");
5200        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5201            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5202            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5203            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5204            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5205        }}}).to_string()).unwrap();
5206        crate::daemon_config::load(&path)
5207            .unwrap()
5208            .unwrap()
5209            .modules
5210            .pop()
5211            .unwrap()
5212    }
5213
5214    #[cfg(unix)]
5215    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5216        timeout(Duration::from_secs(5), async {
5217            loop {
5218                if module.status().unwrap().health.status == status {
5219                    break;
5220                }
5221                sleep(Duration::from_millis(5)).await;
5222            }
5223        })
5224        .await
5225        .unwrap_or_else(|_| {
5226            panic!(
5227                "expected {status:?}, got {:?}",
5228                module.status().unwrap().health
5229            )
5230        });
5231    }
5232
5233    #[cfg(unix)]
5234    #[tokio::test]
5235    async fn http_health_status_flips_ok_failing_ok() {
5236        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5237        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5238        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5239        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5240        let serving_status = status.clone();
5241        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5242        let server = tokio::spawn(async move {
5243            loop {
5244                let (mut stream, _) = listener.accept().await.unwrap();
5245                let mut request = [0u8; 2048];
5246                let count = stream.read(&mut request).await.unwrap();
5247                assert!(count > 0, "a probe must send an HTTP request");
5248                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5249                let body = if code == 200 {
5250                    "ready"
5251                } else {
5252                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5253                };
5254                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5255                let _ = stream.write_all(response.as_bytes()).await;
5256            }
5257        });
5258        let configured = http_fixture(&dir, &url, 1000);
5259        let module = Supervisor::default()
5260            .supervise_configured_with_health(
5261                configured.module_spec(),
5262                true,
5263                configured.health,
5264                configured.drain_timeout_ms,
5265                configured.restart,
5266            )
5267            .unwrap();
5268        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5269        status.store(503, std::sync::atomic::Ordering::SeqCst);
5270        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5271        assert!(module
5272            .status()
5273            .unwrap()
5274            .health
5275            .detail
5276            .unwrap()
5277            .contains("scratch failure"));
5278        status.store(200, std::sync::atomic::Ordering::SeqCst);
5279        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5280        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5281        let before = module.status().unwrap();
5282        let (spec, mut health) = module.configuration().unwrap();
5283        health.http = None;
5284        module
5285            .update_configuration(spec.clone(), health.clone(), Some(10))
5286            .await
5287            .unwrap();
5288        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5289        health.http = Some(url);
5290        module
5291            .update_configuration(spec, health, Some(10))
5292            .await
5293            .unwrap();
5294        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5295        assert_eq!(
5296            module.status().unwrap().pid,
5297            before.pid,
5298            "changing a probe must apply live, not restart its process"
5299        );
5300        let (spec, mut health) = module.configuration().unwrap();
5301        health.failure_threshold = 2;
5302        module
5303            .update_configuration(spec, health, Some(10))
5304            .await
5305            .unwrap();
5306        status.store(503, std::sync::atomic::Ordering::SeqCst);
5307        timeout(Duration::from_secs(5), async {
5308            while module.status().unwrap().spawn_generation == before.spawn_generation {
5309                sleep(Duration::from_millis(5)).await;
5310            }
5311        })
5312        .await
5313        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5314        module.drain().await.unwrap();
5315        server.abort();
5316    }
5317
5318    #[cfg(unix)]
5319    #[tokio::test]
5320    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5321        let dir = subc_test_support::TestTempDir::new("http-refused");
5322        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5323        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5324        drop(unused);
5325        let configured = http_fixture(&dir, &url, 2);
5326        let module = Supervisor::default()
5327            .supervise_configured_with_health(
5328                configured.module_spec(),
5329                true,
5330                configured.health,
5331                configured.drain_timeout_ms,
5332                configured.restart,
5333            )
5334            .unwrap();
5335        let before = module.status().unwrap().spawn_generation;
5336        timeout(Duration::from_secs(5), async {
5337            loop {
5338                let status = module.status().unwrap();
5339                if status.spawn_generation > before {
5340                    assert!(status.lifetime_restarts > 0);
5341                    break;
5342                }
5343                sleep(Duration::from_millis(5)).await;
5344            }
5345        })
5346        .await
5347        .expect("sustained HTTP refusal must trigger the health restart policy");
5348        module.drain().await.unwrap();
5349    }
5350
5351    #[cfg(unix)]
5352    #[tokio::test]
5353    async fn http_health_timeout_honours_deadline() {
5354        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5355        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5356        let server = tokio::spawn(async move {
5357            let _held = listener.accept().await.unwrap();
5358            std::future::pending::<()>().await;
5359        });
5360        let error = timeout(
5361            Duration::from_secs(1),
5362            probe_http_health(&url, Duration::from_millis(10)),
5363        )
5364        .await
5365        .expect("the probe must enforce its own deadline")
5366        .unwrap_err();
5367        server.abort();
5368        assert!(error.to_string().contains("timed out"));
5369    }
5370
5371    #[cfg(unix)]
5372    #[tokio::test]
5373    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5374        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5375        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5376        let url = format!(
5377            "http://localhost:{}/healthz",
5378            listener.local_addr().unwrap().port()
5379        );
5380        let server = tokio::spawn(async move {
5381            let (mut stream, _) = listener.accept().await.unwrap();
5382            let mut request = [0u8; 2048];
5383            assert!(stream.read(&mut request).await.unwrap() > 0);
5384            stream
5385                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5386                .await
5387                .unwrap();
5388        });
5389        assert_eq!(
5390            probe_http_health(&url, Duration::from_secs(1))
5391                .await
5392                .unwrap()
5393                .status,
5394            HealthStatus::Ok
5395        );
5396        server.await.unwrap();
5397    }
5398
5399    #[cfg(unix)]
5400    #[tokio::test]
5401    async fn http_health_timeout_keeps_partial_status_and_body() {
5402        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5403        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5404        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5405        let server = tokio::spawn(async move {
5406            let (mut stream, _) = listener.accept().await.unwrap();
5407            let mut request = [0u8; 2048];
5408            assert!(stream.read(&mut request).await.unwrap() > 0);
5409            stream
5410                .write_all(
5411                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5412                )
5413                .await
5414                .unwrap();
5415            std::future::pending::<()>().await;
5416        });
5417        let error = probe_http_health(&url, Duration::from_secs(1))
5418            .await
5419            .unwrap_err()
5420            .to_string();
5421        server.abort();
5422        assert!(
5423            error.contains("timed out")
5424                && error.contains("503 Unavailable")
5425                && error.contains("partial diagnostic"),
5426            "{error}"
5427        );
5428    }
5429
5430    #[cfg(unix)]
5431    #[tokio::test]
5432    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5433        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5434        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5435        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5436        let server = tokio::spawn(async move {
5437            let (mut stream, _) = listener.accept().await.unwrap();
5438            let mut request = [0u8; 2048];
5439            assert!(stream.read(&mut request).await.unwrap() > 0);
5440            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5441            let response = format!(
5442                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5443                body.len()
5444            );
5445            stream.write_all(response.as_bytes()).await.unwrap();
5446        });
5447        let error = probe_http_health(&url, Duration::from_secs(1))
5448            .await
5449            .unwrap_err()
5450            .to_string();
5451        server.await.unwrap();
5452        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5453        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5454        assert!(!error.contains("not-in-diagnostic"));
5455    }
5456
5457    #[cfg(unix)]
5458    #[tokio::test]
5459    async fn http_health_real_nats_server_monitoring() {
5460        if std::process::Command::new("nats-server")
5461            .arg("--version")
5462            .env("XDG_DATA_HOME", std::env::temp_dir())
5463            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5464            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5465            .output()
5466            .is_err()
5467        {
5468            eprintln!(
5469                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5470            );
5471            return;
5472        }
5473        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5474        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5475        let port = monitor.local_addr().unwrap().port();
5476        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5477        let client_port = client.local_addr().unwrap().port();
5478        let config = dir.join("server.conf");
5479        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5480        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5481        configured.program = PathBuf::from("nats-server");
5482        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5483        drop(monitor);
5484        drop(client);
5485        let module = Supervisor::default()
5486            .supervise_configured_with_health(
5487                configured.module_spec(),
5488                true,
5489                configured.health,
5490                configured.drain_timeout_ms,
5491                configured.restart,
5492            )
5493            .unwrap();
5494        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5495        module.drain().await.unwrap();
5496    }
5497
5498    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5499    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5500        let supervisor = Supervisor::default();
5501        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5502        let initial = ModuleSpec {
5503            module_id: "rescan-preserves-spawn-facts".to_string(),
5504            program: PathBuf::from("/spawned/module"),
5505            args: Vec::new(),
5506            env: Vec::new(),
5507            reserved: false,
5508            reserved_prefixes: Vec::new(),
5509            protocol: ModuleProtocol::Subc,
5510            overlap: Default::default(),
5511        };
5512        let module = supervisor.supervised_module(
5513            initial.clone(),
5514            supervisor.runtime_config(),
5515            snapshot,
5516            None,
5517        );
5518        let before = module.status().unwrap();
5519        let mut replacement = initial;
5520        replacement.program = PathBuf::from("/rescanned/replacement-module");
5521
5522        module
5523            .update_configuration(replacement, HealthConfig::default(), None)
5524            .await
5525            .unwrap();
5526
5527        let after = module.status().unwrap();
5528        assert_eq!(after.pid, before.pid);
5529        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5530        assert_eq!(after.spawned_from, before.spawned_from);
5531        drop(module);
5532    }
5533}
5534
5535fn unix_ms_now() -> u64 {
5536    SystemTime::now()
5537        .duration_since(UNIX_EPOCH)
5538        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5539        .unwrap_or(0)
5540}
5541
5542async fn supervise_loop(
5543    mut spec: ModuleSpec,
5544    mut runtime: SupervisorRuntimeConfig,
5545    registry: Arc<Registry>,
5546    process_liveness: Arc<SupervisorProcessLiveness>,
5547    snapshot: SharedSnapshot,
5548    mut child: Option<SupervisedChild>,
5549    mut commands: mpsc::Receiver<SupervisorCommand>,
5550) {
5551    let mut health_probe = HealthProbeRuntime::default();
5552    // All restart backoffs run here, including health and operator requests.
5553    // While one is pending the loop serves commands, so disable or drain can
5554    // cancel the replacement without spawning a process just to stop it.
5555    let mut pending_respawn: Option<PendingRespawn> = None;
5556    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5557    // before anything else so a stop that interrupted a swap runs at once.
5558    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5559    loop {
5560        if let Some(scheduled) = runtime
5561            .scheduled_respawn
5562            .lock()
5563            .unwrap_or_else(|p| p.into_inner())
5564            .take()
5565        {
5566            pending_respawn = Some(scheduled);
5567        }
5568        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5569            pending_respawn = None;
5570            cancel_deferred_reload(
5571                &runtime,
5572                &spec.module_id,
5573                "respawn cancelled by a supervisor command",
5574            );
5575        }
5576        if child.is_none() && pending_respawn.is_none() {
5577            cancel_deferred_reload(
5578                &runtime,
5579                &spec.module_id,
5580                "respawn cancelled before a replacement was spawned",
5581            );
5582            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5583                state.respawn_pending = false;
5584                state.coalesced_restart_pending = false;
5585                if matches!(
5586                    state.state,
5587                    ModuleState::Restarting
5588                        | ModuleState::Starting
5589                        | ModuleState::Draining
5590                        | ModuleState::Unresponsive
5591                ) {
5592                    error!(module_id = %spec.module_id, state = ?state.state,
5593                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5594                    state.state = ModuleState::Failed;
5595                    clear_current_process_facts(state);
5596                }
5597            });
5598        }
5599        if let Some(command) = requeued.pop_front() {
5600            if !handle_supervisor_command(
5601                command,
5602                &mut spec,
5603                &mut runtime,
5604                &registry,
5605                &process_liveness,
5606                &snapshot,
5607                &mut child,
5608                &mut commands,
5609                &mut requeued,
5610            )
5611            .await
5612            {
5613                return;
5614            }
5615            if child.is_some() || !respawn_still_pending(&snapshot) {
5616                pending_respawn = None;
5617                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5618                    state.respawn_pending = false
5619                });
5620            }
5621            continue;
5622        }
5623        if child.is_some() {
5624            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
5625            let probe_sleep = sleep(health_probe.wake_after());
5626            tokio::pin!(probe_sleep);
5627            let active_child = child.as_mut().expect("child checked above");
5628            tokio::select! {
5629                wait_result = active_child.wait() => {
5630                    // Every arm below that gives up on the CHILD must keep the
5631                    // supervision task itself alive (child = None, loop
5632                    // continues into command-serving mode). Returning here
5633                    // closes the command channel, which makes the module
5634                    // permanently unrestartable in-band: a clean child exit
5635                    // of an enabled module once wedged the fleet this way
5636                    // ('supervisor command channel is closed') and required a
5637                    // full daemon restart to recover.
5638                    let exit_report = match wait_result {
5639                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
5640                        Err(err) => {
5641                            active_child.drain_stderr(&spec.module_id).await;
5642                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
5643                            // Every other exit path (on_child_exit's Clean/Crash arms,
5644                            // the reload-registration-failure path) records a terminal
5645                            // before moving on. Without one here, a module whose wait()
5646                            // itself errored (e.g. already reaped) leaves no terminal
5647                            // record at all -- an empty ring reads as "nothing died".
5648                            record_wait_error_terminal(
5649                                &spec.module_id,
5650                                &runtime.terminal_ring,
5651                                &runtime.spawn_events,
5652                            );
5653                            untrack_if_registration_released(
5654                                &process_liveness,
5655                                &registry,
5656                                &spec.module_id,
5657                                &snapshot,
5658                            );
5659                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
5660                            child = None;
5661                            continue;
5662                        }
5663                    };
5664                    active_child.drain_stderr(&spec.module_id).await;
5665
5666                    let next = on_child_exit(
5667                        &spec,
5668                        runtime.restart_policy,
5669                        &registry,
5670                        &snapshot,
5671                        &runtime.terminal_ring,
5672                        &runtime.spawn_events,
5673                        &runtime.child_roster,
5674                        exit_report,
5675                    ).await;
5676                    // The exit is recorded, so a daemon shutdown may stop
5677                    // waiting for this child (see `SupervisedChild::wait`).
5678                    active_child.release_roster();
5679                    match next {
5680                        NextAction::Stop { registration_released } => {
5681                            if registration_released {
5682                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
5683                            }
5684                            child = None;
5685                        }
5686                        NextAction::Restart { schedule } => {
5687                            let delay = schedule.map_or(
5688                                runtime.restart_policy.delay_for_restart(0),
5689                                |schedule| schedule.delay,
5690                            );
5691                            if let Some(schedule) = schedule {
5692                                log_crash_respawn(&spec.module_id, schedule);
5693                            }
5694                            // The exited child is fully recorded at this point,
5695                            // so release it and count the backoff down in the
5696                            // command-serving branch below rather than sleeping
5697                            // here: commands cannot be received from inside this
5698                            // select arm, and an operator disable or drain that
5699                            // arrives during the backoff must cancel the pending
5700                            // respawn instead of waiting for it to spawn first.
5701                            child = None;
5702                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
5703                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
5704                        }
5705                    }
5706                }
5707                command = commands.recv() => {
5708                    let Some(command) = command else {
5709                        return;
5710                    };
5711                    if !handle_supervisor_command(
5712                        command,
5713                        &mut spec,
5714                        &mut runtime,
5715                        &registry,
5716                        &process_liveness,
5717                        &snapshot,
5718                        &mut child,
5719                        &mut commands,
5720                        &mut requeued,
5721                    ).await {
5722                        return;
5723                    }
5724                }
5725                _ = &mut probe_sleep => {
5726                    if health_probe.due() {
5727                        run_health_probe_cycle(
5728                            &spec,
5729                            &runtime,
5730                            &registry,
5731                            &process_liveness,
5732                            &snapshot,
5733                            &mut child,
5734                        ).await;
5735                        if child.is_some() {
5736                            health_probe.schedule_next(&spec, runtime.health.cadence);
5737                        }
5738                    }
5739                }
5740            }
5741        } else if let Some(pending) = pending_respawn {
5742            tokio::select! {
5743                _ = sleep_until(pending.deadline) => {
5744                    pending_respawn = None;
5745                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
5746                    // A command handled below while the backoff elapsed may
5747                    // have stopped the module; never respawn past an operator's
5748                    // disable or drain.
5749                    if !respawn_still_pending(&snapshot) {
5750                        continue;
5751                    }
5752                    // The daemon began shutting down during the backoff: the
5753                    // spawn would be refused anyway, and refusing it here
5754                    // leaves the module stopped instead of reporting a
5755                    // failed restart.
5756                    if runtime.child_roster.is_closed() {
5757                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5758                            state.state = ModuleState::Stopped;
5759                        });
5760                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
5761                        continue;
5762                    }
5763                    if let Err(err) = release_dead_registration(
5764                        &registry,
5765                        runtime.forwarding.as_deref(),
5766                        &snapshot,
5767                        &spec.module_id,
5768                    ).await {
5769                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
5770                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
5771                        continue;
5772                    }
5773
5774                    if matches!(pending.kind, RespawnKind::Reload) {
5775                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
5776                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
5777                        if let Some(reply) = reply { let _ = reply.send(result); }
5778                        continue;
5779                    }
5780                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
5781                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
5782                        Ok(next_child) => {
5783                            child = Some(next_child);
5784                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
5785                        }
5786                        Err(err) => {
5787                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
5788                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
5789                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
5790                        }
5791                    }
5792                }
5793                command = commands.recv() => {
5794                    let Some(command) = command else {
5795                        return;
5796                    };
5797                    if !handle_supervisor_command(
5798                        command,
5799                        &mut spec,
5800                        &mut runtime,
5801                        &registry,
5802                        &process_liveness,
5803                        &snapshot,
5804                        &mut child,
5805                        &mut commands,
5806                        &mut requeued,
5807                    ).await {
5808                        return;
5809                    }
5810                    // Reconcile the pending respawn with what the command did:
5811                    // a start may already have spawned a fresh child,
5812                    // while a disable or drain moved the snapshot out of the
5813                    // state the respawn was counting down from.
5814                    if child.is_some() || !respawn_still_pending(&snapshot) {
5815                        pending_respawn = None;
5816                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
5817                    }
5818                }
5819            }
5820        } else {
5821            let Some(command) = commands.recv().await else {
5822                return;
5823            };
5824            if !handle_supervisor_command(
5825                command,
5826                &mut spec,
5827                &mut runtime,
5828                &registry,
5829                &process_liveness,
5830                &snapshot,
5831                &mut child,
5832                &mut commands,
5833                &mut requeued,
5834            )
5835            .await
5836            {
5837                return;
5838            }
5839        }
5840    }
5841}
5842
5843fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
5844    info!(
5845        module_id,
5846        restart_in_window = schedule.restart_in_window,
5847        delay_ms = schedule.delay.as_millis() as u64,
5848        "respawning after crash"
5849    );
5850}
5851
5852/// Whether the respawn a backoff was counting down to is still wanted. A
5853/// disable or drain handled while the backoff elapsed moves the snapshot out
5854/// of `Restarting`, and the operator's stop must win over the pending respawn,
5855/// so every sleep-then-spawn path re-validates against the live snapshot
5856/// instead of assuming the state it left behind still holds.
5857fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
5858    matches!(
5859        lock_snapshot(snapshot),
5860        Ok(state) if state.enabled && state.state == ModuleState::Restarting
5861    )
5862}
5863
5864enum NextAction {
5865    Stop {
5866        registration_released: bool,
5867    },
5868    Restart {
5869        schedule: Option<CrashRestartSchedule>,
5870    },
5871}
5872
5873#[allow(clippy::too_many_arguments)]
5874async fn handle_supervisor_command(
5875    command: SupervisorCommand,
5876    spec: &mut ModuleSpec,
5877    runtime: &mut SupervisorRuntimeConfig,
5878    registry: &Arc<Registry>,
5879    process_liveness: &SupervisorProcessLiveness,
5880    snapshot: &SharedSnapshot,
5881    child: &mut Option<SupervisedChild>,
5882    commands: &mut mpsc::Receiver<SupervisorCommand>,
5883    requeued: &mut VecDeque<SupervisorCommand>,
5884) -> bool {
5885    match command {
5886        SupervisorCommand::Drain { reply } => {
5887            // A plain stop runs no forwarding drain, so nothing reaches the
5888            // module over its connection before the wait: ask by signal.
5889            let result = drain_optional_child(
5890                &spec.module_id,
5891                spec.protocol,
5892                StopNotice::NotSent,
5893                registry,
5894                runtime.forwarding.as_deref(),
5895                snapshot,
5896                &runtime.terminal_ring,
5897                &runtime.spawn_events,
5898                child,
5899                runtime.drain_timeout,
5900                ModuleState::Stopped,
5901                None,
5902            )
5903            .await;
5904            let registration_released = result.is_ok();
5905            let _ = reply.send(result);
5906            if registration_released {
5907                process_liveness.untrack_if_current(&spec.module_id, snapshot);
5908            }
5909            false
5910        }
5911        SupervisorCommand::Retire { reply } => {
5912            let result = async {
5913                let stop_notice = begin_forwarding_drain_if_configured(
5914                    spec,
5915                    runtime,
5916                    registry,
5917                    snapshot,
5918                    None,
5919                    RouteCloseReason::Disable,
5920                )
5921                .await?;
5922                drain_optional_child(
5923                    &spec.module_id,
5924                    spec.protocol,
5925                    stop_notice,
5926                    registry,
5927                    runtime.forwarding.as_deref(),
5928                    snapshot,
5929                    &runtime.terminal_ring,
5930                    &runtime.spawn_events,
5931                    child,
5932                    runtime.drain_timeout,
5933                    ModuleState::Stopped,
5934                    None,
5935                )
5936                .await
5937            }
5938            .await;
5939            let registration_released = result.is_ok();
5940            let _ = reply.send(result);
5941            if registration_released {
5942                process_liveness.untrack_if_current(&spec.module_id, snapshot);
5943            }
5944            false
5945        }
5946        SupervisorCommand::Restart {
5947            drain_timeout_ms,
5948            received_at_generation,
5949            queued_at,
5950            reply,
5951        } => {
5952            // Without this line a restart that waited in the queue (behind a
5953            // health probe cycle or another command) was invisible: the log
5954            // showed only the drain timing out, minutes after the operator's call.
5955            info!(
5956                module_id = %spec.module_id,
5957                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
5958                "restart command dequeued"
5959            );
5960            // ACK AT INITIATION, not completion. The blocking form deadlocked any
5961            // caller whose own request lane rides the module being restarted: the
5962            // caller's in-flight request keeps the drain from quiescing, the drain
5963            // keeps the restart from completing, and the completion keeps the reply
5964            // from releasing the caller — so the drain always timed out and cut the
5965            // initiator with a GOODBYE, even on a healthy module. Replying once the
5966            // restart is validated lets a self-lane caller settle, which is exactly
5967            // what makes the drain succeed. Completion is observable via
5968            // supervisor.list / module status; a post-ack failure lands the module
5969            // in a visible terminal state below rather than in a reply nobody can
5970            // receive.
5971            let validation = match lock_snapshot(snapshot) {
5972                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
5973                    module_id: spec.module_id.clone(),
5974                }),
5975                Ok(_) => Ok(()),
5976                Err(err) => Err(err),
5977            };
5978            let initiated = validation.is_ok();
5979            let _ = reply.send(validation);
5980            // A restart asks for a fresh process. Commands run one at a time,
5981            // so a restart queued behind another restart (two operator calls
5982            // in quick succession) is dequeued the moment the first one has
5983            // spawned its replacement -- before that process has sent HELLO.
5984            // Running it would drain and kill the process the first restart
5985            // just produced, which is the opposite of what both callers asked
5986            // for. If a process spawned after this request was received is
5987            // still supervised, the request is already satisfied. Not when the
5988            // configuration changed since that spawn: then the newer process
5989            // predates the spec this restart may exist to apply.
5990            let satisfied_by_generation = if initiated && child.is_some() {
5991                lock_snapshot(snapshot).ok().and_then(|state| {
5992                    (state.spawn_generation > received_at_generation
5993                        && !state.configuration_updated_since_spawn)
5994                        .then_some(state.spawn_generation)
5995                })
5996            } else {
5997                None
5998            };
5999            let satisfied_by_pending = initiated
6000                && child.is_none()
6001                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6002                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6003                    if pending {
6004                        state.coalesced_restart_pending = true;
6005                    }
6006                    pending
6007                });
6008            if satisfied_by_pending {
6009                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6010            } else if let Some(generation) = satisfied_by_generation {
6011                info!(
6012                    module_id = %spec.module_id,
6013                    received_at_generation,
6014                    "restart already satisfied by generation {generation}; not restarting again"
6015                );
6016            } else if initiated {
6017                // Precedence: this restart's operator override, else the module's
6018                // configured budget (already resolved into the runtime).
6019                let drain_timeout = drain_timeout_ms
6020                    .map(Duration::from_millis)
6021                    .unwrap_or(runtime.drain_timeout);
6022                if let Err(err) = restart_child(
6023                    spec,
6024                    runtime,
6025                    registry,
6026                    process_liveness,
6027                    snapshot,
6028                    child,
6029                    drain_timeout,
6030                )
6031                .await
6032                {
6033                    warn!(
6034                        module_id = %spec.module_id,
6035                        error = %err,
6036                        "operator restart failed after initiation ack; module state carries the outcome"
6037                    );
6038                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6039                        state.state = ModuleState::Failed;
6040                        clear_current_process_facts(state);
6041                    });
6042                }
6043            }
6044            true
6045        }
6046        SupervisorCommand::Reload { reply } => {
6047            let result =
6048                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6049            if result.is_ok()
6050                && runtime
6051                    .scheduled_respawn
6052                    .lock()
6053                    .unwrap_or_else(|p| p.into_inner())
6054                    .as_ref()
6055                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6056            {
6057                *runtime
6058                    .deferred_reload_reply
6059                    .lock()
6060                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6061            } else {
6062                let _ = reply.send(result);
6063            }
6064            true
6065        }
6066        SupervisorCommand::SetEnabled { enabled, reply } => {
6067            let result = set_child_enabled(
6068                spec,
6069                runtime,
6070                registry,
6071                process_liveness,
6072                snapshot,
6073                child,
6074                enabled,
6075            )
6076            .await;
6077            let _ = reply.send(result);
6078            true
6079        }
6080        SupervisorCommand::UpdateConfiguration {
6081            spec: next_spec,
6082            health,
6083            drain_timeout_ms,
6084            reply,
6085        } => {
6086            if let Some(handle) = &runtime.supervisor_handle {
6087                handle.apply_identity_configuration(&next_spec);
6088            }
6089            *spec = next_spec;
6090            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6091                state.configuration_updated_since_spawn = true;
6092            });
6093            let health_changed = runtime.health != health;
6094            runtime.health = health;
6095            // Reset the cadence and old endpoint's failure streak on a live
6096            // health-policy change rather than waiting for its old deadline.
6097            if health_changed {
6098                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6099                    state.health = ModuleHealthStatus::default();
6100                });
6101            }
6102            runtime.drain_timeout = drain_timeout_ms
6103                .map(Duration::from_millis)
6104                .unwrap_or(runtime.default_drain_timeout);
6105            *runtime
6106                .effective_drain_timeout
6107                .lock()
6108                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6109            let _ = reply.send(());
6110            true
6111        }
6112        SupervisorCommand::Swap {
6113            ready_timeout,
6114            reply,
6115        } => {
6116            let end = swap::run_swap(
6117                spec,
6118                runtime,
6119                registry,
6120                process_liveness,
6121                snapshot,
6122                child,
6123                commands,
6124                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6125                reply,
6126            )
6127            .await;
6128            requeued.extend(end.requeue);
6129            true
6130        }
6131    }
6132}
6133
6134async fn restart_child(
6135    spec: &ModuleSpec,
6136    runtime: &SupervisorRuntimeConfig,
6137    registry: &Registry,
6138    process_liveness: &SupervisorProcessLiveness,
6139    snapshot: &SharedSnapshot,
6140    child: &mut Option<SupervisedChild>,
6141    drain_timeout: Duration,
6142) -> Result<(), SuperviseError> {
6143    // Restart cycles a running module; it must not silently start a disabled one.
6144    if !lock_snapshot(snapshot)?.enabled {
6145        return Err(SuperviseError::Disabled {
6146            module_id: spec.module_id.clone(),
6147        });
6148    }
6149    let stop_notice = begin_forwarding_drain_with_timeout(
6150        spec,
6151        runtime,
6152        registry,
6153        snapshot,
6154        None,
6155        RouteCloseReason::Restart,
6156        drain_timeout,
6157    )
6158    .await?;
6159
6160    if child.is_some() {
6161        drain_optional_child(
6162            &spec.module_id,
6163            spec.protocol,
6164            stop_notice,
6165            registry,
6166            runtime.forwarding.as_deref(),
6167            snapshot,
6168            &runtime.terminal_ring,
6169            &runtime.spawn_events,
6170            child,
6171            drain_timeout,
6172            ModuleState::Restarting,
6173            Some(true),
6174        )
6175        .await?;
6176    } else {
6177        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6178            state.enabled = true;
6179            state.state = ModuleState::Restarting;
6180            clear_current_process_facts(state);
6181        })?;
6182        release_dead_registration(
6183            registry,
6184            runtime.forwarding.as_deref(),
6185            snapshot,
6186            &spec.module_id,
6187        )
6188        .await?;
6189    }
6190
6191    reset_restart_count(snapshot, &spec.module_id)?;
6192    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6193    schedule_respawn(
6194        runtime,
6195        snapshot,
6196        &spec.module_id,
6197        runtime.restart_policy.backoff,
6198        RespawnKind::Spawn,
6199    )
6200}
6201
6202async fn reload_child(
6203    spec: &ModuleSpec,
6204    runtime: &SupervisorRuntimeConfig,
6205    registry: &Registry,
6206    process_liveness: &SupervisorProcessLiveness,
6207    snapshot: &SharedSnapshot,
6208    child: &mut Option<SupervisedChild>,
6209) -> Result<(), SuperviseError> {
6210    // Reload cycles a running module; it must not silently start a disabled one.
6211    if !lock_snapshot(snapshot)?.enabled {
6212        return Err(SuperviseError::Disabled {
6213            module_id: spec.module_id.clone(),
6214        });
6215    }
6216    let stop_notice = begin_forwarding_drain(
6217        spec,
6218        runtime,
6219        registry,
6220        snapshot,
6221        Some(true),
6222        RouteCloseReason::Reload,
6223    )
6224    .await?;
6225
6226    if child.is_some() {
6227        drain_optional_child(
6228            &spec.module_id,
6229            spec.protocol,
6230            stop_notice,
6231            registry,
6232            runtime.forwarding.as_deref(),
6233            snapshot,
6234            &runtime.terminal_ring,
6235            &runtime.spawn_events,
6236            child,
6237            runtime.drain_timeout,
6238            ModuleState::Restarting,
6239            Some(true),
6240        )
6241        .await?;
6242    } else {
6243        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6244            state.enabled = true;
6245            state.state = ModuleState::Restarting;
6246            clear_current_process_facts(state);
6247        })?;
6248        release_dead_registration(
6249            registry,
6250            runtime.forwarding.as_deref(),
6251            snapshot,
6252            &spec.module_id,
6253        )
6254        .await?;
6255    }
6256
6257    reset_restart_count(snapshot, &spec.module_id)?;
6258    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6259    schedule_respawn(
6260        runtime,
6261        snapshot,
6262        &spec.module_id,
6263        runtime.restart_policy.backoff,
6264        RespawnKind::Reload,
6265    )
6266}
6267
6268async fn finish_reload_child(
6269    spec: &ModuleSpec,
6270    runtime: &SupervisorRuntimeConfig,
6271    registry: &Registry,
6272    process_liveness: &SupervisorProcessLiveness,
6273    snapshot: &SharedSnapshot,
6274    child: &mut Option<SupervisedChild>,
6275) -> Result<(), SuperviseError> {
6276    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6277    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6278        Ok(next_child) => next_child,
6279        Err(err) => {
6280            return handle_reload_spawn_failure(
6281                spec,
6282                runtime,
6283                process_liveness,
6284                snapshot,
6285                child,
6286                format!("new child failed to spawn: {err}"),
6287            )
6288            .await;
6289        }
6290    };
6291    *child = Some(next_child);
6292
6293    let wait_outcome = {
6294        let active_child = child.as_mut().expect("new reload child was just stored");
6295        wait_for_registration_after_reload(
6296            registry,
6297            &spec.module_id,
6298            snapshot,
6299            active_child,
6300            REGISTRY_RELEASE_TIMEOUT,
6301        )
6302        .await?
6303    };
6304
6305    match wait_outcome {
6306        RegistrationWaitOutcome::Registered => {
6307            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6308            Ok(())
6309        }
6310        RegistrationWaitOutcome::Exited(exit_report) => {
6311            if let Some(active_child) = child.as_mut() {
6312                active_child.drain_stderr(&spec.module_id).await;
6313            }
6314            // Keep the reaped child's roster guard until its terminal is written.
6315            // Shutdown waits on that guard, not on the child Option used for respawn.
6316            let mut exited_child = child.take().expect("exited reload child is still stored");
6317            #[cfg(test)]
6318            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6319                gate.reached.notify_one();
6320                gate.resume.notified().await;
6321            }
6322            let result = handle_reload_child_registration_failure(
6323                spec,
6324                runtime,
6325                registry,
6326                process_liveness,
6327                snapshot,
6328                child,
6329                ReloadRegistrationFailure {
6330                    exit_report: registration_failure_exit_report(exit_report),
6331                    reason: "new child exited before registering".to_string(),
6332                },
6333            )
6334            .await;
6335            exited_child.release_roster();
6336            result
6337        }
6338        RegistrationWaitOutcome::TimedOut => {
6339            let mut timed_out_child = child
6340                .take()
6341                .expect("timed-out reload child is still running");
6342            timed_out_child
6343                .start_kill()
6344                .map_err(|source| SuperviseError::Kill {
6345                    module_id: spec.module_id.clone(),
6346                    source,
6347                })?;
6348            let status = timed_out_child
6349                .wait()
6350                .await
6351                .map_err(|source| SuperviseError::Wait {
6352                    module_id: spec.module_id.clone(),
6353                    source,
6354                })?;
6355            timed_out_child.drain_stderr(&spec.module_id).await;
6356            handle_reload_child_registration_failure(
6357                spec,
6358                runtime,
6359                registry,
6360                process_liveness,
6361                snapshot,
6362                child,
6363                ReloadRegistrationFailure {
6364                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6365                        snapshot,
6366                        &timed_out_child,
6367                        &status,
6368                    )),
6369                    reason: format!(
6370                        "new child did not register within {:?}",
6371                        REGISTRY_RELEASE_TIMEOUT
6372                    ),
6373                },
6374            )
6375            .await
6376        }
6377    }
6378}
6379
6380async fn set_child_enabled(
6381    spec: &ModuleSpec,
6382    runtime: &SupervisorRuntimeConfig,
6383    registry: &Registry,
6384    process_liveness: &SupervisorProcessLiveness,
6385    snapshot: &SharedSnapshot,
6386    child: &mut Option<SupervisedChild>,
6387    enabled: bool,
6388) -> Result<bool, SuperviseError> {
6389    let (current_enabled, current_state, respawn_pending) = {
6390        let state = lock_snapshot(snapshot)?;
6391        (state.enabled, state.state, state.respawn_pending)
6392    };
6393    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6394    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6395    // clean (Stopped) has no live process and no other in-band recovery — the
6396    // operator's start is the explicit recovery act and resets the budget. Without
6397    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6398    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6399    // the one providing every agent's shell.
6400    let revive_terminal = enabled
6401        && current_enabled
6402        && child.is_none()
6403        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6404            || (current_state == ModuleState::Restarting && !respawn_pending));
6405    if current_enabled == enabled && !revive_terminal {
6406        return Ok(false);
6407    }
6408
6409    if enabled {
6410        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6411            state.enabled = true;
6412            state.state = ModuleState::Starting;
6413            clear_current_process_facts(state);
6414        })?;
6415        #[cfg(test)]
6416        if runtime.test_seed_stale_facts_before_enable_spawn {
6417            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6418                state.process_alive = true;
6419                state.pid = Some(41);
6420                state.spawned_at_ms = Some(42);
6421                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6422                state.spawned_file_identity = Some(SpawnedFileIdentity {
6423                    device: 43,
6424                    inode: 44,
6425                });
6426            })?;
6427        }
6428        release_dead_registration(
6429            registry,
6430            runtime.forwarding.as_deref(),
6431            snapshot,
6432            &spec.module_id,
6433        )
6434        .await?;
6435        reset_restart_count(snapshot, &spec.module_id)?;
6436        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6437        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6438            Ok(next_child) => next_child,
6439            Err(err) => {
6440                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6441                    state.state = ModuleState::Failed;
6442                    clear_current_process_facts(state);
6443                }) {
6444                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6445                }
6446                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6447                return Err(err);
6448            }
6449        };
6450        *child = Some(next_child);
6451        debug!(module_id = %spec.module_id, "supervised module enabled");
6452        Ok(true)
6453    } else {
6454        let stop_notice = begin_forwarding_drain_if_configured(
6455            spec,
6456            runtime,
6457            registry,
6458            snapshot,
6459            Some(false),
6460            RouteCloseReason::Disable,
6461        )
6462        .await?;
6463        drain_optional_child(
6464            &spec.module_id,
6465            spec.protocol,
6466            stop_notice,
6467            registry,
6468            runtime.forwarding.as_deref(),
6469            snapshot,
6470            &runtime.terminal_ring,
6471            &runtime.spawn_events,
6472            child,
6473            runtime.drain_timeout,
6474            ModuleState::Disabled,
6475            Some(false),
6476        )
6477        .await?;
6478        debug!(module_id = %spec.module_id, "supervised module disabled");
6479        Ok(true)
6480    }
6481}
6482
6483#[allow(clippy::too_many_arguments)]
6484async fn on_child_exit(
6485    spec: &ModuleSpec,
6486    policy: RestartPolicy,
6487    registry: &Registry,
6488    snapshot: &SharedSnapshot,
6489    terminal_ring: &Arc<Mutex<TerminalRing>>,
6490    spawn_events: &SpawnEventFeed,
6491    roster: &ChildRoster,
6492    exit_report: ExitReport,
6493) -> NextAction {
6494    // Once the daemon has begun shutting down, no exit is a crash to recover
6495    // from: the module is exiting because the daemon is going away (EOF on its
6496    // connection, or a service manager signalling the whole cgroup). Record it
6497    // as such and never schedule a respawn, which would only start a process
6498    // for the shutdown to end again.
6499    if roster.is_closed() {
6500        return on_child_exit_during_daemon_shutdown(
6501            spec,
6502            registry,
6503            snapshot,
6504            terminal_ring,
6505            spawn_events,
6506            exit_report,
6507        )
6508        .await;
6509    }
6510    // Every stop the supervisor itself asks for (operator stop, disable,
6511    // restart, reload, swap, a health restart, a drain that runs out of budget)
6512    // takes the child out of the supervise loop and reaps it in
6513    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
6514    // that reaches this point was not requested by the daemon.
6515    //
6516    // For a subc-wire module a clean exit is still a stop: those modules are
6517    // written to re-raise SIGTERM, so a stray outside signal already reads as a
6518    // crash, and exiting 0 is a deliberate choice the module made. A
6519    // `protocol: "none"` module is a stock program we cannot change, and many
6520    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
6521    // stop would leave the module down for good after any stray signal, so it
6522    // goes through the crash path instead: it spends restart budget, respawns
6523    // with the crash backoff, and ends `failed` when the budget runs out.
6524    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6525        && running_protocol(spec, snapshot) == ModuleProtocol::None;
6526    match exit_report.kind {
6527        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6528            info!(
6529                module_id = %spec.module_id,
6530                exit_code = ?exit_report.code,
6531                exit_signal = ?exit_report.signal,
6532                "supervised module exited cleanly"
6533            );
6534            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6535                state.state = ModuleState::Stopped;
6536                clear_current_process_facts(state);
6537                state.last_exit = Some(exit_report.clone());
6538            }) {
6539                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6540            }
6541            record_terminal(
6542                &spec.module_id,
6543                terminal_ring,
6544                spawn_events,
6545                &exit_report,
6546                TerminalDisposition::Stopped,
6547            );
6548            let registration_released = match wait_for_registration_release(
6549                registry,
6550                &spec.module_id,
6551                REGISTRY_RELEASE_TIMEOUT,
6552            )
6553            .await
6554            {
6555                Ok(()) => true,
6556                Err(err) => {
6557                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6558                    false
6559                }
6560            };
6561            NextAction::Stop {
6562                registration_released,
6563            }
6564        }
6565        ExitKind::Clean | ExitKind::Crash => {
6566            if unrequested_clean_exit_of_protocol_none {
6567                warn!(
6568                    module_id = %spec.module_id,
6569                    exit_code = ?exit_report.code,
6570                    exit_signal = ?exit_report.signal,
6571                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
6572                );
6573            } else {
6574                warn!(
6575                    module_id = %spec.module_id,
6576                    exit_code = ?exit_report.code,
6577                    exit_signal = ?exit_report.signal,
6578                    "supervised module exited abnormally (crash)"
6579                );
6580            }
6581            let mut restart_schedule = None;
6582            let mut disposition = TerminalDisposition::Disabled;
6583            // Set only when the budget is what stopped the module, so the
6584            // terminal record says which limit was hit rather than leaving
6585            // `failed` to be read as "crashed once, badly".
6586            let mut disposition_detail = None;
6587            let now = Instant::now();
6588            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6589                clear_current_process_facts(state);
6590                state.last_exit = Some(exit_report.clone());
6591                if state.enabled {
6592                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
6593                        state.state = ModuleState::Restarting;
6594                        restart_schedule = Some(schedule);
6595                        disposition = TerminalDisposition::Restarting;
6596                    } else {
6597                        state.state = ModuleState::Failed;
6598                        disposition = TerminalDisposition::Failed;
6599                        disposition_detail = Some(policy.budget_exhausted_detail());
6600                    }
6601                } else {
6602                    state.state = ModuleState::Disabled;
6603                    disposition = TerminalDisposition::Disabled;
6604                }
6605            }) {
6606                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
6607                return NextAction::Stop {
6608                    registration_released: false,
6609                };
6610            }
6611            if disposition_detail.is_some() {
6612                // The window is in the message, not only in the fields: this line
6613                // is read in a scrollback where a bare `max_restarts=3` reads as a
6614                // lifetime cap and sends the operator looking for three crashes
6615                // that never happened together.
6616                error!(
6617                    module_id = %spec.module_id,
6618                    max_restarts = policy.max_restarts,
6619                    window_secs = policy.window.as_secs(),
6620                    "module stopped: {}",
6621                    policy.budget_exhausted_detail()
6622                );
6623            }
6624            record_terminal_with_detail(
6625                &spec.module_id,
6626                terminal_ring,
6627                spawn_events,
6628                &exit_report,
6629                disposition,
6630                disposition_detail,
6631            );
6632
6633            if let Some(schedule) = restart_schedule {
6634                NextAction::Restart {
6635                    schedule: Some(schedule),
6636                }
6637            } else {
6638                let registration_released = match wait_for_registration_release(
6639                    registry,
6640                    &spec.module_id,
6641                    REGISTRY_RELEASE_TIMEOUT,
6642                )
6643                .await
6644                {
6645                    Ok(()) => true,
6646                    Err(err) => {
6647                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
6648                        false
6649                    }
6650                };
6651                NextAction::Stop {
6652                    registration_released,
6653                }
6654            }
6655        }
6656        ExitKind::DeliberateSeverance => {
6657            warn!(
6658                module_id = %spec.module_id,
6659                exit_code = ?exit_report.code,
6660                exit_signal = ?exit_report.signal,
6661                "supervised module exited after deliberate connection severance"
6662            );
6663            let mut should_restart = false;
6664            let mut disposition = TerminalDisposition::Disabled;
6665            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6666                clear_current_process_facts(state);
6667                state.last_exit = Some(exit_report.clone());
6668                state.lifetime_restarts += 1;
6669                if state.enabled {
6670                    state.state = ModuleState::Restarting;
6671                    should_restart = true;
6672                    disposition = TerminalDisposition::Restarting;
6673                } else {
6674                    state.state = ModuleState::Disabled;
6675                }
6676            }) {
6677                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
6678                return NextAction::Stop {
6679                    registration_released: false,
6680                };
6681            }
6682            record_terminal(
6683                &spec.module_id,
6684                terminal_ring,
6685                spawn_events,
6686                &exit_report,
6687                disposition,
6688            );
6689
6690            if should_restart {
6691                NextAction::Restart { schedule: None }
6692            } else {
6693                let registration_released = match wait_for_registration_release(
6694                    registry,
6695                    &spec.module_id,
6696                    REGISTRY_RELEASE_TIMEOUT,
6697                )
6698                .await
6699                {
6700                    Ok(()) => true,
6701                    Err(err) => {
6702                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
6703                        false
6704                    }
6705                };
6706                NextAction::Stop {
6707                    registration_released,
6708                }
6709            }
6710        }
6711    }
6712}
6713
6714async fn on_child_exit_during_daemon_shutdown(
6715    spec: &ModuleSpec,
6716    registry: &Registry,
6717    snapshot: &SharedSnapshot,
6718    terminal_ring: &Arc<Mutex<TerminalRing>>,
6719    spawn_events: &SpawnEventFeed,
6720    exit_report: ExitReport,
6721) -> NextAction {
6722    info!(
6723        module_id = %spec.module_id,
6724        exit_code = ?exit_report.code,
6725        exit_signal = ?exit_report.signal,
6726        exit_kind = ?exit_report.kind,
6727        "supervised module exited during daemon shutdown; not restarting it"
6728    );
6729    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6730        state.state = ModuleState::Stopped;
6731        clear_current_process_facts(state);
6732        state.last_exit = Some(exit_report.clone());
6733    }) {
6734        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
6735    }
6736    record_terminal(
6737        &spec.module_id,
6738        terminal_ring,
6739        spawn_events,
6740        &exit_report,
6741        TerminalDisposition::DaemonShutdown,
6742    );
6743    let registration_released =
6744        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
6745            .await
6746            .is_ok();
6747    NextAction::Stop {
6748        registration_released,
6749    }
6750}
6751
6752fn record_wait_error_terminal(
6753    module_id: &str,
6754    terminal_ring: &Arc<Mutex<TerminalRing>>,
6755    spawn_events: &SpawnEventFeed,
6756) {
6757    record_terminal(
6758        module_id,
6759        terminal_ring,
6760        spawn_events,
6761        &wait_error_exit_report(),
6762        TerminalDisposition::Failed,
6763    );
6764}
6765
6766fn record_terminal(
6767    module_id: &str,
6768    terminal_ring: &Arc<Mutex<TerminalRing>>,
6769    spawn_events: &SpawnEventFeed,
6770    exit_report: &ExitReport,
6771    disposition: TerminalDisposition,
6772) {
6773    record_terminal_with_detail(
6774        module_id,
6775        terminal_ring,
6776        spawn_events,
6777        exit_report,
6778        disposition,
6779        None,
6780    );
6781}
6782
6783/// The ring lock is held only to capture the read (see
6784/// `TerminalJournal::capture_read`), so this module's exits keep recording
6785/// while the journal files are read. Blocking: it reads files.
6786fn durable_terminal_history_of(
6787    terminal_ring: &Mutex<TerminalRing>,
6788    module_id: &str,
6789) -> subc_control::TerminalHistory {
6790    let read = terminal_ring
6791        .lock()
6792        .unwrap_or_else(|p| p.into_inner())
6793        .capture_durable_history();
6794    read.read(module_id)
6795}
6796
6797fn record_terminal_with_detail(
6798    module_id: &str,
6799    terminal_ring: &Arc<Mutex<TerminalRing>>,
6800    spawn_events: &SpawnEventFeed,
6801    exit_report: &ExitReport,
6802    disposition: TerminalDisposition,
6803    disposition_detail: Option<String>,
6804) {
6805    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
6806    let record = TerminalRecord {
6807        exit_code: exit_report.code,
6808        exit_signal: exit_report.signal,
6809        at_ms: exit_report.at_ms,
6810        disposition,
6811        exit_kind: exit_report.kind.into(),
6812        disposition_detail,
6813    };
6814    terminal_ring
6815        .lock()
6816        .unwrap_or_else(|poisoned| poisoned.into_inner())
6817        .record_exit(module_id, record);
6818}
6819
6820fn untrack_if_registration_released(
6821    process_liveness: &SupervisorProcessLiveness,
6822    registry: &Registry,
6823    module_id: &str,
6824    snapshot: &SharedSnapshot,
6825) {
6826    match registry.get_module(module_id) {
6827        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
6828        Ok(Some(_)) => {}
6829        Err(err) => {
6830            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
6831        }
6832    }
6833}
6834
6835/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
6836/// then apply the module's configured entries minus daemon-private capture keys.
6837///
6838/// Separated from `spawn_child` only so it can be asserted without spawning a
6839/// process — a duplicate of this logic in a test would pass while the real one
6840/// drifted, which is the defect class this function exists to avoid.
6841/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
6842/// nonce. A `protocol: "none"` module gets neither, because it cannot use
6843/// either and the argument would stop a stock binary from starting at all.
6844/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
6845///
6846/// The plain-spawn form, kept for the tests that assert its plan; spawns go
6847/// through [`apply_wire_spawn_args_for_role`].
6848#[cfg(test)]
6849fn apply_wire_spawn_args(
6850    command: &mut Command,
6851    spec: &ModuleSpec,
6852    connection_file_path: Option<&std::path::Path>,
6853    handle: Option<&SupervisorHandle>,
6854) -> Result<Option<NonceHandoff>, SuperviseError> {
6855    apply_wire_spawn_args_for_role(
6856        command,
6857        spec,
6858        connection_file_path,
6859        handle,
6860        SpawnRole::Plain,
6861    )
6862}
6863
6864/// The read end of a spawn's launch-nonce pipe, prepared by
6865/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
6866/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
6867/// handoff and keeps only the environment copy.
6868#[cfg(unix)]
6869type NonceHandoff = subc_os::LaunchNonceHandoff;
6870#[cfg(not(unix))]
6871type NonceHandoff = std::convert::Infallible;
6872
6873/// Prepare wire identity for a plain spawn or a swap candidate.
6874///
6875/// A plain spawn replaces the module's recorded nonce. A swap candidate records
6876/// a separate candidate token so the still-serving incumbent and its consumers
6877/// keep their nonce. Both records are installed before the process exists, so
6878/// the child's initial HELLO registration cannot arrive ahead of its nonce.
6879///
6880/// On Unix the nonce is delivered only through a pipe. It is written into
6881/// a pipe whose read end the child gets as descriptor 3, named by
6882/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
6883/// process of the same user cannot read it with `ps eww`. That handoff is
6884/// returned rather than installed here, because installing it replaces
6885/// whatever the child has at descriptor 3 and so must be the last pre-exec
6886/// step, after the Linux cgroup placement that the caller registers later.
6887/// Windows retains the environment handoff until restricted handle inheritance
6888/// can be implemented outside std's process primitives.
6889fn apply_wire_spawn_args_for_role(
6890    command: &mut Command,
6891    spec: &ModuleSpec,
6892    connection_file_path: Option<&std::path::Path>,
6893    handle: Option<&SupervisorHandle>,
6894    role: SpawnRole,
6895) -> Result<Option<NonceHandoff>, SuperviseError> {
6896    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
6897    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
6898    // included: a daemon started from a module's process tree inherits it,
6899    // and passing it on would point the child at a descriptor it does not
6900    // have.
6901    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
6902    // Remove inherited or configured copies too: withholding must mean absent.
6903    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
6904    if spec.protocol == ModuleProtocol::None {
6905        return Ok(None);
6906    }
6907    if let Some(connection_file_path) = connection_file_path {
6908        command.arg(SUBC_ARG).arg(connection_file_path);
6909    }
6910
6911    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
6912    // route.open attestation. Reserved modules additionally use the same nonce
6913    // for HELLO id-squatting protection. A respawn rotates both records.
6914    let nonce = generate_launch_nonce()?;
6915    if let Some(handle) = handle {
6916        match role {
6917            SpawnRole::Plain => {
6918                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
6919                if spec.reserved {
6920                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
6921                }
6922            }
6923            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
6924        }
6925    }
6926    #[cfg(unix)]
6927    let handoff = {
6928        let handoff =
6929            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
6930                program: spec.program.clone(),
6931                source,
6932                cgroup_path: None,
6933            })?;
6934        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
6935        Some(handoff)
6936    };
6937    #[cfg(not(unix))]
6938    let handoff = None;
6939    // Windows keeps the environment copy: std cannot restrict an inherited pipe
6940    // handle to this child without leaking it to concurrently spawned processes.
6941    #[cfg(not(unix))]
6942    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
6943    Ok(handoff)
6944}
6945
6946fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
6947    command.env_remove(CK_LOG_ENV);
6948    // The spawn role is the supervisor's to set, and only on a swap candidate
6949    // (see `apply_spawn_role`). Removing it here, rather than just not setting
6950    // it, is what makes it absent on a plain spawn: the daemon's own
6951    // environment could carry it, and so could a spec built outside daemon
6952    // config (config refuses it as an `env` key). A module reading it on a
6953    // plain restart would pick the long swap budget and leave callers waiting.
6954    command.env_remove(SUBC_SPAWN_ROLE_ENV);
6955    for (key, value) in &spec.env {
6956        // cortexkit-log currently exposes retention only as a Rust struct, not
6957        // environment names. These values are daemon-private sink metadata and
6958        // must never become a public child-process contract by being inherited.
6959        if matches!(
6960            key.as_str(),
6961            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
6962        ) || key == SUBC_SPAWN_ROLE_ENV
6963        {
6964            continue;
6965        }
6966        command.env(key, value);
6967    }
6968}
6969
6970/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
6971/// of a blue/green swap.
6972#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6973enum SpawnRole {
6974    Plain,
6975    SwapCandidate,
6976}
6977
6978/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
6979/// `apply_child_env` has already removed the variable for every spawn.
6980fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
6981    if role == SpawnRole::SwapCandidate {
6982        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
6983    }
6984}
6985
6986fn spawn_child(
6987    spec: &ModuleSpec,
6988    connection_file_path: Option<&std::path::Path>,
6989    handle: Option<&SupervisorHandle>,
6990    ring: &Arc<Mutex<StderrRing>>,
6991    capture_logs_dir: Option<&std::path::Path>,
6992    roster: &ChildRoster,
6993    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
6994) -> Result<SupervisedChild, SuperviseError> {
6995    spawn_child_in_slot(
6996        spec,
6997        connection_file_path,
6998        handle,
6999        ring,
7000        capture_logs_dir,
7001        roster,
7002        #[cfg(target_os = "linux")]
7003        cgroup_placement,
7004        SpawnRole::Plain,
7005        false,
7006    )
7007}
7008
7009/// Spawn one process of `spec` into a slot.
7010///
7011/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7012/// A swap candidate needs a different cgroup from the process it is replacing,
7013/// which is still alive: in the same cgroup the two would be one kill domain,
7014/// and killing a failed candidate could take the incumbent with it.
7015///
7016/// The stderr capture file is `<module_id>.stderr.log` for every process of
7017/// the module, whichever slot it is in, because that is the one file
7018/// `ck module logs` reads. During a swap's overlap both processes append to it;
7019/// the daemon writes whole lines, so the two interleave by line, which is also
7020/// the merged view an operator wants while a swap runs.
7021#[allow(clippy::too_many_arguments)]
7022fn spawn_child_in_slot(
7023    spec: &ModuleSpec,
7024    connection_file_path: Option<&std::path::Path>,
7025    handle: Option<&SupervisorHandle>,
7026    ring: &Arc<Mutex<StderrRing>>,
7027    capture_logs_dir: Option<&std::path::Path>,
7028    roster: &ChildRoster,
7029    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7030    role: SpawnRole,
7031    alternate_slot: bool,
7032) -> Result<SupervisedChild, SuperviseError> {
7033    if roster.is_closed() {
7034        return Err(SuperviseError::Spawn {
7035            program: spec.program.clone(),
7036            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7037            cgroup_path: None,
7038        });
7039    }
7040    #[cfg(target_os = "linux")]
7041    let cgroup_name = {
7042        // Slot names alone are not kill domains: a retired incumbent may still
7043        // be draining when a later enable/restart spawns into the same slot.
7044        // Decimal entropy keeps the suffix unambiguous; Placement performs
7045        // the module-id escaping and constructs the filesystem path.
7046        if cgroup_placement.is_none() {
7047            swap::cgroup_name(&spec.module_id, alternate_slot)
7048        } else {
7049            let nonce = generate_launch_nonce()?;
7050            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7051            // Leave room for byte escaping and the suffix under NAME_MAX. The
7052            // label is only for humans; the nonce identifies the kill domain.
7053            let mut end = spec.module_id.len().min(64);
7054            while !spec.module_id.is_char_boundary(end) {
7055                end -= 1;
7056            }
7057            format!(
7058                "{}_{suffix}",
7059                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7060            )
7061        }
7062    };
7063    #[cfg(not(target_os = "linux"))]
7064    let _ = alternate_slot;
7065    let mut command = Command::new(&spec.program);
7066    command.args(&spec.args);
7067    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7068    // that is the whole of the intent, so remove that one key rather than the
7069    // environment.
7070    //
7071    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7072    // and took the POSIX environment with it. Modules spawned that way had no
7073    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7074    // logging:
7075    //
7076    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7077    //     both unset it fell back to the temp dir alone and `ck` could not find
7078    //     a daemon running on the same machine from inside any module's process
7079    //     tree — reporting a path the file has never lived at, which reads as
7080    //     "the daemon did not write its file".
7081    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7082    //     the RELATIVE `.local/share`, so a module deriving its own store path
7083    //     resolved it against its own CWD. That is the store-fragmentation
7084    //     defect the daemon already refuses in config (`parse_doc` rejects a
7085    //     relative `storage.data_home`) arriving by derivation instead.
7086    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7087    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7088    //     quietly rather than erroring.
7089    //
7090    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7091    // offered one candidate under /tmp while the file sat in /run/user/1000.
7092    //
7093    // A configured module is unaffected either way: `module_spec()` puts the
7094    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7095    // wins over anything ambient.
7096    apply_child_env(&mut command, spec);
7097    apply_spawn_role(&mut command, role);
7098    let nonce_handoff =
7099        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7100
7101    #[cfg(target_os = "linux")]
7102    let cgroup_path = cgroup_placement
7103        .map(|placement| placement.module_path(&cgroup_name))
7104        .transpose()
7105        .map_err(|source| SuperviseError::Cgroup {
7106            module_id: spec.module_id.clone(),
7107            source,
7108        })?;
7109    #[cfg(not(target_os = "linux"))]
7110    let cgroup_path: Option<PathBuf> = None;
7111    #[cfg(target_os = "linux")]
7112    if let Some(path) = &cgroup_path {
7113        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7114            if let Some(placement) = cgroup_placement {
7115                remove_module_cgroup(placement, &cgroup_name);
7116            }
7117            return Err(error);
7118        }
7119    }
7120
7121    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7122        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7123        match ChildOutputSink::open(&path, capture_retention(spec)) {
7124            Ok(sink) => sink,
7125            Err(error) => {
7126                warn!(
7127                    module_id = %spec.module_id,
7128                    path = %path.display(),
7129                    error = %error,
7130                    "could not open child output capture file; forwarding to stderr"
7131                );
7132                ChildOutputSink::Stderr
7133            }
7134        }
7135    } else {
7136        ChildOutputSink::Stderr
7137    };
7138
7139    command.stdout(Stdio::piped());
7140    command.stderr(Stdio::piped());
7141    command.kill_on_drop(true);
7142    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7143    // before exec). In the daemon's group, a service manager that kills the
7144    // job's process group when the daemon exits (launchd's default) killed
7145    // every module at the same moment its control connection closed, so no
7146    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7147    // module is reached only by the daemon: the EOF it sees when its
7148    // connection closes, and the bounded stop in `child_roster` for anything
7149    // still running after that. On Linux this composes with the cgroup
7150    // placement above: that is a pre_exec write to cgroup.procs, std performs
7151    // setpgid in the child before running pre_exec callbacks, and the two
7152    // change independent process attributes.
7153    //
7154    // stdin is /dev/null because a process outside the terminal's foreground
7155    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7156    // by hand would otherwise hand down. Under a service manager stdin is
7157    // already /dev/null.
7158    #[cfg(unix)]
7159    command.process_group(0);
7160    command.stdin(Stdio::null());
7161    // The LAST pre-exec step, after the cgroup placement above: installing the
7162    // nonce at descriptor 3 replaces whatever the child had there, which could
7163    // be the descriptor an earlier step writes through.
7164    #[cfg(unix)]
7165    if let Some(handoff) = nonce_handoff {
7166        handoff.install_last(command.as_std_mut());
7167    }
7168    #[cfg(not(unix))]
7169    let _ = nonce_handoff;
7170
7171    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7172    // cannot run a single instruction -- and therefore cannot spawn a
7173    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7174    // other two steps and why the window matters.
7175    #[cfg(windows)]
7176    subc_jobobject::suspend_on_create_async(&mut command);
7177    let mut child = match command.spawn() {
7178        Ok(child) => child,
7179        Err(source) => {
7180            #[cfg(target_os = "linux")]
7181            if let Some(placement) = cgroup_placement {
7182                remove_module_cgroup(placement, &cgroup_name);
7183            }
7184            return Err(SuperviseError::Spawn {
7185                program: spec.program.clone(),
7186                source,
7187                cgroup_path,
7188            });
7189        }
7190    };
7191
7192    // Containment, steps 2 and 3: assign while suspended, then resume.
7193    #[cfg(windows)]
7194    let job = contain_spawned_child(&child, spec)?;
7195    let spawned_at_ms = unix_ms_now();
7196    let spawned_from = spec.program.clone();
7197    let spawned_file_identity = spawned_file_identity(&spawned_from);
7198    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7199        program: spec.program.clone(),
7200        source: io::Error::other("spawned child exposed no live pid"),
7201        cgroup_path: cgroup_path.clone(),
7202    })?;
7203    let process_start_time = crate::provenance::process_start_time(pid);
7204    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7205    // Unix spawn returns after exec's error pipe closes. The kernel image is
7206    // therefore the executable to compare during a future orphan sweep: PATH
7207    // lookup and shebang interpretation may select a different file from the
7208    // configured program. Keep the literal program's identity for provenance,
7209    // but never use it as proof that a recorded pid may be signalled.
7210    let recorded_image = subc_os::Process::open(pid)
7211        .ok()
7212        .flatten()
7213        .and_then(|process| process.observe());
7214    #[cfg(target_os = "linux")]
7215    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7216    #[cfg(not(target_os = "linux"))]
7217    let recorded_cgroup_name = None;
7218    let roster_guard = roster.admit(
7219        spec.module_id.clone(),
7220        pid,
7221        spec.protocol,
7222        process_start_time,
7223        crate::child_roster::RecordedIdentity {
7224            start_time: recorded_image.map(|image| image.start_time),
7225            executable: recorded_image
7226                .and_then(|image| image.executable)
7227                .map(crate::live_children::ExecutableIdentity::from),
7228            cgroup_name: recorded_cgroup_name,
7229            #[cfg(target_os = "linux")]
7230            cgroup_placement: cgroup_placement.cloned(),
7231        },
7232    );
7233    // The check at the top of this function can pass just before daemon
7234    // shutdown begins, and the process is only in the roster from here on.
7235    // The shutdown stop returns as soon as it finds the roster empty, so a
7236    // process admitted after that look would outlive the daemon. The roster
7237    // is closed before the stop first reads it and admission happens under
7238    // the roster's lock, so either the stop sees this process or this check
7239    // sees the roster closed: end the process now rather than start a module
7240    // the daemon is about to stop.
7241    if roster.is_closed() {
7242        // This child was never admitted, so there is no module protocol shutdown to wait for.
7243        #[cfg(target_os = "linux")]
7244        kill_module_cgroup(cgroup_placement, &cgroup_name);
7245        if let Err(error) = child.start_kill() {
7246            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7247        }
7248        #[cfg(target_os = "linux")]
7249        if let Some(placement) = cgroup_placement {
7250            // This spawn was never admitted, so shutdown has no roster entry
7251            // to await. Do not detach its cleanup: the runtime could exit
7252            // before that task reaps the rejected child and removes its group.
7253            while matches!(child.try_wait(), Ok(None)) {
7254                std::thread::yield_now();
7255            }
7256            if matches!(
7257                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7258                subc_cgroup::KillOutcome::Killed
7259            ) {
7260                if let Ok(path) = placement.module_path(&cgroup_name) {
7261                    while std::fs::read_to_string(path.join("cgroup.events"))
7262                        .ok()
7263                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7264                    {
7265                        std::thread::yield_now();
7266                    }
7267                }
7268            }
7269            remove_module_cgroup(placement, &cgroup_name);
7270        }
7271        drop(roster_guard);
7272        return Err(SuperviseError::Spawn {
7273            program: spec.program.clone(),
7274            source: io::Error::other(
7275                "the daemon began shutting down while this process was starting; ended it",
7276            ),
7277            cgroup_path,
7278        });
7279    }
7280
7281    let stdout_pump = match child.stdout.take() {
7282        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7283        None => {
7284            warn!(
7285                module_id = %spec.module_id,
7286                "spawned child exposed no stdout pipe; file capture will be incomplete"
7287            );
7288            None
7289        }
7290    };
7291    let stderr_pump = match child.stderr.take() {
7292        Some(stderr) => {
7293            let generation = ring
7294                .lock()
7295                .unwrap_or_else(|poisoned| poisoned.into_inner())
7296                .begin_process();
7297            Some(StderrPump {
7298                task: tokio::spawn(pump_stderr_to(
7299                    stderr,
7300                    Arc::clone(ring),
7301                    generation,
7302                    output_sink,
7303                )),
7304                generation,
7305            })
7306        }
7307        None => {
7308            // Spawning succeeded but the pipe did not materialise. Recording it as
7309            // uncaptured keeps the tail honest: the alternative is an empty tail
7310            // that reads as a module which printed nothing.
7311            ring.lock()
7312                .unwrap_or_else(|poisoned| poisoned.into_inner())
7313                .mark_not_captured("stderr pipe was not available on spawn");
7314            warn!(
7315                module_id = %spec.module_id,
7316                "spawned child exposed no stderr pipe; tail will be unavailable"
7317            );
7318            None
7319        }
7320    };
7321
7322    Ok(SupervisedChild {
7323        child,
7324        protocol: spec.protocol,
7325        #[cfg(target_os = "linux")]
7326        module_id: cgroup_name,
7327        #[cfg(target_os = "linux")]
7328        cgroup_placement: cgroup_placement.cloned(),
7329        #[cfg(windows)]
7330        job,
7331        stdout_pump,
7332        stderr_pump,
7333        stderr_ring: Arc::clone(ring),
7334        spawned_at_ms,
7335        spawned_from,
7336        spawned_file_identity,
7337        process_start_time,
7338        process_identity,
7339        pid,
7340        roster_guard: Some(roster_guard),
7341    })
7342}
7343
7344#[cfg(target_os = "linux")]
7345pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7346    use subc_cgroup::KillOutcome;
7347    match subc_cgroup::kill_module(placement, module_id) {
7348        KillOutcome::Killed => {}
7349        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7350            debug!(
7351                module_id,
7352                "cgroup tree kill unavailable; using direct-child kill"
7353            );
7354        }
7355        KillOutcome::IoError { path, error } => {
7356            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7357        }
7358    }
7359}
7360
7361/// Contain a freshly spawned Windows child and start it.
7362///
7363/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7364/// child assigned **while it is still suspended** (step 1 is
7365/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7366///
7367/// A child that is never resumed hangs forever holding a pid, so a resume
7368/// failure kills the child and fails the spawn rather than returning a
7369/// `SupervisedChild` that can never run.
7370///
7371/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7372/// it did before this existed, whereas refusing to start one would be a new
7373/// outage. It is logged at warn because it means a helper process could leak.
7374#[cfg(windows)]
7375fn contain_spawned_child(
7376    child: &Child,
7377    spec: &ModuleSpec,
7378) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7379    let module_id = spec.module_id.as_str();
7380    let Some(pid) = child.id() else {
7381        // The child exited between spawn and here. Its tree, if it made one,
7382        // needs no containment: nothing is left to contain.
7383        warn!(
7384            module_id,
7385            "spawned child had already exited before containment; no job object attached"
7386        );
7387        return Ok(None);
7388    };
7389
7390    let job = match subc_jobobject::JobObject::new() {
7391        Ok(job) => job,
7392        Err(source) => {
7393            warn!(
7394                module_id,
7395                error = %source,
7396                "could not create a job object; this module's helper processes will not be \
7397                 reaped on teardown"
7398            );
7399            // Resume regardless: leaving the child suspended would turn a
7400            // containment gap into a hung module.
7401            resume_suspended_child(pid, spec)?;
7402            return Ok(None);
7403        }
7404    };
7405
7406    if let Err(source) = job.assign(child) {
7407        warn!(
7408            module_id,
7409            error = %source,
7410            "could not assign the child to its job object; this module's helper processes \
7411             will not be reaped on teardown"
7412        );
7413        resume_suspended_child(pid, spec)?;
7414        return Ok(None);
7415    }
7416
7417    resume_suspended_child(pid, spec)?;
7418    Ok(Some(job))
7419}
7420
7421/// Resume a suspended child, killing it if it cannot be started.
7422///
7423/// A suspended process holds a pid and does nothing, so there is no useful
7424/// state to return: the caller gets an error and the spawn fails.
7425#[cfg(windows)]
7426fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7427    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7428        // Kill it here rather than leaving a suspended process for the caller
7429        // to notice; `kill_on_drop` would eventually do this, but the module
7430        // would have been reported as running in between.
7431        let _ = std::process::Command::new("taskkill.exe")
7432            .args(["/PID", &pid.to_string(), "/T", "/F"])
7433            .stdin(Stdio::null())
7434            .stdout(Stdio::null())
7435            .stderr(Stdio::null())
7436            .status();
7437        return Err(SuperviseError::Spawn {
7438            program: spec.program.clone(),
7439            source,
7440            cgroup_path: None,
7441        });
7442    }
7443    Ok(())
7444}
7445
7446#[cfg(target_os = "linux")]
7447fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7448    match placement.remove_module(module_id) {
7449        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7450        Err(error) => warn!(
7451            module_id,
7452            error = %error,
7453            "could not remove module cgroup after process exit; continuing teardown"
7454        ),
7455    }
7456}
7457
7458#[cfg(target_os = "linux")]
7459async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7460    // Reaping the direct child is not proof its descendants exited. End the
7461    // residual tree and wait for the kernel's population fact before rmdir;
7462    // otherwise a successful parent wait leaks a directory on each restart.
7463    if matches!(
7464        subc_cgroup::kill_module(Some(placement), module_id),
7465        subc_cgroup::KillOutcome::Killed
7466    ) {
7467        if let Ok(path) = placement.module_path(module_id) {
7468            while std::fs::read_to_string(path.join("cgroup.events"))
7469                .ok()
7470                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7471            {
7472                sleep(Duration::from_millis(1)).await;
7473            }
7474        }
7475    }
7476    remove_module_cgroup(placement, module_id);
7477}
7478
7479#[cfg(target_os = "linux")]
7480fn apply_cgroup_placement(
7481    command: &mut Command,
7482    spec: &ModuleSpec,
7483    path: &std::path::Path,
7484) -> Result<(), SuperviseError> {
7485    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7486        module_id: spec.module_id.clone(),
7487        source,
7488    })
7489}
7490
7491fn capture_retention(spec: &ModuleSpec) -> Retention {
7492    let defaults = Retention::default();
7493    let value = |name: &str| {
7494        spec.env
7495            .iter()
7496            .rev()
7497            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7498    };
7499    Retention {
7500        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7501            .and_then(|value| value.parse().ok())
7502            .unwrap_or(defaults.max_file_mb),
7503        keep: value(CAPTURE_KEEP_ENV)
7504            .and_then(|value| value.parse().ok())
7505            .unwrap_or(defaults.keep),
7506        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7507            .and_then(|value| value.parse().ok())
7508            .unwrap_or(defaults.max_age_days),
7509    }
7510}
7511
7512/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
7513/// module's registration to the exact process the supervisor spawned.
7514fn generate_launch_nonce() -> Result<String, SuperviseError> {
7515    let mut bytes = [0u8; 32];
7516    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7517        reason: source.to_string(),
7518    })?;
7519    let mut hex = String::with_capacity(64);
7520    for b in bytes {
7521        use std::fmt::Write;
7522        let _ = write!(hex, "{b:02x}");
7523    }
7524    Ok(hex)
7525}
7526
7527/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
7528/// signal about how many leading bytes matched.
7529fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7530    if a.len() != b.len() {
7531        return false;
7532    }
7533    let mut diff = 0u8;
7534    for (x, y) in a.iter().zip(b.iter()) {
7535        diff |= x ^ y;
7536    }
7537    diff == 0
7538}
7539
7540fn spawn_and_mark_running(
7541    spec: &ModuleSpec,
7542    runtime: &SupervisorRuntimeConfig,
7543    snapshot: &SharedSnapshot,
7544) -> Result<SupervisedChild, SuperviseError> {
7545    let child = spawn_child(
7546        spec,
7547        runtime.connection_file_path.as_deref(),
7548        runtime.supervisor_handle.as_ref(),
7549        &runtime.stderr_ring,
7550        runtime.capture_logs_dir.as_deref(),
7551        &runtime.child_roster,
7552        #[cfg(target_os = "linux")]
7553        runtime.cgroup_placement.as_ref(),
7554    )?;
7555    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
7556    Ok(child)
7557}
7558
7559enum RegistrationWaitOutcome {
7560    Registered,
7561    Exited(ExitReport),
7562    TimedOut,
7563}
7564
7565struct ReloadRegistrationFailure {
7566    exit_report: ExitReport,
7567    reason: String,
7568}
7569
7570#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7571enum BusyGaugeObservation {
7572    Quiescent,
7573    Busy,
7574    Omitted,
7575}
7576
7577fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
7578    let Some(metrics) = metrics.and_then(Value::as_object) else {
7579        return BusyGaugeObservation::Omitted;
7580    };
7581    let mut sum = 0u128;
7582    for gauge in gauges {
7583        let Some(value) = metrics.get(gauge) else {
7584            return BusyGaugeObservation::Omitted;
7585        };
7586        let Some(value) = value.as_u64() else {
7587            return BusyGaugeObservation::Busy;
7588        };
7589        sum = sum.saturating_add(u128::from(value));
7590    }
7591    if sum == 0 {
7592        BusyGaugeObservation::Quiescent
7593    } else {
7594        BusyGaugeObservation::Busy
7595    }
7596}
7597
7598fn declared_busy_gauges(
7599    registry: &Registry,
7600    module_id: &str,
7601) -> Result<Vec<String>, SuperviseError> {
7602    busy_gauges_of(
7603        registry
7604            .get_module(module_id)
7605            .map_err(SuperviseError::Registry)?,
7606    )
7607}
7608
7609/// [`declared_busy_gauges`] for the registration a connection holds, in any
7610/// slot: after cutover the incumbent is no longer the id's active
7611/// registration, and its own manifest is the one that names its gauges.
7612fn declared_busy_gauges_for_connection(
7613    registry: &Registry,
7614    connection_id: ConnectionId,
7615) -> Result<Vec<String>, SuperviseError> {
7616    busy_gauges_of(
7617        registry
7618            .get_module_by_connection(connection_id)
7619            .map_err(SuperviseError::Registry)?,
7620    )
7621}
7622
7623fn busy_gauges_of(
7624    registration: Option<crate::registry::ModuleRegistration>,
7625) -> Result<Vec<String>, SuperviseError> {
7626    let Some(registration) = registration else {
7627        return Ok(Vec::new());
7628    };
7629    let Some(self_signals) = registration.manifest.self_signals else {
7630        return Ok(Vec::new());
7631    };
7632
7633    let mut gauges = Vec::new();
7634    for declaration in self_signals {
7635        if declaration.kind != SelfSignalKind::Busy {
7636            continue;
7637        }
7638        match declaration.anchored_to {
7639            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
7640                gauges.extend(declared)
7641            }
7642            _ => {
7643                // An invalid Busy anchor is fail-safe: the empty name cannot be
7644                // present in a conforming health report, so this drain stays busy.
7645                gauges.push(String::new());
7646            }
7647        }
7648    }
7649    Ok(gauges)
7650}
7651
7652/// Wait for `endpoint` to have nothing in flight and, when the module declares
7653/// busy gauges, for a health probe to report them quiet. The probe is addressed
7654/// by `scope`: a swap's superseded incumbent must be asked about its own
7655/// gauges, and by module id the probe would reach the promoted candidate.
7656async fn wait_for_forwarding_quiescence(
7657    forwarding: &ForwardingTable,
7658    module_id: &str,
7659    runtime: &SupervisorRuntimeConfig,
7660    endpoint: crate::ModuleEndpointId,
7661    deadline: Instant,
7662    busy_gauges: &[String],
7663    scope: DrainScope,
7664) -> Result<bool, SuperviseError> {
7665    let mut gauges_quiescent = busy_gauges.is_empty();
7666    let mut next_probe_at = Instant::now();
7667    let mut omission_counted = false;
7668
7669    loop {
7670        let now = Instant::now();
7671        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
7672            let report = match scope {
7673                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
7674                DrainScope::Endpoint(endpoint) => {
7675                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
7676                }
7677            };
7678            gauges_quiescent = match report {
7679                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
7680                    BusyGaugeObservation::Quiescent => true,
7681                    BusyGaugeObservation::Busy => false,
7682                    BusyGaugeObservation::Omitted => {
7683                        if !omission_counted {
7684                            forwarding
7685                                .counters()
7686                                .increment_drains_with_undeclared_gauge();
7687                            omission_counted = true;
7688                        }
7689                        false
7690                    }
7691                },
7692                Err(err) => {
7693                    warn!(
7694                        module_id,
7695                        error = %err,
7696                        "drain health.check did not produce declared busy gauges; treating module as busy"
7697                    );
7698                    false
7699                }
7700            };
7701            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
7702        }
7703
7704        let in_flight = forwarding
7705            .endpoint_in_flight_count(endpoint)
7706            .map_err(SuperviseError::Forwarding)?;
7707        if in_flight == 0 && gauges_quiescent {
7708            return Ok(true);
7709        }
7710
7711        let now = Instant::now();
7712        if now >= deadline {
7713            return Ok(false);
7714        }
7715        let mut wait = deadline
7716            .saturating_duration_since(now)
7717            .min(REGISTRY_RELEASE_POLL);
7718        if !busy_gauges.is_empty() {
7719            wait = wait.min(next_probe_at.saturating_duration_since(now));
7720        }
7721        sleep(wait).await;
7722    }
7723}
7724
7725/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
7726///
7727/// `Ok` is always honest and passed straight through -- the wait actually measured
7728/// in-flight state. `Err` means the wait produced no measurement at all (the
7729/// forwarding table's lock was poisoned), so `false` is reported as the one honest
7730/// constant: the drain did not complete. Never recomputed from route state, never a
7731/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
7732fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
7733    match wait_result {
7734        Ok(drained) => *drained,
7735        Err(_) => false,
7736    }
7737}
7738
7739fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
7740    for released in released_routes {
7741        let frame = match Frame::build_with_version(
7742            released.negotiated_ver,
7743            FrameType::Goodbye,
7744            control_flags(),
7745            released.channel,
7746            released.epoch,
7747            0,
7748            Vec::new(),
7749        ) {
7750            Ok(frame) => frame,
7751            Err(err) => {
7752                warn!(
7753                    route_channel = released.channel,
7754                    error = %err,
7755                    "failed to build supervisor drain route GOODBYE frame"
7756                );
7757                continue;
7758            }
7759        };
7760        if !released.close_on_delivery_failure() {
7761            crate::forwarding::send_module_route_goodbye(
7762                &forwarding.counters(),
7763                &released.sink,
7764                frame,
7765                released.module_id.as_deref(),
7766                "supervisor drain",
7767            );
7768            continue;
7769        }
7770        if let Err(err) = released.sink.try_send(frame) {
7771            warn!(
7772                target_connection_id = released.connection_id.get(),
7773                route_channel = released.channel,
7774                error = %err,
7775                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
7776            );
7777            let _ = forwarding.escalate_client_delivery_failure(
7778                released.connection_id,
7779                released.channel,
7780                released.epoch,
7781                CloseReason::new(
7782                    "route_goodbye_delivery_failed",
7783                    format!(
7784                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
7785                        released.channel
7786                    ),
7787                ),
7788                crate::forwarding::UndeliveredFrame {
7789                    module_id: released.module_id.as_deref(),
7790                    sink: &released.sink,
7791                },
7792            );
7793        }
7794    }
7795}
7796
7797fn send_module_draining(
7798    module_id: &str,
7799    reason: RouteCloseReason,
7800    deadline_ms: u64,
7801    target: &ModuleDrainTarget,
7802) {
7803    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
7804        reason,
7805        deadline_ms,
7806    }) {
7807        Ok(body) => body,
7808        Err(err) => {
7809            warn!(
7810                module_id,
7811                error = %err,
7812                "failed to encode module draining command"
7813            );
7814            return;
7815        }
7816    };
7817    let frame = match Frame::build_with_version(
7818        target.negotiated_ver,
7819        FrameType::Push,
7820        control_flags(),
7821        0,
7822        0,
7823        0,
7824        body,
7825    ) {
7826        Ok(frame) => frame,
7827        Err(err) => {
7828            warn!(
7829                module_id,
7830                error = %err,
7831                "failed to build module draining command frame"
7832            );
7833            return;
7834        }
7835    };
7836    if let Err(err) = target.sink.try_send(frame) {
7837        warn!(
7838            module_id,
7839            target_connection_id = target.endpoint.connection_id.get(),
7840            error = %err,
7841            "module draining command was not delivered to peer"
7842        );
7843    }
7844}
7845
7846/// The channel-0 GOODBYE that tells a module its stop is planned.
7847fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
7848    match Frame::build_with_version(
7849        negotiated_ver,
7850        FrameType::Goodbye,
7851        control_flags(),
7852        0,
7853        0,
7854        0,
7855        Vec::new(),
7856    ) {
7857        Ok(frame) => Some(frame),
7858        Err(err) => {
7859            warn!(
7860                module_id,
7861                error = %err,
7862                "failed to build module GOODBYE frame"
7863            );
7864            None
7865        }
7866    }
7867}
7868
7869/// Send every registered module connection its module GOODBYE at daemon
7870/// shutdown, then request that connection's close.
7871///
7872/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
7873/// before EOF, so the GOODBYE must reach the socket before the close. A close
7874/// request does not wait for the connection's queued frames: its writer gets a
7875/// bounded grace after the close, is aborted if it overruns it, and the daemon
7876/// process may exit before that grace ends. So with `wait_for_flush`, each
7877/// connection is closed only after its writer has acknowledged writing the
7878/// GOODBYE, or once a short shared budget runs out, so one module that is not
7879/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
7880/// are only queued, for a shutdown the operator has told to stop waiting.
7881/// A connection that is already gone is skipped.
7882#[cfg(unix)]
7883async fn send_module_goodbyes_for_daemon_shutdown(
7884    forwarding: &Arc<ForwardingTable>,
7885    reason: &CloseReason,
7886    wait_for_flush: bool,
7887) {
7888    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
7889    let targets = match forwarding.module_connections() {
7890        Ok(targets) => targets,
7891        Err(err) => {
7892            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
7893            return;
7894        }
7895    };
7896    let deadline = Instant::now() + GOODBYE_BUDGET;
7897    let mut sends = tokio::task::JoinSet::new();
7898    for target in targets {
7899        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
7900            continue;
7901        };
7902        if !wait_for_flush {
7903            if let Err(err) = target.sink.try_send(frame) {
7904                debug!(
7905                    module_id = %target.module_id,
7906                    error = %err,
7907                    "shutdown module GOODBYE was not queued"
7908                );
7909            }
7910            continue;
7911        }
7912        let forwarding = Arc::clone(forwarding);
7913        let reason = reason.clone();
7914        sends.spawn(async move {
7915            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
7916                Ok(Ok(())) => {}
7917                Ok(Err(err)) => debug!(
7918                    module_id = %target.module_id,
7919                    error = %err,
7920                    "module connection closed before its shutdown GOODBYE was written"
7921                ),
7922                Err(_) => warn!(
7923                    module_id = %target.module_id,
7924                    budget = ?GOODBYE_BUDGET,
7925                    "shutdown module GOODBYE was not written within its budget; closing anyway"
7926                ),
7927            }
7928            forwarding.request_connection_close(target.endpoint.connection_id, reason);
7929        });
7930    }
7931    // Every task ends by the shared deadline, so this wait is bounded too.
7932    while sends.join_next().await.is_some() {}
7933}
7934
7935fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
7936    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
7937        return;
7938    };
7939    if let Err(err) = target.sink.try_send(frame) {
7940        warn!(
7941            module_id,
7942            target_connection_id = target.endpoint.connection_id.get(),
7943            error = %err,
7944            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
7945        );
7946        forwarding.request_connection_close(
7947            target.endpoint.connection_id,
7948            CloseReason::new(
7949                "module_goodbye_delivery_failed",
7950                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
7951            ),
7952        );
7953    }
7954}
7955
7956#[derive(Clone, Copy)]
7957struct ForwardingDrainContext<'a> {
7958    spec: &'a ModuleSpec,
7959    runtime: &'a SupervisorRuntimeConfig,
7960    registry: &'a Registry,
7961    scope: DrainScope,
7962}
7963
7964/// Which process a forwarding drain addresses.
7965#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7966enum DrainScope {
7967    /// Whatever endpoint is active for the module id: every plain stop,
7968    /// restart and reload. Also moves the module's state to `Draining`.
7969    Active,
7970    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
7971    /// module id would resolve to the promoted candidate and leave neither
7972    /// process routable. The module's state is left alone, since the promoted
7973    /// candidate is what it describes and that process is running.
7974    Endpoint(crate::ModuleEndpointId),
7975}
7976
7977/// Whether a child being drained has already been asked to stop by the time
7978/// its drain wait starts.
7979///
7980/// The drain wait is the same budget whatever this says. What it decides is
7981/// whether the supervisor must ask by signal before that wait begins: a child
7982/// that nobody asked will sit out the whole budget and then be SIGKILLed,
7983/// healthy or not.
7984#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7985enum StopNotice {
7986    /// The module was sent `module.draining` and a module GOODBYE over its own
7987    /// registered connection, and stops itself.
7988    SentOverConnection,
7989    /// The forwarding drain found no registered connection for the module: a
7990    /// subc child spawned moments ago that has not sent HELLO yet, or a
7991    /// `protocol: "none"` child, which never registers.
7992    NoConnection,
7993    /// This path sends nothing over the module's connection: the supervisor has
7994    /// no forwarding table, or the caller stops the child without a forwarding
7995    /// drain.
7996    NotSent,
7997}
7998
7999async fn begin_forwarding_drain(
8000    spec: &ModuleSpec,
8001    runtime: &SupervisorRuntimeConfig,
8002    registry: &Registry,
8003    snapshot: &SharedSnapshot,
8004    enabled: Option<bool>,
8005    reason: RouteCloseReason,
8006) -> Result<StopNotice, SuperviseError> {
8007    let Some(forwarding) = runtime.forwarding.as_ref() else {
8008        return Err(SuperviseError::ReloadUnavailable {
8009            module_id: spec.module_id.clone(),
8010            reason: "supervisor was not configured with a forwarding table".to_string(),
8011        });
8012    };
8013
8014    begin_forwarding_drain_with(
8015        forwarding,
8016        ForwardingDrainContext {
8017            spec,
8018            runtime,
8019            registry,
8020            scope: DrainScope::Active,
8021        },
8022        snapshot,
8023        enabled,
8024        reason,
8025        runtime.drain_timeout,
8026    )
8027    .await
8028}
8029
8030async fn begin_forwarding_drain_if_configured(
8031    spec: &ModuleSpec,
8032    runtime: &SupervisorRuntimeConfig,
8033    registry: &Registry,
8034    snapshot: &SharedSnapshot,
8035    enabled: Option<bool>,
8036    reason: RouteCloseReason,
8037) -> Result<StopNotice, SuperviseError> {
8038    begin_forwarding_drain_with_timeout(
8039        spec,
8040        runtime,
8041        registry,
8042        snapshot,
8043        enabled,
8044        reason,
8045        runtime.drain_timeout,
8046    )
8047    .await
8048}
8049
8050/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8051/// budget, for paths where the operator overrides the module's configured one
8052/// (`supervisor.restart{drain_timeout_ms}`).
8053async fn begin_forwarding_drain_with_timeout(
8054    spec: &ModuleSpec,
8055    runtime: &SupervisorRuntimeConfig,
8056    registry: &Registry,
8057    snapshot: &SharedSnapshot,
8058    enabled: Option<bool>,
8059    reason: RouteCloseReason,
8060    drain_timeout: Duration,
8061) -> Result<StopNotice, SuperviseError> {
8062    let Some(forwarding) = runtime.forwarding.as_ref() else {
8063        return Ok(StopNotice::NotSent);
8064    };
8065
8066    begin_forwarding_drain_with(
8067        forwarding,
8068        ForwardingDrainContext {
8069            spec,
8070            runtime,
8071            registry,
8072            scope: DrainScope::Active,
8073        },
8074        snapshot,
8075        enabled,
8076        reason,
8077        drain_timeout,
8078    )
8079    .await
8080}
8081
8082async fn begin_forwarding_drain_with(
8083    forwarding: &ForwardingTable,
8084    context: ForwardingDrainContext<'_>,
8085    snapshot: &SharedSnapshot,
8086    enabled: Option<bool>,
8087    reason: RouteCloseReason,
8088    drain_timeout: Duration,
8089) -> Result<StopNotice, SuperviseError> {
8090    let ForwardingDrainContext {
8091        spec,
8092        runtime,
8093        registry,
8094        scope,
8095    } = context;
8096    debug_assert_ne!(reason, RouteCloseReason::Crash);
8097    let terminal = matches!(reason, RouteCloseReason::Disable);
8098    let drain_started_at = Instant::now();
8099    let drain_deadline = drain_started_at + drain_timeout;
8100    let deadline_ms =
8101        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8102    let busy_gauges = match scope {
8103        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8104        DrainScope::Endpoint(endpoint) => {
8105            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8106        }
8107    };
8108
8109    // Admission gate first: route.open/commit and route REQUEST admission are closed
8110    // before the first quiescence check, so the outstanding count can only fall.
8111    let gate_started = Instant::now();
8112    let drain_target = match scope {
8113        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8114        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8115    }
8116    .map_err(SuperviseError::Forwarding)?;
8117    // The instant admission closed, and how long taking the forwarding write
8118    // lock to close it took. The timeout line reports only the quiescence
8119    // wait, so without this a drain that started late looked like one that
8120    // started on time.
8121    info!(
8122        module_id = %spec.module_id,
8123        ?reason,
8124        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8125        connected = drain_target.is_some(),
8126        "module drain began; route admission closed"
8127    );
8128    if scope == DrainScope::Active {
8129        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8130            state.state = ModuleState::Draining;
8131            state.draining_to_replace =
8132                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8133            if let Some(enabled) = enabled {
8134                state.enabled = enabled;
8135            }
8136        })?;
8137    }
8138
8139    let Some(target) = drain_target.as_ref() else {
8140        // Nothing was sent: the module has no registered connection to carry
8141        // `module.draining` or a GOODBYE. The caller must not assume the child
8142        // was asked to stop.
8143        return Ok(StopNotice::NoConnection);
8144    };
8145    {
8146        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8147        let routes = forwarding
8148            .endpoint_routes(target.endpoint)
8149            .map_err(SuperviseError::Forwarding)?;
8150        let routes_notified = routes.len();
8151        crate::control::send_route_control_pushes(
8152            forwarding,
8153            routes.clone(),
8154            ClientControlPush::RouteClosing {
8155                module_id: spec.module_id.clone(),
8156                channels: Vec::new(),
8157                reason,
8158            },
8159        );
8160        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8161
8162        // `route.closing` was just sent above: from here on every return path,
8163        // including an early one, MUST send `route.closed` before propagating
8164        // anything else. A client holds `closing` as a promise that a verdict is
8165        // coming; leaving early without `closed` strands it waiting forever, since
8166        // `closing` carries no timeout of its own.
8167        let wait_result = wait_for_forwarding_quiescence(
8168            forwarding,
8169            &spec.module_id,
8170            runtime,
8171            target.endpoint,
8172            drain_deadline,
8173            &busy_gauges,
8174            scope,
8175        )
8176        .await;
8177        let drained = drained_after_quiescence_wait(&wait_result);
8178        if let Err(err) = &wait_result {
8179            error!(
8180                module_id = %spec.module_id,
8181                ?reason,
8182                error = %err,
8183                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8184            );
8185        } else if !drained {
8186            // Name what the drain waited on. Without it the line says only that
8187            // something did not settle, and "one wedged call" and "every
8188            // session's held stream" read the same; the first is a module bug,
8189            // the second is a module that should end its streams on
8190            // module.draining. Read before teardown releases the routes.
8191            let holdouts = forwarding
8192                .endpoint_drain_holdouts(target.endpoint)
8193                .unwrap_or_default();
8194            warn!(
8195                module_id = %spec.module_id,
8196                waited = ?drain_timeout,
8197                ?reason,
8198                held_requests = holdouts.requests,
8199                held_routes = holdouts.routes,
8200                total_routes = holdouts.total_routes,
8201                top_connections = ?holdouts.top_connections,
8202                // `module_channel:corr`, so the module can find each held request
8203                // in its own log; capped, so `held_requests` is the full count.
8204                held = %holdouts
8205                    .held
8206                    .iter()
8207                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8208                    .collect::<Vec<_>>()
8209                    .join(","),
8210                "route drain timed out before request quiescence; forcing teardown"
8211            );
8212        }
8213        crate::control::send_route_control_pushes(
8214            forwarding,
8215            routes,
8216            ClientControlPush::RouteClosed {
8217                module_id: spec.module_id.clone(),
8218                channels: Vec::new(),
8219                reason,
8220                drained,
8221                abandoned: target.abandoned_bindings.len() as u32,
8222                excluded_subscriptions: target.excluded_subscriptions,
8223                terminal: Some(terminal),
8224            },
8225        );
8226        wait_result?;
8227
8228        // `route.closed` has now been sent unconditionally above. From here the
8229        // remaining steps are cleanup (route + module GOODBYE) rather than a
8230        // promise the client is waiting on, but a lock-poisoned
8231        // `release_module_endpoint_routes` would otherwise skip the module
8232        // GOODBYE silently too -- send it before propagating the error.
8233        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8234            Ok(routes) => routes,
8235            Err(err) => {
8236                warn!(
8237                    module_id = %spec.module_id,
8238                    ?reason,
8239                    error = %err,
8240                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8241                );
8242                send_module_goodbye(&spec.module_id, forwarding, target);
8243                return Err(SuperviseError::Forwarding(err));
8244            }
8245        };
8246        let route_goodbye_count = released_routes.len();
8247        send_route_goodbyes(forwarding, released_routes);
8248        send_module_goodbye(&spec.module_id, forwarding, target);
8249
8250        // The drain's happy path was previously silent: every emission above is
8251        // best-effort with only its failure arm logged, so "were consumers told"
8252        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8253        // hang where the open question was exactly whether teardown notice went
8254        // out). One summary line makes that class decidable in one grep.
8255        info!(
8256            module_id = %spec.module_id,
8257            ?reason,
8258            routes_notified,
8259            route_goodbyes = route_goodbye_count,
8260            abandoned_reservations = target.abandoned_bindings.len(),
8261            excluded_subscriptions = target.excluded_subscriptions,
8262            drained,
8263            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8264        );
8265    }
8266
8267    Ok(StopNotice::SentOverConnection)
8268}
8269
8270/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8271/// the only slot a plain (non-swap) spawn can register into.
8272async fn wait_for_registration_after_reload(
8273    registry: &Registry,
8274    module_id: &str,
8275    snapshot: &SharedSnapshot,
8276    child: &mut SupervisedChild,
8277    wait: Duration,
8278) -> Result<RegistrationWaitOutcome, SuperviseError> {
8279    wait_for_slot_registration(
8280        registry,
8281        crate::registry::RegistrationSlot::Active(module_id),
8282        module_id,
8283        snapshot,
8284        child,
8285        wait,
8286    )
8287    .await
8288}
8289
8290/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8291///
8292/// Keyed on the slot rather than the bare module id because during a swap the
8293/// id's active slot is already held by the incumbent: an id-keyed wait would
8294/// report the incumbent's registration as the candidate's and a candidate that
8295/// never registers would look registered. A swap candidate waits on
8296/// `crate::registry::RegistrationSlot::Candidate`.
8297async fn wait_for_slot_registration(
8298    registry: &Registry,
8299    slot: crate::registry::RegistrationSlot<'_>,
8300    module_id: &str,
8301    snapshot: &SharedSnapshot,
8302    child: &mut SupervisedChild,
8303    wait: Duration,
8304) -> Result<RegistrationWaitOutcome, SuperviseError> {
8305    let deadline = Instant::now() + wait;
8306    loop {
8307        if registry
8308            .registration(slot)
8309            .map_err(SuperviseError::Registry)?
8310            .is_some()
8311        {
8312            return Ok(RegistrationWaitOutcome::Registered);
8313        }
8314
8315        let now = Instant::now();
8316        if now >= deadline {
8317            return Ok(RegistrationWaitOutcome::TimedOut);
8318        }
8319        let remaining = deadline.saturating_duration_since(now);
8320        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8321
8322        tokio::select! {
8323            wait_result = child.wait() => {
8324                let status = wait_result.map_err(|source| SuperviseError::Wait {
8325                    module_id: module_id.to_string(),
8326                    source,
8327                })?;
8328                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8329                    snapshot,
8330                    child,
8331                    &status,
8332                )));
8333            }
8334            _ = sleep(poll) => {}
8335        }
8336    }
8337}
8338
8339fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8340    // A replacement process that exits before HELLO did not provide service, even
8341    // if it used status 0. Count it against the restart cap as a new-binary failure.
8342    if exit_report.kind != ExitKind::DeliberateSeverance {
8343        exit_report.kind = ExitKind::Crash;
8344    }
8345    exit_report
8346}
8347
8348async fn handle_reload_child_registration_failure(
8349    spec: &ModuleSpec,
8350    runtime: &SupervisorRuntimeConfig,
8351    registry: &Registry,
8352    process_liveness: &SupervisorProcessLiveness,
8353    snapshot: &SharedSnapshot,
8354    _child: &mut Option<SupervisedChild>,
8355    failure: ReloadRegistrationFailure,
8356) -> Result<(), SuperviseError> {
8357    let ReloadRegistrationFailure {
8358        exit_report,
8359        reason,
8360    } = failure;
8361    match on_child_exit(
8362        spec,
8363        runtime.restart_policy,
8364        registry,
8365        snapshot,
8366        &runtime.terminal_ring,
8367        &runtime.spawn_events,
8368        &runtime.child_roster,
8369        exit_report,
8370    )
8371    .await
8372    {
8373        NextAction::Stop {
8374            registration_released,
8375        } => {
8376            if registration_released {
8377                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8378            }
8379        }
8380        NextAction::Restart { schedule } => {
8381            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8382                schedule.delay
8383            });
8384            if let Some(schedule) = schedule {
8385                log_crash_respawn(&spec.module_id, schedule);
8386            }
8387            schedule_respawn(
8388                runtime,
8389                snapshot,
8390                &spec.module_id,
8391                delay,
8392                RespawnKind::Spawn,
8393            )?;
8394        }
8395    }
8396    Err(SuperviseError::ReloadFailed {
8397        module_id: spec.module_id.clone(),
8398        reason,
8399    })
8400}
8401
8402async fn handle_reload_spawn_failure(
8403    spec: &ModuleSpec,
8404    runtime: &SupervisorRuntimeConfig,
8405    process_liveness: &SupervisorProcessLiveness,
8406    snapshot: &SharedSnapshot,
8407    _child: &mut Option<SupervisedChild>,
8408    reason: String,
8409) -> Result<(), SuperviseError> {
8410    let now = Instant::now();
8411    let mut schedule = None;
8412    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8413        clear_current_process_facts(state);
8414        if state.enabled {
8415            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8416            state.state = if schedule.is_some() {
8417                ModuleState::Restarting
8418            } else {
8419                ModuleState::Failed
8420            };
8421        } else {
8422            state.state = ModuleState::Disabled;
8423        }
8424    })?;
8425    if let Some(schedule) = schedule {
8426        schedule_respawn(
8427            runtime,
8428            snapshot,
8429            &spec.module_id,
8430            schedule.delay,
8431            RespawnKind::Spawn,
8432        )?;
8433    } else {
8434        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8435    }
8436    Err(SuperviseError::ReloadFailed {
8437        module_id: spec.module_id.clone(),
8438        reason,
8439    })
8440}
8441
8442fn control_flags() -> Flags {
8443    Flags::new(false, Priority::Passive, false)
8444}
8445
8446#[allow(clippy::too_many_arguments)]
8447async fn drain_optional_child(
8448    module_id: &str,
8449    protocol: ModuleProtocol,
8450    stop_notice: StopNotice,
8451    registry: &Registry,
8452    forwarding: Option<&ForwardingTable>,
8453    snapshot: &SharedSnapshot,
8454    terminal_ring: &Arc<Mutex<TerminalRing>>,
8455    spawn_events: &SpawnEventFeed,
8456    child: &mut Option<SupervisedChild>,
8457    drain_timeout: Duration,
8458    final_state: ModuleState,
8459    enabled: Option<bool>,
8460) -> Result<(), SuperviseError> {
8461    if let Some(child) = child.take() {
8462        drain_child_to_state(
8463            module_id,
8464            protocol,
8465            stop_notice,
8466            registry,
8467            forwarding,
8468            snapshot,
8469            terminal_ring,
8470            spawn_events,
8471            child,
8472            drain_timeout,
8473            final_state,
8474            enabled,
8475        )
8476        .await
8477    } else {
8478        update_snapshot(snapshot, Some(module_id), |state| {
8479            state.state = final_state;
8480            if let Some(enabled) = enabled {
8481                state.enabled = enabled;
8482            }
8483            clear_current_process_facts(state);
8484        })?;
8485        release_dead_registration(registry, forwarding, snapshot, module_id).await
8486    }
8487}
8488
8489#[allow(clippy::too_many_arguments)]
8490async fn drain_child_to_state(
8491    module_id: &str,
8492    _protocol: ModuleProtocol,
8493    stop_notice: StopNotice,
8494    registry: &Registry,
8495    forwarding: Option<&ForwardingTable>,
8496    snapshot: &SharedSnapshot,
8497    terminal_ring: &Arc<Mutex<TerminalRing>>,
8498    spawn_events: &SpawnEventFeed,
8499    mut child: SupervisedChild,
8500    drain_timeout: Duration,
8501    final_state: ModuleState,
8502    enabled: Option<bool>,
8503) -> Result<(), SuperviseError> {
8504    let protocol = child.protocol;
8505    update_snapshot(snapshot, Some(module_id), |state| {
8506        state.state = ModuleState::Draining;
8507        state.draining_to_replace = final_state == ModuleState::Restarting;
8508        if let Some(enabled) = enabled {
8509            state.enabled = enabled;
8510        }
8511    })?;
8512
8513    // The wait below is the same budget in every case; what differs is
8514    // whether anything has ASKED the child to stop before it starts. Only a
8515    // forwarding drain that reached the module's registered connection has
8516    // (`module.draining`, then a module GOODBYE). Every other child was told
8517    // nothing: a `protocol: "none"` module, which never registers; a subc
8518    // module spawned moments ago that has not sent HELLO yet; or a stop that
8519    // runs no forwarding drain. Without a signal the budget is only a delay
8520    // in front of SIGKILL -- and the not-yet-registered child is the worst
8521    // case, because it registers into a module that is already draining,
8522    // is never told, and is killed while healthy.
8523    if stop_notice != StopNotice::SentOverConnection {
8524        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8525            info!(
8526                module_id,
8527                pid = child.pid,
8528                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8529                "module has no connection yet; requesting stop by signal"
8530            );
8531        }
8532        request_graceful_stop(module_id, &child);
8533    }
8534
8535    let exit_report = match timeout(drain_timeout, child.wait()).await {
8536        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
8537        Ok(Err(source)) => {
8538            fail_snapshot(snapshot, Some(module_id), None);
8539            return Err(SuperviseError::Wait {
8540                module_id: module_id.to_string(),
8541                source,
8542            });
8543        }
8544        Err(_) => {
8545            // Mirror the sibling arm above: state is already `Draining`, and an
8546            // error propagated from here would strand it there -- a state
8547            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
8548            // `Failed | Stopped`), leaving an operator Restart as the only exit.
8549            // `Failed` before `?` keeps the module operator-visible and
8550            // revivable. Trigger is an ESRCH race (process exits between the
8551            // drain timeout firing and the kill) or a post-kill wait failure
8552            // (issue #34).
8553            //
8554            // Logged because the kill is otherwise visible only as signal 9 in
8555            // the terminal ring, and the budget it follows can be long enough
8556            // that consumers see a stretch of refusals with no stated cause.
8557            warn!(
8558                module_id,
8559                pid = child.pid,
8560                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8561                reason = ?final_state,
8562                ?stop_notice,
8563                "drain budget expired before the module exited; killing it"
8564            );
8565            child.start_kill().map_err(|source| {
8566                fail_snapshot(snapshot, Some(module_id), None);
8567                SuperviseError::Kill {
8568                    module_id: module_id.to_string(),
8569                    source,
8570                }
8571            })?;
8572            let status = child.wait().await.map_err(|source| {
8573                fail_snapshot(snapshot, Some(module_id), None);
8574                SuperviseError::Wait {
8575                    module_id: module_id.to_string(),
8576                    source,
8577                }
8578            })?;
8579            classify_reaped_child_exit(snapshot, &child, &status)
8580        }
8581    };
8582
8583    update_snapshot(snapshot, Some(module_id), |state| {
8584        state.state = final_state;
8585        if let Some(enabled) = enabled {
8586            state.enabled = enabled;
8587        }
8588        clear_current_process_facts(state);
8589        state.last_exit = Some(exit_report.clone());
8590        if exit_report.kind == ExitKind::DeliberateSeverance {
8591            state.lifetime_restarts += 1;
8592        }
8593    })?;
8594    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
8595    record_terminal_with_detail(
8596        module_id,
8597        terminal_ring,
8598        spawn_events,
8599        &exit_report,
8600        terminal_disposition(final_state),
8601        detail,
8602    );
8603    child.drain_stderr(module_id).await;
8604
8605    release_dead_registration(registry, forwarding, snapshot, module_id).await
8606}
8607
8608/// Ask a child that nothing else has asked to stop, by signal.
8609///
8610/// A registered subc module is asked over its own connection: the drain sends
8611/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
8612/// module GOODBYE, and the module stops itself. A module that speaks no subc
8613/// wire receives none of that, and neither does a subc module that has not
8614/// registered yet, so for them the drain budget would be pure delay in front of
8615/// a SIGKILL -- and for a process with a store to flush (JetStream is the
8616/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
8617/// into a recovery on the next start.
8618///
8619/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
8620/// rule rather than an optimisation: that module's graceful stop is already
8621/// running by the time its child is drained, and a signal would race it.
8622///
8623/// Best-effort by construction. A child that has already exited is the ordinary
8624/// case rather than an error (the kill lands on a reaped or exiting pid), so a
8625/// failure is logged at debug and the wait-then-kill below still decides the
8626/// outcome.
8627#[cfg(unix)]
8628fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
8629    let Some(pid) = child
8630        .id()
8631        .and_then(|pid| i32::try_from(pid).ok())
8632        .and_then(rustix::process::Pid::from_raw)
8633    else {
8634        debug!(
8635            module_id,
8636            "no pid to signal for teardown; falling through to the drain wait"
8637        );
8638        return;
8639    };
8640    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
8641        Ok(()) => debug!(
8642            module_id,
8643            "sent SIGTERM to a module nothing else asked to stop"
8644        ),
8645        Err(err) => debug!(
8646            module_id,
8647            error = %err,
8648            "SIGTERM to module failed; the drain wait and kill still apply"
8649        ),
8650    }
8651}
8652
8653/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
8654/// Windows does offer need cooperation this supervisor cannot assume: a console
8655/// control event requires sharing a console with the child, and `WM_CLOSE`
8656/// requires the child to pump a message loop. A supervised server process does
8657/// neither, so there is nothing to send and teardown is the wait followed by the
8658/// kill. Emulating a signal here would mean inventing a stop protocol, which is
8659/// the thing `protocol: "none"` exists to avoid.
8660#[cfg(not(unix))]
8661fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
8662    debug!(
8663        module_id,
8664        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
8665    );
8666}
8667
8668fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
8669    match final_state {
8670        ModuleState::Stopped => TerminalDisposition::Stopped,
8671        ModuleState::Disabled => TerminalDisposition::Disabled,
8672        ModuleState::Restarting => TerminalDisposition::Restarting,
8673        ModuleState::Failed => TerminalDisposition::Failed,
8674        ModuleState::Starting
8675        | ModuleState::Running
8676        | ModuleState::Unresponsive
8677        | ModuleState::Draining => {
8678            unreachable!("terminal exits only finish in terminal or restarting states")
8679        }
8680    }
8681}
8682
8683/// Release a reaped child's registration before allowing another spawn.
8684///
8685/// EOF is not a process-lifetime signal: an inherited socket can stay open
8686/// indefinitely, and serial frame dispatch can be waiting on egress instead of
8687/// reading EOF. After the normal release grace, request connection close (which
8688/// cancels both reads and dispatch), then allow one more release grace for the
8689/// connection guard's forwarding cleanup. Never evict a different connection.
8690async fn release_dead_registration(
8691    registry: &Registry,
8692    forwarding: Option<&ForwardingTable>,
8693    snapshot: &SharedSnapshot,
8694    module_id: &str,
8695) -> Result<(), SuperviseError> {
8696    let result = async {
8697        let registration = registry
8698            .get_module(module_id)
8699            .map_err(SuperviseError::Registry)?;
8700        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
8701            Ok(()) => return Ok(()),
8702            Err(SuperviseError::RegistrationStillActive { .. }) => {}
8703            Err(err) => return Err(err),
8704        }
8705        let pid = lock_snapshot(snapshot)?.reaped_pid;
8706        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
8707            warn!(
8708                module_id,
8709                pid,
8710                connection_id = registration.connection_id.get(),
8711                "reaped module registration outlived release grace; closing dead connection"
8712            );
8713            forwarding.request_connection_close(
8714                registration.connection_id,
8715                CloseReason::new(
8716                    "supervised_process_reaped",
8717                    format!("module '{module_id}' pid {pid} exited"),
8718                ),
8719            );
8720            wait_for_slot_registration_release(
8721                registry,
8722                crate::registry::RegistrationSlot::Connection(registration.connection_id),
8723                REGISTRY_RELEASE_TIMEOUT,
8724            )
8725            .await?;
8726        }
8727        wait_for_registration_release(registry, module_id, Duration::ZERO).await
8728    }
8729    .await;
8730    if let Err(err) = &result {
8731        fail_snapshot(snapshot, Some(module_id), None);
8732        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
8733    }
8734    result
8735}
8736
8737/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
8738/// plain stop or restart waits for before it spawns a replacement.
8739async fn wait_for_registration_release(
8740    registry: &Registry,
8741    module_id: &str,
8742    wait: Duration,
8743) -> Result<(), SuperviseError> {
8744    wait_for_slot_registration_release(
8745        registry,
8746        crate::registry::RegistrationSlot::Active(module_id),
8747        wait,
8748    )
8749    .await
8750}
8751
8752/// Wait for the registration in `slot` to go away.
8753///
8754/// Keyed on the slot rather than the bare module id because a successful swap
8755/// never empties the id's active slot (the promoted candidate is in it), so an
8756/// id-keyed wait for the incumbent's release would always time out. Draining a
8757/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
8758/// incumbent's connection instead.
8759async fn wait_for_slot_registration_release(
8760    registry: &Registry,
8761    slot: crate::registry::RegistrationSlot<'_>,
8762    wait: Duration,
8763) -> Result<(), SuperviseError> {
8764    let deadline = Instant::now() + wait;
8765    let mut release_events = registration_release_events().subscribe();
8766    let still_active = |registration: &crate::registry::ModuleRegistration| {
8767        SuperviseError::RegistrationStillActive {
8768            module_id: registration.manifest.module_id.clone(),
8769            waited: wait,
8770        }
8771    };
8772    loop {
8773        let _observed_generation = *release_events.borrow_and_update();
8774        let Some(registration) = registry
8775            .registration(slot)
8776            .map_err(SuperviseError::Registry)?
8777        else {
8778            return Ok(());
8779        };
8780
8781        let now = Instant::now();
8782        if now >= deadline {
8783            return Err(still_active(&registration));
8784        }
8785
8786        let remaining = deadline.saturating_duration_since(now);
8787        match timeout(remaining, release_events.changed()).await {
8788            Ok(Ok(())) | Ok(Err(_)) => {}
8789            Err(_) => return Err(still_active(&registration)),
8790        }
8791    }
8792}
8793
8794#[cfg(test)]
8795mod slot_registration_wait_tests {
8796    use super::*;
8797    use crate::registry::{ConnectionId, RegistrationSlot};
8798    use subc_protocol::manifest::ModuleManifest;
8799
8800    #[tokio::test]
8801    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
8802        let registry = Arc::new(Registry::default());
8803        let supervisor = Supervisor::new(Arc::clone(&registry), RestartPolicy::default());
8804        let runtime = supervisor.runtime_config();
8805        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
8806        let spec = ModuleSpec {
8807            module_id: "enable-stale-registration".to_string(),
8808            program: PathBuf::from("/missing/enable-retry-test"),
8809            args: Vec::new(),
8810            env: Vec::new(),
8811            reserved: false,
8812            reserved_prefixes: Vec::new(),
8813            protocol: ModuleProtocol::Subc,
8814            overlap: Default::default(),
8815        };
8816        let connection = ConnectionId::new(90);
8817        registry
8818            .register_with_control_ops(
8819                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
8820                1,
8821                connection,
8822                Vec::new(),
8823            )
8824            .unwrap();
8825        let mut child = None;
8826        let err = set_child_enabled(
8827            &spec,
8828            &runtime,
8829            &registry,
8830            &supervisor.process_liveness,
8831            &snapshot,
8832            &mut child,
8833            true,
8834        )
8835        .await
8836        .unwrap_err();
8837        assert!(matches!(
8838            err,
8839            SuperviseError::RegistrationStillActive { .. }
8840        ));
8841        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
8842        assert!(child.is_none());
8843        registry.deregister_connection(connection).unwrap();
8844        let err = set_child_enabled(
8845            &spec,
8846            &runtime,
8847            &registry,
8848            &supervisor.process_liveness,
8849            &snapshot,
8850            &mut child,
8851            true,
8852        )
8853        .await
8854        .unwrap_err();
8855        assert!(
8856            matches!(err, SuperviseError::Spawn { .. }),
8857            "second enable must attempt a spawn: {err}"
8858        );
8859        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
8860    }
8861
8862    const INCUMBENT: u64 = 1;
8863    const CANDIDATE: u64 = 2;
8864
8865    fn swapped_registry() -> Arc<Registry> {
8866        let registry = Arc::new(Registry::default());
8867        let manifest = ModuleManifest::builder("m", "0.1.0").build();
8868        registry
8869            .register_with_control_ops(
8870                manifest.clone(),
8871                1,
8872                ConnectionId::new(INCUMBENT),
8873                Vec::new(),
8874            )
8875            .unwrap();
8876        registry
8877            .register_candidate_with_control_ops(
8878                manifest,
8879                1,
8880                ConnectionId::new(CANDIDATE),
8881                Vec::new(),
8882            )
8883            .unwrap();
8884        registry
8885    }
8886
8887    /// After a promotion the id's active slot is held by the new process, so an
8888    /// id-keyed wait for the incumbent's release can never succeed; the
8889    /// connection-keyed wait completes as soon as the incumbent deregisters.
8890    #[tokio::test]
8891    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
8892        let registry = swapped_registry();
8893        registry.promote_candidate("m").unwrap().unwrap();
8894
8895        assert!(matches!(
8896            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
8897            Err(SuperviseError::RegistrationStillActive { .. })
8898        ));
8899
8900        // Still held while the incumbent's connection has not deregistered.
8901        assert!(matches!(
8902            wait_for_slot_registration_release(
8903                &registry,
8904                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
8905                Duration::from_millis(50),
8906            )
8907            .await,
8908            Err(SuperviseError::RegistrationStillActive { .. })
8909        ));
8910
8911        let releaser = Arc::clone(&registry);
8912        let release = tokio::spawn(async move {
8913            sleep(Duration::from_millis(20)).await;
8914            releaser
8915                .deregister_connection(ConnectionId::new(INCUMBENT))
8916                .unwrap();
8917            notify_registration_release();
8918        });
8919        wait_for_slot_registration_release(
8920            &registry,
8921            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
8922            Duration::from_secs(5),
8923        )
8924        .await
8925        .expect("the incumbent's own registration is released");
8926        release.await.unwrap();
8927        assert!(registry.get_module("m").unwrap().is_some());
8928    }
8929
8930    /// The candidate slot is waited on separately from the active slot: the
8931    /// incumbent's registration neither holds up nor stands in for it.
8932    #[tokio::test]
8933    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
8934        let registry = swapped_registry();
8935        assert!(matches!(
8936            wait_for_slot_registration_release(
8937                &registry,
8938                RegistrationSlot::Candidate("m"),
8939                Duration::from_millis(50),
8940            )
8941            .await,
8942            Err(SuperviseError::RegistrationStillActive { .. })
8943        ));
8944        registry
8945            .deregister_connection(ConnectionId::new(CANDIDATE))
8946            .unwrap();
8947        wait_for_slot_registration_release(
8948            &registry,
8949            RegistrationSlot::Candidate("m"),
8950            Duration::from_millis(50),
8951        )
8952        .await
8953        .expect("a candidate slot with no candidate is released");
8954        assert!(registry
8955            .registration(RegistrationSlot::Active("m"))
8956            .unwrap()
8957            .is_some());
8958    }
8959}
8960
8961fn classify_exit(status: &ExitStatus) -> ExitReport {
8962    ExitReport {
8963        kind: if status.success() {
8964            ExitKind::Clean
8965        } else {
8966            ExitKind::Crash
8967        },
8968        code: status.code(),
8969        signal: exit_signal(status),
8970        at_ms: unix_ms_now(),
8971    }
8972}
8973
8974/// The terminal record for a module whose `wait()` call itself errored (e.g. the
8975/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
8976/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
8977/// disposition still must be `Failed` so the terminal ring is not silently missing
8978/// an entry, matching what `fail_snapshot` records for this same arm.
8979fn wait_error_exit_report() -> ExitReport {
8980    ExitReport {
8981        kind: ExitKind::Crash,
8982        code: None,
8983        signal: None,
8984        at_ms: unix_ms_now(),
8985    }
8986}
8987
8988#[cfg(unix)]
8989fn exit_signal(status: &ExitStatus) -> Option<i32> {
8990    use std::os::unix::process::ExitStatusExt;
8991
8992    status.signal()
8993}
8994
8995#[cfg(not(unix))]
8996fn exit_signal(_status: &ExitStatus) -> Option<i32> {
8997    None
8998}
8999
9000/// Give an operator-touched module its full crash budget back.
9001///
9002/// Named for the counter it used to zero; it now empties the in-window ring,
9003/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9004/// ledger of what happened survives every operator action.
9005fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9006    update_snapshot(snapshot, Some(module_id), |state| {
9007        state.clear_crash_restarts();
9008    })
9009}
9010
9011fn set_running(
9012    snapshot: &SharedSnapshot,
9013    child: &SupervisedChild,
9014    module_id: &str,
9015    spawn_events: &SpawnEventFeed,
9016) -> Result<(), SuperviseError> {
9017    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9018        module_id: Some(module_id.to_string()),
9019    })?;
9020    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9021    if std::mem::take(&mut state.coalesced_restart_pending) {
9022        let generation = state.spawn_generation;
9023        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9024    }
9025    state.drain_disposition_detail = None;
9026    // Every caller of this is a plain spawn, which always uses the primary key;
9027    // a promoted swap candidate sets the flag itself after this returns.
9028    state.in_alternate_slot = false;
9029    state.configuration_updated_since_spawn = false;
9030    state.spawned_protocol = Some(child.protocol);
9031    state.state = ModuleState::Running;
9032    state.enabled = true;
9033    state.process_alive = true;
9034    state.pid = child.id();
9035    state.spawned_at_ms = Some(child.spawned_at_ms);
9036    state.spawned_from = Some(child.spawned_from.clone());
9037    state.spawned_file_identity = child.spawned_file_identity;
9038    state.process_start_time = child.process_start_time;
9039    Ok(())
9040}
9041
9042fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9043    state.process_alive = false;
9044    state.spawned_protocol = None;
9045    state.pid = None;
9046    state.spawned_at_ms = None;
9047    state.spawned_from = None;
9048    state.spawned_file_identity = None;
9049    state.process_start_time = None;
9050    state.deliberate_severance = None;
9051}
9052
9053#[cfg(test)]
9054fn record_deliberate_severance(
9055    snapshot: &SharedSnapshot,
9056    identity: ProcessIdentity,
9057) -> Result<(), SuperviseError> {
9058    update_snapshot(snapshot, None, |state| {
9059        state.deliberate_severance = Some(identity);
9060    })
9061}
9062
9063fn apply_deliberate_severance_marker(
9064    snapshot: &SharedSnapshot,
9065    exited_identity: Option<ProcessIdentity>,
9066    mut exit_report: ExitReport,
9067) -> ExitReport {
9068    let marker = lock_snapshot(snapshot)
9069        .ok()
9070        .and_then(|mut state| state.deliberate_severance.take());
9071    if marker.is_some() && marker == exited_identity {
9072        exit_report.kind = ExitKind::DeliberateSeverance;
9073    }
9074    exit_report
9075}
9076
9077fn classify_reaped_child_exit(
9078    snapshot: &SharedSnapshot,
9079    child: &SupervisedChild,
9080    status: &ExitStatus,
9081) -> ExitReport {
9082    let _ = update_snapshot(snapshot, None, |state| state.reaped_pid = Some(child.pid));
9083    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9084}
9085
9086fn fail_snapshot(
9087    snapshot: &SharedSnapshot,
9088    module_id: Option<&str>,
9089    last_exit: Option<ExitReport>,
9090) {
9091    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9092        state.state = ModuleState::Failed;
9093        clear_current_process_facts(state);
9094        if let Some(last_exit) = last_exit {
9095            state.last_exit = Some(last_exit);
9096        }
9097    }) {
9098        error!(error = %err, "failed to mark supervisor state failed");
9099    }
9100}
9101
9102fn update_snapshot(
9103    snapshot: &SharedSnapshot,
9104    module_id: Option<&str>,
9105    update: impl FnOnce(&mut SupervisorSnapshot),
9106) -> Result<(), SuperviseError> {
9107    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9108        module_id: module_id.map(ToOwned::to_owned),
9109    })?;
9110    update(&mut state);
9111    Ok(())
9112}
9113
9114const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9115
9116fn lock_snapshot_for_control<'a>(
9117    snapshot: &'a SharedSnapshot,
9118    module_id: &str,
9119    caller: &'static str,
9120) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9121    let started_at = Instant::now();
9122    let guard = lock_snapshot(snapshot)?;
9123    let waited = started_at.elapsed();
9124    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9125        warn!(
9126            module_id = %module_id,
9127            waited_ms = waited.as_millis() as u64,
9128            caller = %caller,
9129            "slow snapshot lock"
9130        );
9131    }
9132    Ok(guard)
9133}
9134
9135fn lock_snapshot(
9136    snapshot: &SharedSnapshot,
9137) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9138    snapshot
9139        .lock()
9140        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9141}
9142
9143#[cfg(test)]
9144mod terminal_history_tests {
9145    use std::{
9146        path::PathBuf,
9147        sync::Arc,
9148        time::{Duration, Instant},
9149    };
9150
9151    use tokio::time::sleep;
9152
9153    use super::{
9154        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9155        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9156        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9157        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9158        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9159        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
9160        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
9161    };
9162    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9163    // use for their own wall-clock deadlines: crash-restart instants must be on
9164    // the same clock the production code stamps them with, which is tokio's (and
9165    // is what `start_paused` tests can move).
9166    use super::Instant as ClockInstant;
9167    use crate::{
9168        registry::Registry,
9169        terminal_ring::{TerminalRing, TerminalRingConfig},
9170    };
9171    use std::sync::Mutex;
9172    use subc_control::TerminalDisposition;
9173
9174    /// See the twin in `control.rs` for why this derives the path from
9175    /// `current_exe()` and why the existence check is here: `--lib` alone does
9176    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9177    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9178    pub(super) fn fake_aft_stub_path() -> PathBuf {
9179        let mut path = std::env::current_exe().expect("current_exe available in tests");
9180        path.pop();
9181        path.pop();
9182        path.push(if cfg!(windows) {
9183            "fake-aft-stub.exe"
9184        } else {
9185            "fake-aft-stub"
9186        });
9187        assert!(
9188            path.exists(),
9189            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9190             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9191            path.display()
9192        );
9193        path
9194    }
9195
9196    #[test]
9197    fn reserved_never_spawned_refuses_every_hello() {
9198        // The canary hole: a reserved id whose module has never spawned had NO
9199        // gate entry and admitted anyone -- the reservation protected the nonce
9200        // holder, not the NAME. Now the entry is present with no legitimate
9201        // holder and refuses all comers.
9202        let supervisor = SupervisorHandle::default();
9203        supervisor.apply_identity_configuration(&ModuleSpec {
9204            module_id: "never-spawned".to_string(),
9205            program: PathBuf::from("/usr/bin/false"),
9206            args: Vec::new(),
9207            env: Vec::new(),
9208            reserved: true,
9209            reserved_prefixes: Vec::new(),
9210            protocol: ModuleProtocol::Subc,
9211            overlap: Default::default(),
9212        });
9213        assert!(
9214            supervisor
9215                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9216                .is_some(),
9217            "forged nonce must refuse on a reserved never-spawned id"
9218        );
9219        assert!(
9220            supervisor
9221                .reserved_hello_rejection("never-spawned", None)
9222                .is_some(),
9223            "absent nonce must refuse on a reserved never-spawned id"
9224        );
9225        // And a real spawn nonce minted later admits exactly that nonce.
9226        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9227        supervisor.apply_identity_configuration(&ModuleSpec {
9228            module_id: "never-spawned".to_string(),
9229            program: PathBuf::from("/usr/bin/false"),
9230            args: Vec::new(),
9231            env: Vec::new(),
9232            reserved: true,
9233            reserved_prefixes: Vec::new(),
9234            protocol: ModuleProtocol::Subc,
9235            overlap: Default::default(),
9236        });
9237        assert!(supervisor
9238            .reserved_hello_rejection("never-spawned", Some("minted"))
9239            .is_none());
9240        assert!(supervisor
9241            .reserved_hello_rejection("never-spawned", Some("forged"))
9242            .is_some());
9243    }
9244
9245    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9246    /// happened, which is what "spent budget" looks like to every reader.
9247    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9248        let now = ClockInstant::now();
9249        for _ in 0..count {
9250            state.crash_restarts.push_back(now);
9251        }
9252    }
9253
9254    /// Age the oldest recorded restart out of `window`, standing in for the hours
9255    /// that would otherwise have to pass. Injecting the instant is the point: a
9256    /// test that slept a real window would take ten minutes and still prove less.
9257    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9258        let aged = state
9259            .crash_restarts
9260            .front()
9261            .expect("a crash restart must be recorded before it can be aged")
9262            .checked_sub(window + Duration::from_secs(1))
9263            .expect("the test clock is far enough from its origin to age an instant");
9264        state.crash_restarts[0] = aged;
9265    }
9266
9267    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9268        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9269        seed_crash_restarts(&mut state, count);
9270        state
9271    }
9272
9273    #[test]
9274    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9275        let policy = RestartPolicy::new(3, Duration::ZERO);
9276        let now = ClockInstant::now();
9277        assert!(daemon_will_restart(
9278            &mut snapshot_with_restarts(true, 2),
9279            &policy,
9280            now
9281        ));
9282        assert!(!daemon_will_restart(
9283            &mut snapshot_with_restarts(true, 3),
9284            &policy,
9285            now
9286        ));
9287        assert!(!daemon_will_restart(
9288            &mut snapshot_with_restarts(false, 0),
9289            &policy,
9290            now
9291        ));
9292    }
9293
9294    #[test]
9295    fn crash_restart_backoff_escalates_with_in_window_count() {
9296        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9297            .with_max_backoff(Duration::from_secs(30));
9298        let now = ClockInstant::now();
9299        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9300        let schedules = (0..4)
9301            .map(|_| {
9302                state
9303                    .next_crash_restart(&policy, now)
9304                    .expect("the test policy allows four crash restarts")
9305            })
9306            .collect::<Vec<_>>();
9307
9308        assert_eq!(
9309            schedules
9310                .iter()
9311                .map(|schedule| schedule.restart_in_window)
9312                .collect::<Vec<_>>(),
9313            vec![0, 1, 2, 3]
9314        );
9315        assert_eq!(
9316            schedules
9317                .iter()
9318                .map(|schedule| schedule.delay)
9319                .collect::<Vec<_>>(),
9320            vec![
9321                Duration::from_millis(100),
9322                Duration::from_secs(1),
9323                Duration::from_secs(10),
9324                Duration::from_secs(30),
9325            ]
9326        );
9327    }
9328
9329    #[test]
9330    fn crash_restart_backoff_resets_after_ring_clear() {
9331        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9332        let now = ClockInstant::now();
9333        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9334        assert_eq!(
9335            state.next_crash_restart(&policy, now).unwrap().delay,
9336            Duration::from_millis(100)
9337        );
9338        assert_eq!(
9339            state.next_crash_restart(&policy, now).unwrap().delay,
9340            Duration::from_secs(1)
9341        );
9342
9343        state.clear_crash_restarts();
9344        let schedule = state
9345            .next_crash_restart(&policy, now)
9346            .expect("a cleared ring must allow another restart");
9347        assert_eq!(schedule.restart_in_window, 0);
9348        assert_eq!(schedule.delay, Duration::from_millis(100));
9349    }
9350
9351    #[test]
9352    fn crash_restart_backoff_ignores_aged_restarts() {
9353        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9354        let now = ClockInstant::now();
9355        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9356        state
9357            .next_crash_restart(&policy, now)
9358            .expect("the first restart is allowed");
9359        state
9360            .next_crash_restart(&policy, now)
9361            .expect("the second restart is allowed");
9362        state.crash_restarts[0] = now
9363            .checked_sub(policy.window + Duration::from_secs(1))
9364            .expect("the fake clock can age a restart past the window");
9365
9366        let schedule = state
9367            .next_crash_restart(&policy, now)
9368            .expect("an aged restart must release its slot");
9369        assert_eq!(schedule.restart_in_window, 1);
9370        assert_eq!(schedule.delay, Duration::from_secs(1));
9371        assert_eq!(state.crash_restarts.len(), 2);
9372    }
9373
9374    /// The budget is a rate: the same three spent restarts refuse a respawn
9375    /// while they are recent and allow one once they have aged past the window.
9376    /// Nothing about the module changed in between, which is the whole point.
9377    #[test]
9378    fn a_budget_spent_before_the_window_no_longer_refuses() {
9379        let policy = RestartPolicy::new(3, Duration::ZERO);
9380        let mut state = snapshot_with_restarts(true, 3);
9381        let now = ClockInstant::now();
9382        assert!(!daemon_will_restart(&mut state, &policy, now));
9383
9384        assert!(daemon_will_restart(
9385            &mut state,
9386            &policy,
9387            now + policy.window + Duration::from_secs(1)
9388        ));
9389        assert!(
9390            state.crash_restarts.is_empty(),
9391            "reading the budget must drop the instants that left the window"
9392        );
9393    }
9394
9395    fn module_with_recovery_snapshot(
9396        state: ModuleState,
9397        enabled: bool,
9398        restart_count: u32,
9399    ) -> SupervisedModule {
9400        let registry = Arc::new(Registry::default());
9401        let supervisor =
9402            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9403        let module = supervisor
9404            .spawn(ModuleSpec {
9405                module_id: "recovery-snapshot".to_string(),
9406                program: fake_aft_stub_path(),
9407                args: Vec::new(),
9408                env: Vec::new(),
9409                reserved: false,
9410                reserved_prefixes: Vec::new(),
9411                protocol: ModuleProtocol::Subc,
9412                overlap: Default::default(),
9413            })
9414            .unwrap();
9415        update_snapshot(
9416            &module.inner.snapshot,
9417            Some("recovery-snapshot"),
9418            |snapshot| {
9419                snapshot.state = state;
9420                snapshot.enabled = enabled;
9421                seed_crash_restarts(snapshot, restart_count);
9422            },
9423        )
9424        .unwrap();
9425        module
9426    }
9427
9428    #[cfg(target_os = "linux")]
9429    #[tokio::test]
9430    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9431        let supervisor = Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
9432            .with_cgroup_placement(None);
9433        let result = supervisor.spawn(ModuleSpec {
9434            module_id: "no-cgroup-placement".to_string(),
9435            program: fake_aft_stub_path(),
9436            args: Vec::new(),
9437            env: Vec::new(),
9438            reserved: false,
9439            reserved_prefixes: Vec::new(),
9440            protocol: ModuleProtocol::Subc,
9441            overlap: Default::default(),
9442        });
9443
9444        assert!(
9445            result.is_ok(),
9446            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9447        );
9448    }
9449
9450    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9451    async fn undecided_snapshot_uses_shared_restart_predicate() {
9452        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9453            .will_recover_after_connection_loss()
9454            .unwrap());
9455        assert!(
9456            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9457                .will_recover_after_connection_loss()
9458                .unwrap()
9459        );
9460    }
9461
9462    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9463    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9464        assert!(
9465            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9466                .will_recover_after_connection_loss()
9467                .unwrap()
9468        );
9469    }
9470
9471    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9472    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9473        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9474            .will_recover_after_connection_loss()
9475            .unwrap());
9476        assert!(
9477            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9478                .will_recover_after_connection_loss()
9479                .unwrap()
9480        );
9481    }
9482
9483    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9484    async fn warming_snapshot_is_limited_to_startup_phases() {
9485        for state in [
9486            ModuleState::Starting,
9487            ModuleState::Running,
9488            ModuleState::Restarting,
9489        ] {
9490            assert!(
9491                module_with_recovery_snapshot(state, true, 0)
9492                    .is_warming()
9493                    .unwrap(),
9494                "{state:?} should be warming"
9495            );
9496        }
9497        for state in [
9498            ModuleState::Unresponsive,
9499            ModuleState::Draining,
9500            ModuleState::Stopped,
9501            ModuleState::Failed,
9502            ModuleState::Disabled,
9503        ] {
9504            assert!(
9505                !module_with_recovery_snapshot(state, true, 0)
9506                    .is_warming()
9507                    .unwrap(),
9508                "{state:?} should not be warming"
9509            );
9510        }
9511    }
9512
9513    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9514    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
9515        let registry = Arc::new(Registry::default());
9516        let supervisor =
9517            Supervisor::new(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
9518        let module = supervisor
9519            .spawn(ModuleSpec {
9520                module_id: "terminal-history".to_string(),
9521                program: fake_aft_stub_path(),
9522                args: Vec::new(),
9523                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9524                reserved: false,
9525                reserved_prefixes: Vec::new(),
9526                protocol: ModuleProtocol::Subc,
9527                overlap: Default::default(),
9528            })
9529            .unwrap();
9530
9531        let deadline = Instant::now() + Duration::from_secs(5);
9532        loop {
9533            let history = module.terminal_history();
9534            if history.entries.len() == 2 {
9535                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
9536                assert_eq!(history.dropped, 0);
9537                assert_eq!(
9538                    history
9539                        .entries
9540                        .iter()
9541                        .map(|entry| entry.exit_code)
9542                        .collect::<Vec<_>>(),
9543                    vec![Some(23), Some(23)]
9544                );
9545                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
9546                return;
9547            }
9548            assert!(
9549                Instant::now() < deadline,
9550                "module did not retain two terminal exits: {history:?}"
9551            );
9552            sleep(Duration::from_millis(10)).await;
9553        }
9554    }
9555
9556    /// A disable issued while a crash respawn is still backing off must preempt
9557    /// that respawn: the operator's stop wins, the disable must not queue behind
9558    /// the backoff, and the module must never come back up afterwards.
9559    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9560    async fn disable_during_crash_backoff_cancels_pending_respawn() {
9561        let backoff = Duration::from_secs(2);
9562        let supervisor = Supervisor::new(
9563            Arc::new(Registry::default()),
9564            RestartPolicy::new(10, backoff),
9565        );
9566        let module = supervisor
9567            .spawn(ModuleSpec {
9568                module_id: "disable-during-backoff".to_string(),
9569                program: fake_aft_stub_path(),
9570                args: Vec::new(),
9571                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9572                reserved: false,
9573                reserved_prefixes: Vec::new(),
9574                protocol: ModuleProtocol::Subc,
9575                overlap: Default::default(),
9576            })
9577            .unwrap();
9578
9579        // Wait for the first crash to put the module into its backoff window.
9580        let deadline = Instant::now() + Duration::from_secs(5);
9581        loop {
9582            if module.status().unwrap().state == ModuleState::Restarting {
9583                break;
9584            }
9585            assert!(
9586                Instant::now() < deadline,
9587                "module never entered the crash backoff"
9588            );
9589            sleep(Duration::from_millis(10)).await;
9590        }
9591
9592        let started = Instant::now();
9593        module.set_enabled(false).await.unwrap();
9594        let waited = started.elapsed();
9595
9596        assert!(
9597            waited < backoff / 2,
9598            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
9599        );
9600        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
9601
9602        // Outlast the backoff: the respawn it was counting down to must never run.
9603        sleep(backoff + Duration::from_millis(500)).await;
9604        let status = module.status().unwrap();
9605        assert_eq!(status.state, ModuleState::Disabled);
9606        assert_eq!(
9607            status.spawn_generation, 1,
9608            "module respawned after the operator disabled it"
9609        );
9610    }
9611
9612    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
9613    /// the shape of nats-server, the program this rule exists for.
9614    #[cfg(unix)]
9615    fn protocol_none_sigterm_exits_clean_spec(
9616        module_id: &str,
9617        dir: &std::path::Path,
9618    ) -> (ModuleSpec, PathBuf, PathBuf) {
9619        let ready = dir.join("ready");
9620        let marker = dir.join("sigterm");
9621        let spec = ModuleSpec {
9622            module_id: module_id.to_string(),
9623            program: fake_aft_stub_path(),
9624            args: Vec::new(),
9625            env: vec![
9626                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
9627                (
9628                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
9629                    marker.display().to_string(),
9630                ),
9631                (
9632                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
9633                    ready.display().to_string(),
9634                ),
9635            ],
9636            reserved: false,
9637            reserved_prefixes: Vec::new(),
9638            protocol: ModuleProtocol::None,
9639            overlap: Default::default(),
9640        };
9641        (spec, ready, marker)
9642    }
9643
9644    /// Wait for a file the child writes, so a signal is never sent before the
9645    /// child's SIGTERM handler is installed (the default disposition would
9646    /// kill it by signal and the exit would not be clean).
9647    #[cfg(unix)]
9648    async fn wait_for_file(path: &std::path::Path) {
9649        let deadline = Instant::now() + Duration::from_secs(10);
9650        while !path.exists() {
9651            assert!(
9652                Instant::now() < deadline,
9653                "{} never appeared",
9654                path.display()
9655            );
9656            sleep(Duration::from_millis(10)).await;
9657        }
9658    }
9659
9660    /// A protocol-none module that exits 0 because something OUTSIDE the
9661    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
9662    /// the crash-path disposition rather than `stopped`.
9663    #[cfg(unix)]
9664    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9665    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
9666        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
9667        let (spec, ready, marker) =
9668            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
9669        let supervisor = Supervisor::new(
9670            Arc::new(Registry::default()),
9671            RestartPolicy::new(3, Duration::ZERO),
9672        );
9673        let module = supervisor.spawn(spec).unwrap();
9674        wait_for_file(&ready).await;
9675        let first_pid = module
9676            .status()
9677            .unwrap()
9678            .pid
9679            .expect("a running module reports its pid");
9680
9681        rustix::process::kill_process(
9682            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
9683            rustix::process::Signal::TERM,
9684        )
9685        .unwrap();
9686
9687        let deadline = Instant::now() + Duration::from_secs(10);
9688        let respawned = loop {
9689            let status = module.status().unwrap();
9690            if status.state == ModuleState::Running
9691                && status.pid.is_some_and(|pid| pid != first_pid)
9692            {
9693                break status;
9694            }
9695            assert!(
9696                Instant::now() < deadline,
9697                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
9698            );
9699            sleep(Duration::from_millis(10)).await;
9700        };
9701        assert_eq!(respawned.spawn_generation, 2);
9702        assert!(
9703            marker.exists(),
9704            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
9705        );
9706
9707        let history = module.terminal_history();
9708        assert_eq!(history.entries.len(), 1, "{history:?}");
9709        let entry = &history.entries[0];
9710        assert_eq!(entry.exit_code, Some(0));
9711        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
9712        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
9713
9714        module.stop().await.unwrap();
9715    }
9716
9717    /// Repeated unrequested clean exits of a protocol-none module spend the
9718    /// restart budget exactly as crashes do, and the module ends `failed` with
9719    /// the budget named.
9720    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9721    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
9722        let supervisor = Supervisor::new(
9723            Arc::new(Registry::default()),
9724            RestartPolicy::new(1, Duration::ZERO),
9725        );
9726        let module = supervisor
9727            .spawn(ModuleSpec {
9728                module_id: "none-clean-exit-budget".to_string(),
9729                program: fake_aft_stub_path(),
9730                args: Vec::new(),
9731                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
9732                reserved: false,
9733                reserved_prefixes: Vec::new(),
9734                protocol: ModuleProtocol::None,
9735                overlap: Default::default(),
9736            })
9737            .unwrap();
9738
9739        let deadline = Instant::now() + Duration::from_secs(10);
9740        loop {
9741            let status = module.status().unwrap();
9742            if status.state == ModuleState::Failed {
9743                break;
9744            }
9745            assert!(
9746                Instant::now() < deadline,
9747                "module never exhausted its budget: {status:?} {:?}",
9748                module.terminal_history()
9749            );
9750            sleep(Duration::from_millis(10)).await;
9751        }
9752        let history = module.terminal_history();
9753        assert_eq!(
9754            history
9755                .entries
9756                .iter()
9757                .map(|entry| (entry.exit_code, entry.disposition.clone()))
9758                .collect::<Vec<_>>(),
9759            vec![
9760                (Some(0), TerminalDisposition::Restarting),
9761                (Some(0), TerminalDisposition::Failed),
9762            ]
9763        );
9764        let detail = history.entries[1]
9765            .disposition_detail
9766            .as_deref()
9767            .expect("a budget failure names the budget");
9768        assert!(detail.contains("max_restarts=1"), "{detail}");
9769        assert_eq!(module.status().unwrap().spawn_generation, 2);
9770    }
9771
9772    /// A stop the supervisor itself requests still stops a protocol-none
9773    /// module, even though the child answers the SIGTERM with exit 0.
9774    #[cfg(unix)]
9775    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9776    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
9777        for disable in [false, true] {
9778            let label = if disable {
9779                "none-requested-disable"
9780            } else {
9781                "none-requested-stop"
9782            };
9783            let dir = subc_test_support::TestTempDir::new(label);
9784            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
9785            let supervisor = Supervisor::new(
9786                Arc::new(Registry::default()),
9787                RestartPolicy::new(3, Duration::ZERO),
9788            );
9789            let module = supervisor.spawn(spec).unwrap();
9790            wait_for_file(&ready).await;
9791
9792            if disable {
9793                module.set_enabled(false).await.unwrap();
9794            } else {
9795                module.stop().await.unwrap();
9796            }
9797            assert!(
9798                marker.exists(),
9799                "{label}: the child must have left through its SIGTERM handler with exit 0"
9800            );
9801
9802            // Long enough for a zero-backoff respawn to have happened if the
9803            // exit had been treated as a crash.
9804            sleep(Duration::from_millis(500)).await;
9805            let status = module.status().unwrap();
9806            let expected = if disable {
9807                ModuleState::Disabled
9808            } else {
9809                ModuleState::Stopped
9810            };
9811            assert_eq!(status.state, expected, "{label}");
9812            assert_eq!(
9813                status.spawn_generation, 1,
9814                "{label}: respawned after a requested stop"
9815            );
9816            let history = module.terminal_history();
9817            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
9818            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
9819            assert_ne!(
9820                history.entries[0].disposition,
9821                TerminalDisposition::Restarting,
9822                "{label}"
9823            );
9824        }
9825    }
9826
9827    /// A subc-wire module that exits 0 on its own is still a stop: the
9828    /// protocol-none rule must not reach it.
9829    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9830    async fn subc_wire_clean_exit_is_still_a_stop() {
9831        let supervisor = Supervisor::new(
9832            Arc::new(Registry::default()),
9833            RestartPolicy::new(3, Duration::ZERO),
9834        );
9835        let module = supervisor
9836            .spawn(ModuleSpec {
9837                module_id: "wire-clean-exit".to_string(),
9838                program: fake_aft_stub_path(),
9839                args: Vec::new(),
9840                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
9841                reserved: false,
9842                reserved_prefixes: Vec::new(),
9843                protocol: ModuleProtocol::Subc,
9844                overlap: Default::default(),
9845            })
9846            .unwrap();
9847
9848        let deadline = Instant::now() + Duration::from_secs(10);
9849        while module.terminal_history().entries.is_empty() {
9850            assert!(Instant::now() < deadline, "module never exited");
9851            sleep(Duration::from_millis(10)).await;
9852        }
9853        // Long enough for a zero-backoff respawn to have happened.
9854        sleep(Duration::from_millis(500)).await;
9855        let status = module.status().unwrap();
9856        assert_eq!(status.state, ModuleState::Stopped);
9857        assert_eq!(status.spawn_generation, 1);
9858        let history = module.terminal_history();
9859        assert_eq!(history.entries.len(), 1, "{history:?}");
9860        assert_eq!(history.entries[0].exit_code, Some(0));
9861        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
9862    }
9863
9864    #[cfg(unix)]
9865    #[tokio::test]
9866    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
9867        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
9868        let record = dir.join("live-children.json");
9869        let supervisor = Supervisor::new(
9870            Arc::new(Registry::default()),
9871            RestartPolicy::new(0, Duration::ZERO),
9872        );
9873        let mut runtime = supervisor.runtime_config();
9874        runtime.child_roster.record_to(record.clone());
9875        let gate = Arc::new(super::ReloadExitRecordGate::default());
9876        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
9877        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9878        let spec = ModuleSpec {
9879            module_id: "reload-exit-roster".into(),
9880            program: fake_aft_stub_path(),
9881            args: Vec::new(),
9882            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
9883            reserved: false,
9884            reserved_prefixes: Vec::new(),
9885            protocol: ModuleProtocol::Subc,
9886            overlap: Default::default(),
9887        };
9888        let mut child = None;
9889        let reload = super::finish_reload_child(
9890            &spec,
9891            &runtime,
9892            &supervisor.registry,
9893            &supervisor.process_liveness,
9894            &snapshot,
9895            &mut child,
9896        );
9897        tokio::pin!(reload);
9898        tokio::select! {
9899            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
9900            _ = gate.reached.notified() => {}
9901        }
9902        assert!(runtime
9903            .terminal_ring
9904            .lock()
9905            .unwrap()
9906            .snapshot()
9907            .entries
9908            .is_empty());
9909        assert_eq!(
9910            crate::live_children::read_record(&record).unwrap().len(),
9911            1,
9912            "shutdown must still wait for the reaped child until its terminal record exists"
9913        );
9914        runtime.child_roster.close();
9915        gate.resume.notify_one();
9916        assert!(reload.await.is_err());
9917        assert!(crate::live_children::read_record(&record)
9918            .unwrap()
9919            .is_empty());
9920        let history = runtime.terminal_ring.lock().unwrap().snapshot();
9921        assert_eq!(history.entries.len(), 1);
9922        assert_eq!(
9923            history.entries[0].disposition,
9924            TerminalDisposition::DaemonShutdown
9925        );
9926    }
9927
9928    /// Each restart-producing arm has its own state transition. Keeping their
9929    /// lifetime count assertions adjacent prevents a later new arm from silently
9930    /// spending budget without recording the historical restart.
9931    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9932    async fn every_restart_increment_path_advances_lifetime_count() {
9933        let supervisor = Supervisor::new(
9934            Arc::new(Registry::default()),
9935            RestartPolicy::new(1, Duration::ZERO),
9936        );
9937        let runtime = supervisor.runtime_config();
9938        let spec = ModuleSpec {
9939            module_id: "lifetime-increment-path".to_string(),
9940            program: PathBuf::from("/unused/lifetime-increment-path"),
9941            args: Vec::new(),
9942            env: Vec::new(),
9943            reserved: false,
9944            reserved_prefixes: Vec::new(),
9945            protocol: ModuleProtocol::Subc,
9946            overlap: Default::default(),
9947        };
9948
9949        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9950        assert!(matches!(
9951            on_child_exit(
9952                &spec,
9953                runtime.restart_policy,
9954                &supervisor.registry,
9955                &crash_snapshot,
9956                &runtime.terminal_ring,
9957                &runtime.spawn_events,
9958                &runtime.child_roster,
9959                ExitReport {
9960                    kind: ExitKind::Crash,
9961                    code: Some(1),
9962                    signal: None,
9963                    at_ms: 1,
9964                },
9965            )
9966            .await,
9967            NextAction::Restart { schedule: _ }
9968        ));
9969        let (crash_restarts, crash_lifetime) = {
9970            let state = lock_snapshot(&crash_snapshot).unwrap();
9971            (state.crash_restarts.len(), state.lifetime_restarts)
9972        };
9973        assert_eq!(crash_restarts, 1);
9974        assert_eq!(crash_lifetime, 1);
9975
9976        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9977        let mut health_child = None;
9978        assert!(matches!(
9979            health_restart_child(
9980                &spec,
9981                &runtime,
9982                &supervisor.registry,
9983                &supervisor.process_liveness,
9984                &health_snapshot,
9985                &mut health_child,
9986                SupervisorHealthStatus::Failing,
9987                None,
9988                2,
9989            )
9990            .await,
9991            Ok(())
9992        ));
9993        assert!(health_child.is_none());
9994        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
9995        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
9996        let (health_restarts, health_lifetime) = {
9997            let state = lock_snapshot(&health_snapshot).unwrap();
9998            (state.crash_restarts.len(), state.lifetime_restarts)
9999        };
10000        assert_eq!(health_restarts, 1);
10001        assert_eq!(health_lifetime, 1);
10002
10003        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10004        let mut reload_child = None;
10005        assert!(matches!(
10006            handle_reload_spawn_failure(
10007                &spec,
10008                &runtime,
10009                &supervisor.process_liveness,
10010                &reload_snapshot,
10011                &mut reload_child,
10012                "forced reload spawn failure".to_string(),
10013            )
10014            .await,
10015            Err(SuperviseError::ReloadFailed { .. })
10016        ));
10017        let (reload_restarts, reload_lifetime) = {
10018            let state = lock_snapshot(&reload_snapshot).unwrap();
10019            (state.crash_restarts.len(), state.lifetime_restarts)
10020        };
10021        assert_eq!(reload_restarts, 1);
10022        assert_eq!(reload_lifetime, 1);
10023    }
10024
10025    #[tokio::test]
10026    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10027        let supervisor = Supervisor::new(
10028            Arc::new(Registry::default()),
10029            RestartPolicy::new(3, Duration::ZERO),
10030        );
10031        let runtime = supervisor.runtime_config();
10032        let spec = ModuleSpec {
10033            module_id: "deliberately-severed".to_string(),
10034            program: PathBuf::from("/unused/deliberately-severed"),
10035            args: Vec::new(),
10036            env: Vec::new(),
10037            reserved: false,
10038            reserved_prefixes: Vec::new(),
10039            protocol: ModuleProtocol::Subc,
10040            overlap: Default::default(),
10041        };
10042        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10043        let process = ProcessIdentity {
10044            pid: 41,
10045            start_time: 101,
10046        };
10047        record_deliberate_severance(&snapshot, process).unwrap();
10048        let exit_report = apply_deliberate_severance_marker(
10049            &snapshot,
10050            Some(process),
10051            ExitReport {
10052                kind: ExitKind::Crash,
10053                code: Some(1),
10054                signal: None,
10055                at_ms: 1,
10056            },
10057        );
10058        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10059
10060        assert!(matches!(
10061            on_child_exit(
10062                &spec,
10063                runtime.restart_policy,
10064                &supervisor.registry,
10065                &snapshot,
10066                &runtime.terminal_ring,
10067                &runtime.spawn_events,
10068                &runtime.child_roster,
10069                exit_report,
10070            )
10071            .await,
10072            NextAction::Restart { schedule: _ }
10073        ));
10074        let state = lock_snapshot(&snapshot).unwrap();
10075        assert_eq!(state.lifetime_restarts, 1);
10076        assert_eq!(state.crash_restarts.len(), 0);
10077    }
10078
10079    #[tokio::test]
10080    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10081        let supervisor = Supervisor::new(
10082            Arc::new(Registry::default()),
10083            RestartPolicy::new(3, Duration::ZERO),
10084        );
10085        let runtime = supervisor.runtime_config();
10086        let spec = ModuleSpec {
10087            module_id: "genuine-crash".to_string(),
10088            program: PathBuf::from("/unused/genuine-crash"),
10089            args: Vec::new(),
10090            env: Vec::new(),
10091            reserved: false,
10092            reserved_prefixes: Vec::new(),
10093            protocol: ModuleProtocol::Subc,
10094            overlap: Default::default(),
10095        };
10096        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10097
10098        assert!(matches!(
10099            on_child_exit(
10100                &spec,
10101                runtime.restart_policy,
10102                &supervisor.registry,
10103                &snapshot,
10104                &runtime.terminal_ring,
10105                &runtime.spawn_events,
10106                &runtime.child_roster,
10107                ExitReport {
10108                    kind: ExitKind::Crash,
10109                    code: Some(1),
10110                    signal: None,
10111                    at_ms: 1,
10112                },
10113            )
10114            .await,
10115            NextAction::Restart { schedule: _ }
10116        ));
10117        let state = lock_snapshot(&snapshot).unwrap();
10118        assert_eq!(state.lifetime_restarts, 1);
10119        assert_eq!(state.crash_restarts.len(), 1);
10120    }
10121
10122    fn crash_exit_report(at_ms: u64) -> ExitReport {
10123        ExitReport {
10124            kind: ExitKind::Crash,
10125            code: Some(1),
10126            signal: None,
10127            at_ms,
10128        }
10129    }
10130
10131    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10132        ModuleSpec {
10133            module_id: module_id.to_string(),
10134            program: PathBuf::from("/unused").join(module_id),
10135            args: Vec::new(),
10136            env: Vec::new(),
10137            reserved: false,
10138            reserved_prefixes: Vec::new(),
10139            protocol: ModuleProtocol::Subc,
10140            overlap: Default::default(),
10141        }
10142    }
10143
10144    /// A real crash loop still stops. Three crashes with nothing aging out spend
10145    /// a budget of two and the third respawn is refused, and both surfaces an
10146    /// operator has -- the log line and the retained terminal record -- name the
10147    /// window rather than only the cap, because `max_restarts=2` alone is what
10148    /// this budget used to mean.
10149    #[tokio::test]
10150    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10151        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10152        let supervisor = Supervisor::new(
10153            Arc::new(Registry::default()),
10154            RestartPolicy::new(2, Duration::ZERO),
10155        );
10156        let runtime = supervisor.runtime_config();
10157        let spec = windowed_crash_spec("crash-loop-in-window");
10158        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10159
10160        for attempt in 1..=2 {
10161            assert!(
10162                matches!(
10163                    on_child_exit(
10164                        &spec,
10165                        runtime.restart_policy,
10166                        &supervisor.registry,
10167                        &snapshot,
10168                        &runtime.terminal_ring,
10169                        &runtime.spawn_events,
10170                        &runtime.child_roster,
10171                        crash_exit_report(attempt),
10172                    )
10173                    .await,
10174                    NextAction::Restart { schedule: _ }
10175                ),
10176                "crash {attempt} is inside the budget and must respawn"
10177            );
10178        }
10179
10180        assert!(matches!(
10181            on_child_exit(
10182                &spec,
10183                runtime.restart_policy,
10184                &supervisor.registry,
10185                &snapshot,
10186                &runtime.terminal_ring,
10187                &runtime.spawn_events,
10188                &runtime.child_roster,
10189                crash_exit_report(3),
10190            )
10191            .await,
10192            NextAction::Stop { .. }
10193        ));
10194
10195        {
10196            let state = lock_snapshot(&snapshot).unwrap();
10197            assert_eq!(state.state, ModuleState::Failed);
10198            assert_eq!(state.crash_restarts.len(), 2);
10199            assert_eq!(state.lifetime_restarts, 2);
10200        }
10201
10202        let history = runtime
10203            .terminal_ring
10204            .lock()
10205            .expect("terminal ring is not poisoned")
10206            .snapshot();
10207        let last = history
10208            .entries
10209            .last()
10210            .expect("the refused crash is retained");
10211        assert_eq!(last.disposition, TerminalDisposition::Failed);
10212        assert_eq!(
10213            last.disposition_detail.as_deref(),
10214            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10215        );
10216
10217        let captured = crate::router::test_log::captured_logs(&logs);
10218        assert!(
10219            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10220            "the stop must be logged with its window: {captured}"
10221        );
10222    }
10223
10224    /// The rate, stated as a test: three crashes where the first has aged past
10225    /// the window are two crashes as far as the budget is concerned, so the
10226    /// third respawn is allowed and the ring holds only the two recent ones.
10227    ///
10228    /// This is the case a lifetime counter got wrong -- and the case the daemon
10229    /// now hits routinely, since a module exits non-zero every time its
10230    /// connection to the daemon drops.
10231    #[tokio::test]
10232    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10233        let supervisor = Supervisor::new(
10234            Arc::new(Registry::default()),
10235            RestartPolicy::new(2, Duration::ZERO),
10236        );
10237        let runtime = supervisor.runtime_config();
10238        let spec = windowed_crash_spec("crash-across-windows");
10239        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10240
10241        for attempt in 1..=2 {
10242            assert!(matches!(
10243                on_child_exit(
10244                    &spec,
10245                    runtime.restart_policy,
10246                    &supervisor.registry,
10247                    &snapshot,
10248                    &runtime.terminal_ring,
10249                    &runtime.spawn_events,
10250                    &runtime.child_roster,
10251                    crash_exit_report(attempt),
10252                )
10253                .await,
10254                NextAction::Restart { schedule: _ }
10255            ));
10256        }
10257
10258        // The oldest crash moves out of the window; nothing else about the
10259        // module changes.
10260        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10261            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10262        })
10263        .unwrap();
10264
10265        assert!(
10266            matches!(
10267                on_child_exit(
10268                    &spec,
10269                    runtime.restart_policy,
10270                    &supervisor.registry,
10271                    &snapshot,
10272                    &runtime.terminal_ring,
10273                    &runtime.spawn_events,
10274                    &runtime.child_roster,
10275                    crash_exit_report(3),
10276                )
10277                .await,
10278                NextAction::Restart { schedule: _ }
10279            ),
10280            "a crash older than the window must not hold a budget slot"
10281        );
10282
10283        let state = lock_snapshot(&snapshot).unwrap();
10284        assert_eq!(state.state, ModuleState::Restarting);
10285        assert_eq!(
10286            state.crash_restarts.len(),
10287            2,
10288            "the aged instant is dropped and the new one takes its place"
10289        );
10290        assert_eq!(
10291            state.lifetime_restarts, 3,
10292            "the ledger counts every restart, including the ones the window forgot"
10293        );
10294    }
10295
10296    /// An operator restart hands the budget back whole, and the ledger keeps
10297    /// counting. Those are different questions -- "how close is this module to
10298    /// being stopped" and "how many times has it been replaced" -- and the
10299    /// operator action answers only the first.
10300    #[tokio::test]
10301    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10302        let supervisor = Supervisor::new(
10303            Arc::new(Registry::default()),
10304            RestartPolicy::new(2, Duration::ZERO),
10305        );
10306        let runtime = supervisor.runtime_config();
10307        let spec = windowed_crash_spec("operator-cleared-budget");
10308        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10309
10310        for attempt in 1..=2 {
10311            assert!(matches!(
10312                on_child_exit(
10313                    &spec,
10314                    runtime.restart_policy,
10315                    &supervisor.registry,
10316                    &snapshot,
10317                    &runtime.terminal_ring,
10318                    &runtime.spawn_events,
10319                    &runtime.child_roster,
10320                    crash_exit_report(attempt),
10321                )
10322                .await,
10323                NextAction::Restart { schedule: _ }
10324            ));
10325        }
10326
10327        reset_restart_count(&snapshot, &spec.module_id).unwrap();
10328        {
10329            let state = lock_snapshot(&snapshot).unwrap();
10330            assert!(
10331                state.crash_restarts.is_empty(),
10332                "an operator restart returns the full budget"
10333            );
10334            assert_eq!(
10335                state.lifetime_restarts, 2,
10336                "clearing the budget must not unmake the crashes"
10337            );
10338        }
10339
10340        assert!(
10341            matches!(
10342                on_child_exit(
10343                    &spec,
10344                    runtime.restart_policy,
10345                    &supervisor.registry,
10346                    &snapshot,
10347                    &runtime.terminal_ring,
10348                    &runtime.spawn_events,
10349                    &runtime.child_roster,
10350                    crash_exit_report(3),
10351                )
10352                .await,
10353                NextAction::Restart { schedule: _ }
10354            ),
10355            "the cleared budget must be spendable again"
10356        );
10357        let state = lock_snapshot(&snapshot).unwrap();
10358        assert_eq!(state.crash_restarts.len(), 1);
10359        assert_eq!(state.lifetime_restarts, 3);
10360    }
10361
10362    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10363    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
10364        let severed = ProcessIdentity {
10365            pid: 41,
10366            start_time: 101,
10367        };
10368        let successor = ProcessIdentity {
10369            pid: 41,
10370            start_time: 202,
10371        };
10372        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
10373        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
10374            state.pid = Some(successor.pid);
10375            state.process_start_time = Some(successor.start_time);
10376        })
10377        .unwrap();
10378        assert!(!module.record_deliberate_severance(severed).unwrap());
10379
10380        let exit_report = apply_deliberate_severance_marker(
10381            &module.inner.snapshot,
10382            Some(successor),
10383            ExitReport {
10384                kind: ExitKind::Crash,
10385                code: Some(1),
10386                signal: None,
10387                at_ms: 1,
10388            },
10389        );
10390
10391        assert_eq!(exit_report.kind, ExitKind::Crash);
10392    }
10393
10394    #[tokio::test]
10395    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
10396        let registry = Registry::default();
10397        let supervisor = Supervisor::new(
10398            Arc::new(Registry::default()),
10399            RestartPolicy::new(3, Duration::ZERO),
10400        );
10401        let runtime = supervisor.runtime_config();
10402        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10403        let spec = ModuleSpec {
10404            module_id: "drain-deliberate-severance".to_string(),
10405            program: fake_aft_stub_path(),
10406            args: Vec::new(),
10407            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10408            reserved: false,
10409            reserved_prefixes: Vec::new(),
10410            protocol: ModuleProtocol::Subc,
10411            overlap: Default::default(),
10412        };
10413        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10414        let process = ProcessIdentity {
10415            pid: 41,
10416            start_time: 101,
10417        };
10418        child.process_identity = Some(process);
10419        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10420            state.pid = Some(process.pid);
10421            state.process_start_time = Some(process.start_time);
10422        })
10423        .unwrap();
10424        record_deliberate_severance(&snapshot, process).unwrap();
10425
10426        drain_child_to_state(
10427            &spec.module_id,
10428            spec.protocol,
10429            // The child exits on its own; no signal may change the exit this
10430            // test classifies.
10431            StopNotice::SentOverConnection,
10432            &registry,
10433            None,
10434            &snapshot,
10435            &runtime.terminal_ring,
10436            &runtime.spawn_events,
10437            child,
10438            Duration::from_secs(1),
10439            ModuleState::Stopped,
10440            Some(false),
10441        )
10442        .await
10443        .unwrap();
10444
10445        let state = lock_snapshot(&snapshot).unwrap();
10446        assert_eq!(
10447            state.last_exit.as_ref().map(|exit| exit.kind),
10448            Some(ExitKind::DeliberateSeverance)
10449        );
10450        assert_eq!(state.lifetime_restarts, 1);
10451        assert_eq!(state.crash_restarts.len(), 0);
10452        drop(state);
10453        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10454        assert_eq!(
10455            history.entries[0].exit_kind,
10456            subc_control::TerminalExitKind::DeliberateSeverance
10457        );
10458    }
10459
10460    #[tokio::test]
10461    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
10462        let registry = Registry::default();
10463        let supervisor = Supervisor::new(
10464            Arc::new(Registry::default()),
10465            RestartPolicy::new(3, Duration::ZERO),
10466        );
10467        let runtime = supervisor.runtime_config();
10468        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10469        let spec = ModuleSpec {
10470            module_id: "ordinary-drain".to_string(),
10471            program: fake_aft_stub_path(),
10472            args: Vec::new(),
10473            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10474            reserved: false,
10475            reserved_prefixes: Vec::new(),
10476            protocol: ModuleProtocol::Subc,
10477            overlap: Default::default(),
10478        };
10479        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10480
10481        drain_child_to_state(
10482            &spec.module_id,
10483            spec.protocol,
10484            // The child exits on its own; no signal may change the exit this
10485            // test classifies.
10486            StopNotice::SentOverConnection,
10487            &registry,
10488            None,
10489            &snapshot,
10490            &runtime.terminal_ring,
10491            &runtime.spawn_events,
10492            child,
10493            Duration::from_secs(1),
10494            ModuleState::Stopped,
10495            Some(false),
10496        )
10497        .await
10498        .unwrap();
10499
10500        let state = lock_snapshot(&snapshot).unwrap();
10501        assert_eq!(
10502            state.last_exit.as_ref().map(|exit| exit.kind),
10503            Some(ExitKind::Crash)
10504        );
10505        assert_eq!(state.lifetime_restarts, 0);
10506        assert_eq!(state.crash_restarts.len(), 0);
10507    }
10508
10509    #[test]
10510    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
10511        // The server's generic fatal-routing branch only knows that the
10512        // connection failed; it does not know that the daemon deliberately
10513        // initiated a process-killing severance. Keep this seam explicit so a
10514        // future connection error path cannot silently reintroduce the stale
10515        // exemption that mislabels a later genuine crash.
10516        assert!(!include_str!("server.rs")
10517            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
10518    }
10519
10520    /// The `route.closed` `drained` value must be the quiescence wait's own
10521    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
10522    /// measurement at all and `false` is the one honest constant. This is the exact
10523    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
10524    /// on every return path, including the one that used to return early via `?`
10525    /// with `route.closing` already sent and no `route.closed` ever following.
10526    #[test]
10527    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
10528        assert!(drained_after_quiescence_wait(&Ok(true)));
10529        assert!(!drained_after_quiescence_wait(&Ok(false)));
10530        assert!(!drained_after_quiescence_wait(&Err(
10531            SuperviseError::StatePoisoned { module_id: None }
10532        )));
10533    }
10534
10535    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
10536    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
10537    /// already reaped out-of-band) still leaves a terminal record rather than none
10538    /// at all. Triggering the real `wait()` I/O error from an integration test would
10539    /// need a genuine already-reaped-child race, which is OS-specific and not
10540    /// something this suite attempts elsewhere; this test instead verifies the
10541    /// record produced for that arm end-to-end through the real `TerminalRing`, and
10542    /// the call site itself is verified by inspection to sit in that exact arm.
10543    #[test]
10544    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
10545        let ring = Arc::new(Mutex::new(TerminalRing::new(
10546            TerminalRingConfig::default(),
10547            0,
10548        )));
10549        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
10550
10551        let snapshot = ring.lock().unwrap().snapshot();
10552        assert_eq!(snapshot.entries.len(), 1);
10553        let entry = &snapshot.entries[0];
10554        assert_eq!(entry.exit_code, None);
10555        assert_eq!(entry.exit_signal, None);
10556        assert_eq!(entry.disposition, TerminalDisposition::Failed);
10557    }
10558
10559    #[test]
10560    fn wait_error_exit_path_preserves_spawn_event_density() {
10561        let feed = super::SpawnEventFeed::default();
10562        feed.configure_incarnation("wait-error-density".to_string());
10563        feed.emit_spawned("wait-error", 41, 1);
10564        let ring = Arc::new(Mutex::new(TerminalRing::new(
10565            TerminalRingConfig::default(),
10566            0,
10567        )));
10568
10569        record_wait_error_terminal("wait-error", &ring, &feed);
10570        feed.emit_spawned("after-wait-error", 42, 2);
10571
10572        let state = feed.0.lock().unwrap();
10573        let sequences = state
10574            .events
10575            .iter()
10576            .map(|event| event.cursor.seq)
10577            .collect::<Vec<_>>();
10578        assert_eq!(sequences, vec![1, 2, 3]);
10579        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
10580        assert_eq!(state.events[1].exit_code, None);
10581        assert_eq!(state.events[1].exit_signal, None);
10582    }
10583
10584    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
10585    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
10586    /// not a clean exit it never actually observed.
10587    #[test]
10588    fn wait_error_exit_report_is_classified_as_a_crash() {
10589        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
10590    }
10591}
10592
10593#[cfg(test)]
10594mod health_evidence_tests {
10595    use super::{HealthProbeError, HealthProbeEvidence};
10596    use std::collections::HashSet;
10597
10598    /// The evidential asymmetry, asserted rather than described.
10599    ///
10600    /// Exactly ONE observation is proof a module cannot serve, and the one that
10601    /// fires under CPU starvation is not it. Before the split, all fifteen
10602    /// construction sites collapsed into a single String, so a timeout carried the
10603    /// same weight as a dead lane -- which is how a healthy module was restarted
10604    /// three times in one day.
10605    #[test]
10606    fn only_a_dead_lane_is_proof_of_death() {
10607        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
10608        // Three non-proof classes, each for a different reason: silence is
10609        // consistent with health, a bad answer proves the module ALIVE, and a
10610        // daemon-side fault never reached the module at all.
10611        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
10612        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
10613        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
10614    }
10615
10616    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
10617    ///
10618    /// A shared label renders two different observations identically in the line an
10619    /// operator reads after an unexplained restart -- the exact confusion this
10620    /// change removes.
10621    #[test]
10622    fn every_evidence_class_has_a_distinct_label() {
10623        let labels = [
10624            HealthProbeError::lane_dead("").label(),
10625            HealthProbeError::no_answer("").label(),
10626            HealthProbeError::bad_answer("").label(),
10627            HealthProbeError::misconfigured("").label(),
10628        ];
10629        let unique: HashSet<_> = labels.iter().collect();
10630        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
10631    }
10632
10633    /// The class is additional information, not a replacement.
10634    ///
10635    /// An operator needs both "this was silence" and the specific text saying how
10636    /// long we waited; a classification that swallowed the message would trade one
10637    /// missing distinction for another.
10638    #[test]
10639    fn classification_preserves_the_original_message() {
10640        let err = HealthProbeError::no_answer("module did not answer within 5s");
10641        assert_eq!(err.to_string(), "module did not answer within 5s");
10642        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
10643    }
10644}
10645
10646#[cfg(test)]
10647mod health_tombstone_tests {
10648    use std::{path::PathBuf, sync::Arc, time::Duration};
10649
10650    use subc_protocol::{
10651        manifest::Concurrency,
10652        session::{HealthStatus, ModuleControlResponse},
10653    };
10654    use tokio::sync::mpsc;
10655
10656    use super::{
10657        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
10658        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
10659    };
10660    use crate::{
10661        control::ControlHandler,
10662        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
10663        registry::{ConnectionId, Registry},
10664        router::FrameSink,
10665    };
10666
10667    struct ProbeHarness {
10668        spec: ModuleSpec,
10669        runtime: SupervisorRuntimeConfig,
10670        forwarding: Arc<ForwardingTable>,
10671        module_connection: ConnectionId,
10672        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
10673        handler: ControlHandler,
10674        module: super::SupervisedModule,
10675    }
10676
10677    fn probe_harness() -> ProbeHarness {
10678        let registry = Arc::new(Registry::default());
10679        let forwarding = Arc::new(ForwardingTable::default());
10680        let supervisor_handle = super::SupervisorHandle::new();
10681        let health = HealthConfig {
10682            http: None,
10683            cadence: Duration::from_secs(30),
10684            deadline: Duration::from_secs(5),
10685            failure_threshold: 3,
10686            on_degraded: HealthAction::Report,
10687            on_failing: HealthAction::Report,
10688            critical: false,
10689        };
10690        let supervisor = Supervisor::new(Arc::clone(&registry), RestartPolicy::default())
10691            .with_forwarding(Arc::clone(&forwarding))
10692            .with_handle(supervisor_handle.clone())
10693            .with_health_config(health);
10694        let spec = ModuleSpec {
10695            module_id: "late-health-module".to_string(),
10696            program: PathBuf::from("disabled-module"),
10697            args: Vec::new(),
10698            env: Vec::new(),
10699            reserved: false,
10700            reserved_prefixes: Vec::new(),
10701            protocol: ModuleProtocol::Subc,
10702            overlap: Default::default(),
10703        };
10704        let module = supervisor
10705            .supervise_configured(spec.clone(), false)
10706            .unwrap();
10707        let runtime = supervisor.runtime_config();
10708        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
10709            .with_supervisor(supervisor_handle);
10710        let module_connection = ConnectionId::new(700);
10711        let (module_tx, module_rx) = mpsc::channel(8);
10712        forwarding
10713            .register_module_connection(
10714                module_connection,
10715                spec.module_id.clone(),
10716                subc_protocol::PROTOCOL_VERSION,
10717                Concurrency::ModuleManaged,
10718                FrameSink::new(module_tx),
10719            )
10720            .unwrap();
10721
10722        ProbeHarness {
10723            spec,
10724            runtime,
10725            forwarding,
10726            module_connection,
10727            module_rx,
10728            handler,
10729            module,
10730        }
10731    }
10732
10733    async fn finish_after(
10734        harness: &mut ProbeHarness,
10735        stall: Duration,
10736    ) -> ModuleControlRpcCompletion {
10737        assert!(stall > harness.runtime.health.deadline);
10738        let deadline = harness.runtime.health.deadline;
10739        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
10740        let answer = async {
10741            let frame = harness.module_rx.recv().await.expect("health.check frame");
10742            tokio::time::advance(deadline).await;
10743            tokio::task::yield_now().await;
10744            tokio::time::advance(stall - deadline).await;
10745            harness
10746                .forwarding
10747                .complete_module_control_rpc(
10748                    harness.module_connection,
10749                    frame.header.corr,
10750                    Some("health.check"),
10751                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
10752                        status: HealthStatus::Ok,
10753                        detail: None,
10754                        metrics: None,
10755                    }),
10756                )
10757                .unwrap()
10758        };
10759        let (probe_result, completion) = tokio::join!(probe, answer);
10760        let err = probe_result.expect_err("probe must miss its deadline");
10761        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
10762        completion
10763    }
10764
10765    async fn time_out_without_answer(harness: &mut ProbeHarness) {
10766        let deadline = harness.runtime.health.deadline;
10767        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
10768        let exhaust_deadline = async {
10769            let _frame = harness.module_rx.recv().await.expect("health.check frame");
10770            tokio::time::advance(deadline).await;
10771            tokio::task::yield_now().await;
10772        };
10773        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
10774        let err = probe_result.expect_err("probe must miss its deadline");
10775        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
10776    }
10777
10778    #[tokio::test(start_paused = true)]
10779    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
10780        let mut harness = probe_harness();
10781
10782        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
10783        let first_latency = match &first {
10784            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
10785            other => panic!("late answer was not retained: {other:?}"),
10786        };
10787        assert!(harness.handler.observe_module_control_completion(first));
10788
10789        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
10790        let second_latency = match &second {
10791            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
10792            other => panic!("late answer was not retained: {other:?}"),
10793        };
10794        assert!(harness.handler.observe_module_control_completion(second));
10795
10796        assert_eq!(first_latency, Duration::from_secs(8));
10797        assert_eq!(
10798            second_latency - first_latency,
10799            Duration::from_secs(3),
10800            "latency must grow linearly with the additional stall"
10801        );
10802        let health = harness.module.status().unwrap().health;
10803        assert_eq!(health.late_answer_count, 2);
10804        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
10805    }
10806
10807    /// A module that answers every probe late must never march to the kill
10808    /// threshold: the late answer proves it is alive, so it must clear the miss
10809    /// streak the timeout recorded. Without the reset, a CPU-starved module
10810    /// that serves every probe seconds past the deadline accumulates
10811    /// `consecutive_failures` to the threshold and is killed — the exact
10812    /// sequence from the 2026-08-14 aft disable, where the daemon logged
10813    /// "proves the module is alive" five times while counting five misses.
10814    #[tokio::test(start_paused = true)]
10815    async fn late_answer_clears_the_consecutive_failure_streak() {
10816        let mut harness = probe_harness();
10817
10818        // Timeout recorded first: the probe path saw no answer in time.
10819        time_out_without_answer(&mut harness).await;
10820        harness
10821            .module
10822            .record_health_probe_failure_for_test("[no-answer] test miss")
10823            .unwrap();
10824        assert_eq!(
10825            harness.module.status().unwrap().health.consecutive_failures,
10826            1,
10827            "precondition: the miss must be on the streak before the late answer"
10828        );
10829
10830        // The stalled reply then lands: proof of life.
10831        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
10832        assert!(matches!(
10833            late,
10834            ModuleControlRpcCompletion::LateHealthAnswer { .. }
10835        ));
10836        assert!(harness.handler.observe_module_control_completion(late));
10837
10838        let health = harness.module.status().unwrap().health;
10839        assert_eq!(
10840            health.consecutive_failures, 0,
10841            "a late answer is an answer: the streak must reset"
10842        );
10843        assert_eq!(health.late_answer_count, 1);
10844    }
10845
10846    #[tokio::test(start_paused = true)]
10847    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
10848        let mut harness = probe_harness();
10849
10850        for _ in 0..20 {
10851            time_out_without_answer(&mut harness).await;
10852            assert_eq!(
10853                harness.forwarding.health_probe_tombstone_count().unwrap(),
10854                1
10855            );
10856        }
10857    }
10858}
10859
10860#[cfg(test)]
10861mod child_env_tests {
10862    use super::{
10863        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
10864        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
10865        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
10866    };
10867    use std::{ffi::OsStr, path::PathBuf};
10868    use tokio::process::Command;
10869
10870    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
10871        ModuleSpec {
10872            module_id: "env-plan".to_string(),
10873            program: PathBuf::from("/nonexistent"),
10874            args: Vec::new(),
10875            env,
10876            reserved: false,
10877            reserved_prefixes: Vec::new(),
10878            protocol: ModuleProtocol::Subc,
10879            overlap: Default::default(),
10880        }
10881    }
10882
10883    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
10884    /// one still gets its own.
10885    ///
10886    /// This is the narrow goal `env_clear()` was reached for, and the reason the
10887    /// fix is `env_remove` rather than deleting the line: an operator's ambient
10888    /// filter silently becoming an unconfigured module's log level is a real
10889    /// defect, just a much smaller one than clearing the environment.
10890    ///
10891    /// Asserted on the command plan rather than a spawned child because proving
10892    /// the ABSENCE of an inherited variable needs the parent's environment
10893    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
10894    /// removal as `(key, None)`, which is exactly the distinction wanted: not
10895    /// "absent because nobody set it" but "explicitly unset for the child".
10896    #[test]
10897    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
10898        let mut command = Command::new("/nonexistent");
10899        apply_child_env(&mut command, &spec(Vec::new()));
10900        let removed = command
10901            .as_std()
10902            .get_envs()
10903            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
10904        assert!(
10905            removed,
10906            "ambient CK_LOG must be explicitly removed for an unconfigured module"
10907        );
10908
10909        let mut configured = Command::new("/nonexistent");
10910        apply_child_env(
10911            &mut configured,
10912            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
10913        );
10914        let effective = configured
10915            .as_std()
10916            .get_envs()
10917            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
10918            .last()
10919            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
10920        assert_eq!(
10921            effective,
10922            Some(Some("debug".to_string())),
10923            "a module's configured CK_LOG must survive the ambient removal"
10924        );
10925    }
10926
10927    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
10928    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
10929    /// the same reason as the CK_LOG test above.
10930    ///
10931    /// The argument is the load-bearing half: a stock binary exits on an
10932    /// unknown flag before it listens, so with `--subc` appended the mode
10933    /// could not supervise the one process it exists for. Found by the first
10934    /// conformance run (nats-server: `flag provided but not defined: -subc`).
10935    #[test]
10936    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
10937        let connection_file = std::path::Path::new("/run/subc-connection.json");
10938        let handle = SupervisorHandle::new();
10939
10940        let mut none_spec = spec(Vec::new());
10941        none_spec.protocol = ModuleProtocol::None;
10942        let mut none = Command::new("/nonexistent");
10943        let none_handoff =
10944            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
10945                .expect("protocol-none spawn args apply");
10946        assert!(
10947            none_handoff.is_none(),
10948            "protocol:none spawn must not receive a nonce descriptor"
10949        );
10950        assert!(
10951            !none.as_std().get_envs().any(|(key, value)| key
10952                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
10953                && value.is_some()),
10954            "protocol:none spawn must not name a nonce descriptor"
10955        );
10956        let none_args: Vec<String> = none
10957            .as_std()
10958            .get_args()
10959            .map(|a| a.to_string_lossy().into_owned())
10960            .collect();
10961        assert!(
10962            !none_args.iter().any(|a| a == SUBC_ARG),
10963            "protocol:none argv must not carry --subc; got {none_args:?}"
10964        );
10965        let none_has_nonce = none
10966            .as_std()
10967            .get_envs()
10968            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
10969        assert!(
10970            !none_has_nonce,
10971            "protocol:none spawn must not receive a launch nonce"
10972        );
10973        let none_has_module_id = none
10974            .as_std()
10975            .get_envs()
10976            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
10977        assert!(
10978            none_has_module_id,
10979            "SUBC_MODULE_ID is inert and stays on every path"
10980        );
10981        assert!(
10982            handle.spawn_nonce(&none_spec.module_id).is_none(),
10983            "no nonce record for a process that will never present one"
10984        );
10985
10986        // Control: the subc-wire path is unchanged by the branch above.
10987        let wire_spec = spec(Vec::new());
10988        let mut wire = Command::new("/nonexistent");
10989        let wire_handoff =
10990            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
10991                .expect("subc-wire spawn args apply");
10992        let wire_fd_env = wire
10993            .as_std()
10994            .get_envs()
10995            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
10996            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
10997        #[cfg(unix)]
10998        assert_eq!(
10999            wire_fd_env,
11000            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11001            "a subc-wire spawn names the pipe it will receive at descriptor 3"
11002        );
11003        #[cfg(not(unix))]
11004        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11005        let wire_args: Vec<String> = wire
11006            .as_std()
11007            .get_args()
11008            .map(|a| a.to_string_lossy().into_owned())
11009            .collect();
11010        assert_eq!(
11011            wire_args,
11012            vec![
11013                SUBC_ARG.to_string(),
11014                connection_file.to_string_lossy().into_owned()
11015            ],
11016            "a subc-wire spawn still carries --subc <path>"
11017        );
11018        assert_eq!(
11019            wire.as_std()
11020                .get_envs()
11021                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11022            !cfg!(unix),
11023            "only Windows supplies the environment nonce"
11024        );
11025        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11026    }
11027
11028    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
11029    /// spec tries to set it; only a swap candidate carries it.
11030    ///
11031    /// "Set it only on candidates" is not enough, because spawn applies the
11032    /// spec's env verbatim and the daemon's own environment is inherited: either
11033    /// could hand a plain restart the swap role, and a module reading it would
11034    /// warm on its long swap budget while callers wait. Asserted as an explicit
11035    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
11036    /// test above gives.
11037    #[test]
11038    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11039        let role = |command: &Command| {
11040            command
11041                .as_std()
11042                .get_envs()
11043                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11044                .last()
11045                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11046        };
11047        let forged = spec(vec![(
11048            SUBC_SPAWN_ROLE_ENV.to_string(),
11049            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11050        )]);
11051
11052        let mut plain = Command::new("/nonexistent");
11053        apply_child_env(&mut plain, &forged);
11054        apply_spawn_role(&mut plain, SpawnRole::Plain);
11055        assert_eq!(
11056            role(&plain),
11057            Some(None),
11058            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11059        );
11060
11061        let mut candidate = Command::new("/nonexistent");
11062        apply_child_env(&mut candidate, &spec(Vec::new()));
11063        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11064        assert_eq!(
11065            role(&candidate),
11066            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11067        );
11068    }
11069
11070    /// Daemon-private capture retention keys never reach the child.
11071    ///
11072    /// cortexkit-log exposes retention as a Rust struct with no environment
11073    /// names, so these entries are supervisor metadata. Passing them through
11074    /// would invent a public child-process contract by accident.
11075    #[test]
11076    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11077        let mut command = Command::new("/nonexistent");
11078        apply_child_env(
11079            &mut command,
11080            &spec(vec![
11081                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11082                ("KEPT".to_string(), "yes".to_string()),
11083            ]),
11084        );
11085        let keys: Vec<String> = command
11086            .as_std()
11087            .get_envs()
11088            .filter(|(_, value)| value.is_some())
11089            .map(|(key, _)| key.to_string_lossy().into_owned())
11090            .collect();
11091        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11092        assert!(
11093            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11094            "daemon-private capture key leaked to the child: {keys:?}"
11095        );
11096    }
11097}
11098
11099#[cfg(test)]
11100mod jitter_tests {
11101    use super::jittered_health_delay;
11102    use std::{collections::HashSet, time::Duration};
11103
11104    /// Module ids drawn from a real fleet, so the dispersal claim is about names
11105    /// that actually occur rather than invented ones.
11106    ///
11107    /// This is a SAMPLE, not a registry: the property under test is that distinct
11108    /// ids disperse, which holds for any set of distinct strings. Several entries
11109    /// are already historical (modules get renamed), and that costs nothing here --
11110    /// but it means a reader must not mistake this for the live module set, and a
11111    /// rename sweep will match it without there being anything to change.
11112    const FLEET: [&str; 14] = [
11113        "aft",
11114        "alfonso-core",
11115        "magic-context",
11116        "broca",
11117        "thalamus",
11118        "quota",
11119        "engram",
11120        "plexus",
11121        "cerebellum",
11122        "astrocyte",
11123        "synapse",
11124        "subc-mcp",
11125        "cortexkit-credentials",
11126        "subc-federation",
11127    ];
11128
11129    /// Probes must not converge after a fleet-wide restart.
11130    ///
11131    /// This is the property the jitter exists for: every module reconnects at
11132    /// once, and without dispersal all fourteen would then probe on the same
11133    /// tick forever. Nothing failed visibly when this went untested -- a
11134    /// convergent fleet still probes correctly, just in a burst, so the symptom
11135    /// is a periodic load spike that looks like whatever else is running.
11136    #[test]
11137    fn probe_delays_disperse_across_the_fleet() {
11138        let cadence = Duration::from_secs(30);
11139        let delays: HashSet<Duration> = FLEET
11140            .iter()
11141            .map(|id| jittered_health_delay(id, 0, cadence))
11142            .collect();
11143        assert_eq!(
11144            delays.len(),
11145            FLEET.len(),
11146            "every supervised module must land on its own probe offset"
11147        );
11148    }
11149
11150    /// The offset may only ever DELAY a probe, never bring it forward.
11151    ///
11152    /// A delay below the cadence would probe a module more often than
11153    /// configured, which is the opposite of what an operator asked for and
11154    /// would tighten the failure budget without anyone changing it.
11155    #[test]
11156    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11157        let cadence = Duration::from_secs(30);
11158        let span = cadence / 10;
11159        for id in FLEET {
11160            for probe_index in 0..8 {
11161                let delay = jittered_health_delay(id, probe_index, cadence);
11162                assert!(
11163                    delay >= cadence,
11164                    "{id}#{probe_index}: jitter must not shorten the cadence"
11165                );
11166                assert!(
11167                    delay < cadence + span,
11168                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11169                );
11170            }
11171        }
11172    }
11173
11174    /// A module keeps its offset across daemon restarts.
11175    ///
11176    /// The delay is derived rather than randomised precisely so a restart does
11177    /// not re-roll every module into a fresh chance of collision. A random
11178    /// source would satisfy the dispersal test above and quietly lose this.
11179    #[test]
11180    fn a_module_offset_is_stable_across_restarts() {
11181        let cadence = Duration::from_secs(30);
11182        for id in FLEET {
11183            assert_eq!(
11184                jittered_health_delay(id, 0, cadence),
11185                jittered_health_delay(id, 0, cadence),
11186                "{id}: the same module and probe index must produce the same offset"
11187            );
11188        }
11189    }
11190
11191    /// A zero cadence disables probing rather than producing a busy loop.
11192    #[test]
11193    fn zero_cadence_yields_zero_delay() {
11194        assert_eq!(
11195            jittered_health_delay("aft", 0, Duration::ZERO),
11196            Duration::ZERO
11197        );
11198    }
11199}
11200
11201#[cfg(all(test, target_os = "linux"))]
11202mod cgroup_placement_tests {
11203    use super::{
11204        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11205        SupervisedChild,
11206    };
11207    use crate::stderr_tail::{StderrRing, StderrTailConfig};
11208    use std::{
11209        fs, io,
11210        path::{Path, PathBuf},
11211        sync::{Arc, Mutex},
11212    };
11213    use subc_test_support::TestTempDir;
11214    use tokio::process::Command;
11215
11216    #[tokio::test]
11217    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11218        use super::*;
11219        let dir = TestTempDir::new("unique-spawn-cgroups");
11220        let root = PathBuf::from(format!(
11221            "/sys/fs/cgroup/subc-unique-test-{}-{}",
11222            std::process::id(),
11223            unix_ms_now()
11224        ));
11225        if let Err(error) = fs::create_dir(&root) {
11226            assert!(
11227                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11228                "required cgroup test cannot execute: {error}"
11229            );
11230            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11231            return;
11232        }
11233        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11234        let group_count = || {
11235            fs::read_dir(root.join("subc-modules"))
11236                .unwrap()
11237                .map(|entry| entry.unwrap().file_type().unwrap())
11238                .filter(|kind| kind.is_dir())
11239                .count()
11240        };
11241        let supervisor = Supervisor::default().with_cgroup_placement(Some(placement.clone()));
11242        let runtime = supervisor.runtime_config();
11243        let mut spec = ModuleSpec {
11244            module_id: "unique-spawn".into(),
11245            program: PathBuf::from("/bin/sleep"),
11246            args: vec!["60".into()],
11247            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11248                .into_iter()
11249                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
11250                .collect(),
11251            reserved: false,
11252            reserved_prefixes: vec![],
11253            protocol: ModuleProtocol::None,
11254            overlap: Default::default(),
11255        };
11256        let spawn = |spec: &ModuleSpec| {
11257            spawn_child(
11258                spec,
11259                None,
11260                None,
11261                &runtime.stderr_ring,
11262                None,
11263                &runtime.child_roster,
11264                Some(&placement),
11265            )
11266            .unwrap()
11267        };
11268        let mut live = spawn(&spec);
11269        for _ in 0..3 {
11270            // A new process can enter the old slot while retirement is pending.
11271            let next = spawn(&spec);
11272            assert_ne!(live.module_id, next.module_id);
11273            live.start_kill().unwrap();
11274            live.wait().await.unwrap();
11275            live = next;
11276            assert!(
11277                live.child.try_wait().unwrap().is_none(),
11278                "retiring the old slot must not kill the replacement"
11279            );
11280            assert_eq!(
11281                group_count(),
11282                1,
11283                "only the live spawn's cgroup should remain"
11284            );
11285        }
11286        supervisor.begin_daemon_shutdown();
11287        let reap = tokio::spawn(async move {
11288            live.wait().await.unwrap();
11289        });
11290        supervisor
11291            .end_children_for_daemon_shutdown(false, std::future::pending())
11292            .await;
11293        reap.await.unwrap();
11294        assert_eq!(group_count(), 0);
11295        // A normal exit uses the same tree-cleanup path as a killed spawn.
11296        spec.program = PathBuf::from("/bin/true");
11297        spec.args.clear();
11298        let fresh_roster = ChildRoster::default();
11299        let mut short = spawn_child(
11300            &spec,
11301            None,
11302            None,
11303            &runtime.stderr_ring,
11304            None,
11305            &fresh_roster,
11306            Some(&placement),
11307        )
11308        .unwrap();
11309        short.wait().await.unwrap();
11310        assert_eq!(group_count(), 0);
11311        spec.module_id = "_".repeat(255);
11312        let mut long_id = spawn_child(
11313            &spec,
11314            None,
11315            None,
11316            &runtime.stderr_ring,
11317            None,
11318            &fresh_roster,
11319            Some(&placement),
11320        )
11321        .unwrap();
11322        long_id.wait().await.unwrap();
11323        assert_eq!(
11324            group_count(),
11325            0,
11326            "valid long module IDs must not exceed cgroup NAME_MAX"
11327        );
11328        fs::remove_dir(root.join("subc-modules")).unwrap();
11329        fs::remove_dir(root).unwrap();
11330        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
11331    }
11332
11333    #[test]
11334    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
11335        let path = Path::new("/definitely-missing-subc-cgroup");
11336        let mut command = Command::new("true");
11337        let error = apply_cgroup_placement(
11338            &mut command,
11339            &ModuleSpec {
11340                module_id: "broken-cgroup".to_string(),
11341                program: PathBuf::from("true"),
11342                args: Vec::new(),
11343                env: Vec::new(),
11344                reserved: false,
11345                reserved_prefixes: Vec::new(),
11346                protocol: ModuleProtocol::Subc,
11347                overlap: Default::default(),
11348            },
11349            path,
11350        )
11351        .expect_err("a parent cgroup open failure must reject the supervised spawn");
11352        let reason = error.to_string();
11353
11354        assert!(
11355            matches!(error, SuperviseError::Cgroup { .. }),
11356            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
11357        );
11358        assert!(
11359            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
11360            "parent cgroup open failure must name cgroup.procs: {reason}"
11361        );
11362    }
11363
11364    #[tokio::test]
11365    async fn reaping_a_child_removes_its_empty_module_cgroup() {
11366        let root = TestTempDir::new("supervisor-reap-cgroup");
11367        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11368        let placement = subc_cgroup::prepare_at(&root)
11369            .expect("prepare scratch cgroup root")
11370            .expect("scratch root has a cgroup.procs marker");
11371        let module_id = "reaped-module";
11372        let module = placement
11373            .module_path(module_id)
11374            .expect("create scratch module cgroup");
11375        let child = Command::new("true")
11376            .env("XDG_DATA_HOME", root.path())
11377            .env("XDG_RUNTIME_DIR", root.path())
11378            .env("XDG_CONFIG_HOME", root.path())
11379            .spawn()
11380            .expect("spawn short-lived child");
11381        let pid = child.id().expect("spawned child has pid");
11382        let mut child = SupervisedChild {
11383            child,
11384            protocol: ModuleProtocol::Subc,
11385            module_id: module_id.to_string(),
11386            cgroup_placement: Some(placement),
11387            stdout_pump: None,
11388            stderr_pump: None,
11389            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
11390            spawned_at_ms: 0,
11391            spawned_from: PathBuf::from("true"),
11392            spawned_file_identity: None,
11393            process_start_time: None,
11394            process_identity: None,
11395            pid,
11396            roster_guard: None,
11397        };
11398
11399        child.wait().await.expect("reap short-lived child");
11400
11401        assert!(
11402            !module.exists(),
11403            "reaping the supervised child must remove its empty cgroup"
11404        );
11405    }
11406
11407    #[test]
11408    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
11409        let root = TestTempDir::new("supervisor-non-empty-cgroup");
11410        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11411        let placement = subc_cgroup::prepare_at(&root)
11412            .expect("prepare scratch cgroup root")
11413            .expect("scratch root has a cgroup.procs marker");
11414        let module = placement
11415            .module_path("surviving-module")
11416            .expect("create scratch module cgroup");
11417        fs::write(module.join("surviving-process"), b"still present")
11418            .expect("make scratch cgroup non-empty");
11419        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
11420
11421        remove_module_cgroup(&placement, "surviving-module");
11422
11423        let logs = crate::router::test_log::captured_logs(&logs);
11424        assert!(
11425            module.exists(),
11426            "failed removal must leave the cgroup intact"
11427        );
11428        assert!(
11429            logs.contains("could not remove module cgroup after process exit; continuing teardown")
11430                && logs.contains("surviving-module"),
11431            "best-effort removal must report the failure without returning it: {logs}"
11432        );
11433    }
11434
11435    #[test]
11436    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
11437        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
11438        let reason = SuperviseError::Spawn {
11439            program: PathBuf::from("/bin/true"),
11440            source: io::Error::from_raw_os_error(13),
11441            cgroup_path: Some(cgroup_path.clone()),
11442        }
11443        .to_string();
11444
11445        assert!(
11446            reason.contains(&cgroup_path.display().to_string()),
11447            "a pre_exec spawn failure must name the cgroup path: {reason}"
11448        );
11449    }
11450}
11451
11452#[cfg(test)]
11453mod spawn_subscriber_lag_tests {
11454    use super::*;
11455
11456    /// A subscriber whose connection stops draining is dropped once its frame
11457    /// channel fills. The client must learn that from a terminal Error frame
11458    /// after the frames already queued for it, not from a stream that simply
11459    /// goes quiet.
11460    #[tokio::test]
11461    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
11462        let feed = SpawnEventFeed::default();
11463        feed.configure_incarnation("lag-incarnation".to_string());
11464        // A one-slot connection queue that nobody reads until the emits are
11465        // done: the forwarder parks on it and the subscriber channel fills.
11466        let (tx, mut rx) = mpsc::channel(1);
11467        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
11468            .expect("subscribe");
11469        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
11470        for index in 0..emitted {
11471            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
11472            // Let the forwarder take what it can so the fill point is the
11473            // subscriber channel, not a scheduling accident.
11474            tokio::task::yield_now().await;
11475        }
11476        assert_eq!(
11477            feed.subscriber_count(),
11478            0,
11479            "the lagged subscriber must be removed"
11480        );
11481
11482        let mut data = Vec::new();
11483        let mut last = None;
11484        loop {
11485            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
11486                .await
11487                .expect("the forwarder must finish once the subscriber is dropped");
11488            let Some(outbound) = next else { break };
11489            let frame = outbound.frame;
11490            if frame.header.ty == FrameType::StreamData {
11491                assert!(last.is_none(), "no data may follow the terminal frame");
11492                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
11493                data.push(event.cursor.seq);
11494            } else {
11495                assert!(last.is_none(), "exactly one terminal frame");
11496                last = Some(frame);
11497            }
11498        }
11499        assert!(!data.is_empty(), "queued frames drain before the terminal");
11500        for pair in data.windows(2) {
11501            assert_eq!(
11502                pair[1],
11503                pair[0] + 1,
11504                "queued frames arrive dense and in order"
11505            );
11506        }
11507        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
11508        assert_eq!(terminal.header.ty, FrameType::Error);
11509        assert_eq!(terminal.header.corr, 7);
11510        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
11511        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
11512        let detail = body.detail.expect("lagged error carries detail");
11513        assert_eq!(
11514            detail["first_undelivered_cursor"]["seq"],
11515            data.last().unwrap() + 1,
11516            "the named cursor is the first event the subscriber did not receive"
11517        );
11518        assert_eq!(
11519            detail["first_undelivered_cursor"]["daemon_incarnation"],
11520            "lag-incarnation"
11521        );
11522    }
11523}
11524
11525#[cfg(test)]
11526mod terminal_history_read_concurrency_tests {
11527    use super::*;
11528    use crate::terminal_journal::read_pause;
11529    use std::sync::mpsc as std_mpsc;
11530    use subc_test_support::TestTempDir;
11531
11532    fn journaled_ring(
11533        journal: &Arc<crate::terminal_journal::TerminalJournal>,
11534    ) -> Arc<Mutex<TerminalRing>> {
11535        Arc::new(Mutex::new(
11536            TerminalRing::new(TerminalRingConfig::default(), 1)
11537                .with_journal(Some(Arc::clone(journal))),
11538        ))
11539    }
11540
11541    fn crash(at_ms: u64) -> ExitReport {
11542        ExitReport {
11543            kind: ExitKind::Crash,
11544            code: Some(1),
11545            signal: None,
11546            at_ms,
11547        }
11548    }
11549
11550    /// Record an exit on another thread and report whether it finished within
11551    /// `bound`. The recorder thread is left running if it did not.
11552    fn record_within(
11553        module_id: &'static str,
11554        ring: &Arc<Mutex<TerminalRing>>,
11555        at_ms: u64,
11556        bound: Duration,
11557    ) -> bool {
11558        let ring = Arc::clone(ring);
11559        let (done, done_rx) = std_mpsc::channel();
11560        std::thread::spawn(move || {
11561            record_terminal(
11562                module_id,
11563                &ring,
11564                &SpawnEventFeed::default(),
11565                &crash(at_ms),
11566                TerminalDisposition::Restarting,
11567            );
11568            let _ = done.send(());
11569        });
11570        done_rx.recv_timeout(bound).is_ok()
11571    }
11572
11573    /// A history read in progress must not hold the journal writer (which every
11574    /// module's exit recording needs) or the module's own ring. Exits recorded
11575    /// while the read is paused complete promptly; the paused read answers as of
11576    /// the moment it started, and the next read has each exit exactly once.
11577    #[test]
11578    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
11579        let dir = TestTempDir::new("terminal-history-concurrent-read");
11580        let path = dir.join("terminals.jsonl");
11581        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
11582            path.clone(),
11583            "daemon".into(),
11584        ));
11585        let reader_ring = journaled_ring(&journal);
11586        let other_ring = journaled_ring(&journal);
11587        assert!(record_within(
11588            "reader-module",
11589            &reader_ring,
11590            10,
11591            Duration::from_secs(5)
11592        ));
11593
11594        let (started, release) = read_pause::install(&path);
11595        let reading = {
11596            let ring = Arc::clone(&reader_ring);
11597            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
11598        };
11599        started
11600            .recv_timeout(Duration::from_secs(5))
11601            .expect("the history read reached its pause");
11602
11603        let bound = Duration::from_secs(1);
11604        assert!(
11605            record_within("other-module", &other_ring, 20, bound),
11606            "another module's exit waited on a history read (journal writer held)"
11607        );
11608        assert!(
11609            record_within("reader-module", &reader_ring, 30, bound),
11610            "the read module's own exit waited on its history read (ring held)"
11611        );
11612
11613        drop(release);
11614        let paused = reading.join().unwrap();
11615        assert_eq!(
11616            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
11617            vec![10],
11618            "an exit recorded after the read began lands in neither half of it"
11619        );
11620        assert_eq!(paused.journal_skipped_lines, 0);
11621        assert_eq!(paused.journal_read_errors, 0);
11622
11623        let after = durable_terminal_history_of(&reader_ring, "reader-module");
11624        assert_eq!(
11625            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
11626            vec![10, 30],
11627            "the next read merges ring and journal with no duplicate"
11628        );
11629        assert_eq!(after.journal_skipped_lines, 0);
11630    }
11631}
11632
11633/// What a restart does with the exited process's stderr reader. These drive
11634/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
11635/// holds, so a reader that has not been scheduled by the bound is a controlled
11636/// input rather than something only a loaded machine produces.
11637#[cfg(test)]
11638mod stderr_settle_tests {
11639    use std::{
11640        future::Future,
11641        io,
11642        pin::Pin,
11643        sync::{Arc, Mutex},
11644        task::{Context, Poll},
11645        time::Duration,
11646    };
11647
11648    use tokio::{
11649        io::{AsyncRead, ReadBuf},
11650        sync::oneshot,
11651        time::Instant,
11652    };
11653
11654    use super::{settle_stderr_pump, StderrPump};
11655    use crate::stderr_tail::{
11656        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
11657    };
11658
11659    const BOUND: Duration = Duration::from_millis(250);
11660
11661    /// Yields `before`, then stays pending until the gate is released, then
11662    /// yields `after` and reaches EOF. The bytes after the gate were written
11663    /// by a process that has already exited; only the reader is behind.
11664    struct HeldReader {
11665        before: Option<Vec<u8>>,
11666        gate: Option<oneshot::Receiver<()>>,
11667        after: io::Cursor<Vec<u8>>,
11668    }
11669
11670    impl AsyncRead for HeldReader {
11671        fn poll_read(
11672            mut self: Pin<&mut Self>,
11673            cx: &mut Context<'_>,
11674            buf: &mut ReadBuf<'_>,
11675        ) -> Poll<io::Result<()>> {
11676            if let Some(bytes) = self.before.take() {
11677                buf.put_slice(&bytes);
11678                return Poll::Ready(Ok(()));
11679            }
11680            if let Some(gate) = self.gate.as_mut() {
11681                match Pin::new(gate).poll(cx) {
11682                    Poll::Pending => return Poll::Pending,
11683                    Poll::Ready(_) => self.gate = None,
11684                }
11685            }
11686            Pin::new(&mut self.after).poll_read(cx, buf)
11687        }
11688    }
11689
11690    struct DiscardSink;
11691
11692    impl OutputSink for DiscardSink {
11693        fn write_line(&mut self, _line: &[u8]) {}
11694    }
11695
11696    fn line(text: &str) -> TailEntry {
11697        TailEntry::Line {
11698            text: text.to_string(),
11699            truncated: false,
11700            at_ms: None,
11701        }
11702    }
11703
11704    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
11705        ring.lock().unwrap()
11706    }
11707
11708    /// Start a reader for a new process generation that delivers `before`
11709    /// immediately and `after` only once the returned sender fires (or is
11710    /// dropped).
11711    fn held_pump(
11712        ring: &Arc<Mutex<StderrRing>>,
11713        before: &str,
11714        after: &str,
11715    ) -> (StderrPump, oneshot::Sender<()>) {
11716        let generation = lock(ring).begin_process();
11717        let (release, gate) = oneshot::channel();
11718        let reader = HeldReader {
11719            before: Some(before.as_bytes().to_vec()),
11720            gate: Some(gate),
11721            after: io::Cursor::new(after.as_bytes().to_vec()),
11722        };
11723        let task = tokio::spawn(pump_stderr_to(
11724            reader,
11725            Arc::clone(ring),
11726            generation,
11727            DiscardSink,
11728        ));
11729        (StderrPump { task, generation }, release)
11730    }
11731
11732    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
11733        for _ in 0..1000 {
11734            if done(&lock(ring)) {
11735                return;
11736            }
11737            tokio::time::sleep(Duration::from_millis(1)).await;
11738        }
11739        panic!(
11740            "ring never reached the expected state: {:?}",
11741            lock(ring).snapshot(None, None)
11742        );
11743    }
11744
11745    #[tokio::test(start_paused = true)]
11746    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
11747        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
11748        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
11749
11750        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
11751        let before_release = lock(&ring).snapshot(None, None);
11752        assert!(
11753            matches!(before_release.capture, CaptureState::Incomplete { .. }),
11754            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
11755        );
11756
11757        // The restart: the next process starts and writes before the old
11758        // reader catches up.
11759        let next = lock(&ring).begin_process();
11760        lock(&ring).push_line_from(next, "next process booting");
11761        release.send(()).unwrap();
11762        wait_until(&ring, |ring| {
11763            ring.snapshot(None, None).capture == CaptureState::Captured
11764        })
11765        .await;
11766
11767        assert_eq!(
11768            untimed(lock(&ring).snapshot(None, None).entries),
11769            vec![
11770                line("booting"),
11771                line("config error: missing storage"),
11772                TailEntry::ProcessStart,
11773                line("next process booting"),
11774            ],
11775            "the crash's last line must survive a slow reader and stay in the crashed process's section"
11776        );
11777    }
11778
11779    #[tokio::test(start_paused = true)]
11780    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
11781    ) {
11782        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
11783        // `_held` is never fired: a descendant keeps the pipe open for the
11784        // whole test.
11785        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
11786
11787        let started = Instant::now();
11788        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
11789        assert_eq!(
11790            started.elapsed(),
11791            BOUND,
11792            "the restart must wait exactly the bound for a pipe that stays open, no longer"
11793        );
11794
11795        let next = lock(&ring).begin_process();
11796        lock(&ring).push_line_from(next, "next process booting");
11797        tokio::time::sleep(Duration::from_secs(60)).await;
11798
11799        let snapshot = lock(&ring).snapshot(None, None);
11800        match &snapshot.capture {
11801            CaptureState::Incomplete { reason } => assert!(
11802                reason.contains("had not reached EOF") && reason.contains("250ms"),
11803                "the reason must say what is missing and after how long: {reason}"
11804            ),
11805            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
11806        }
11807        assert_eq!(
11808            untimed(snapshot.entries),
11809            vec![
11810                line("parent exiting"),
11811                TailEntry::ProcessStart,
11812                line("next process booting"),
11813            ]
11814        );
11815    }
11816
11817    #[tokio::test(start_paused = true)]
11818    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
11819        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
11820        let (pump, release) = held_pump(&ring, "one\n", "two\n");
11821        release.send(()).unwrap();
11822
11823        settle_stderr_pump("clean", &ring, pump, BOUND).await;
11824
11825        let snapshot = lock(&ring).snapshot(None, None);
11826        assert_eq!(snapshot.capture, CaptureState::Captured);
11827        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
11828    }
11829}
11830
11831/// Containment of a module's process tree (issue #109).
11832///
11833/// The behaviour these defend against is a module helper surviving its module:
11834/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
11835/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
11836/// compounds it.
11837///
11838/// They run against the SUPERVISOR rather than the job-object crate because the
11839/// claim is about teardown: a crate-level test proves a job can reap a tree, not
11840/// that the daemon's drain path reaches it.
11841///
11842/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
11843/// lane there is a separate containment path with its own tests.
11844#[cfg(all(test, windows))]
11845mod job_containment_tests {
11846    use super::*;
11847    use std::{
11848        path::{Path, PathBuf},
11849        sync::{Arc, Mutex},
11850        time::{Duration, Instant},
11851    };
11852    use subc_test_support::TestTempDir;
11853
11854    /// The stub, expected beside this test executable.
11855    ///
11856    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
11857    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
11858    /// failure then reads as a broken test rather than an unbuilt dependency.
11859    fn stub_path() -> PathBuf {
11860        let mut path = std::env::current_exe().expect("current_exe available in tests");
11861        path.pop();
11862        path.pop();
11863        path.push("fake-aft-stub.exe");
11864        assert!(
11865            path.exists(),
11866            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
11867             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
11868            path.display()
11869        );
11870        path
11871    }
11872
11873    /// Poll for the grandchild pid the stub records, and parse it.
11874    fn read_grandchild_pid(path: &Path) -> u32 {
11875        let deadline = Instant::now() + Duration::from_secs(10);
11876        loop {
11877            if let Ok(contents) = std::fs::read_to_string(path) {
11878                if let Ok(pid) = contents.trim().parse() {
11879                    return pid;
11880                }
11881            }
11882            assert!(
11883                Instant::now() < deadline,
11884                "the stub never recorded a grandchild pid at {}",
11885                path.display()
11886            );
11887            std::thread::sleep(Duration::from_millis(10));
11888        }
11889    }
11890
11891    /// Everything one fixture run needs, so the two tests below differ in exactly
11892    /// one place: whether the child is contained.
11893    struct Fixture {
11894        _dir: TestTempDir,
11895        module_id: String,
11896        grandchild: u32,
11897        child: Option<SupervisedChild>,
11898        registry: Arc<Registry>,
11899        snapshot: Arc<Mutex<SupervisorSnapshot>>,
11900        terminal_ring: Arc<Mutex<TerminalRing>>,
11901        spawn_events: SpawnEventFeed,
11902    }
11903
11904    fn fixture(label: &str, module_id: &str) -> Fixture {
11905        let dir = TestTempDir::new(label);
11906        let pid_file = dir.join("grandchild.pid");
11907        let supervisor = Supervisor::new(
11908            Arc::new(Registry::default()),
11909            RestartPolicy::new(3, Duration::ZERO),
11910        );
11911        let runtime = supervisor.runtime_config();
11912        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11913        let spec = ModuleSpec {
11914            module_id: module_id.to_string(),
11915            program: stub_path(),
11916            // Zero args deliberately: a `--subc` argument would make the stub dial
11917            // a daemon that is not there, and the failure would land in the same
11918            // stderr ring this fixture exists to keep quiet.
11919            args: Vec::new(),
11920            env: vec![
11921                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
11922                (
11923                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
11924                    pid_file.display().to_string(),
11925                ),
11926            ],
11927            reserved: false,
11928            reserved_prefixes: Vec::new(),
11929            protocol: ModuleProtocol::Subc,
11930            overlap: Default::default(),
11931        };
11932        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
11933            .expect("spawn the supervised fixture");
11934        let grandchild = read_grandchild_pid(&pid_file);
11935        Fixture {
11936            _dir: dir,
11937            module_id: module_id.to_string(),
11938            grandchild,
11939            child: Some(child),
11940            registry: Arc::new(Registry::default()),
11941            snapshot,
11942            terminal_ring: Arc::clone(&runtime.terminal_ring),
11943            spawn_events: SpawnEventFeed::default(),
11944        }
11945    }
11946
11947    impl Fixture {
11948        /// Drain through the supervisor's own teardown path.
11949        async fn drain(&mut self) {
11950            let child = self
11951                .child
11952                .take()
11953                .expect("the fixture child is still present");
11954            drain_child_to_state(
11955                &self.module_id,
11956                ModuleProtocol::Subc,
11957                // No forwarding table in this fixture, so nothing reaches the
11958                // child over a connection.
11959                StopNotice::NotSent,
11960                &self.registry,
11961                None,
11962                &self.snapshot,
11963                &self.terminal_ring,
11964                &self.spawn_events,
11965                child,
11966                Duration::from_millis(500),
11967                ModuleState::Stopped,
11968                Some(false),
11969            )
11970            .await
11971            .expect("drain the supervised fixture");
11972        }
11973    }
11974
11975    /// Teardown reaps the grandchild, not merely the direct child.
11976    ///
11977    /// This is the assertion the change exists for. Before containment the
11978    /// grandchild survived: it is a separate process, and `start_kill` is
11979    /// `TerminateProcess` scoped to one pid.
11980    #[tokio::test]
11981    async fn teardown_reaps_the_grandchild() {
11982        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
11983        let grandchild = fixture.grandchild;
11984
11985        assert!(
11986            subc_jobobject::process_exists(grandchild),
11987            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
11988        );
11989
11990        fixture.drain().await;
11991
11992        assert!(
11993            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
11994            "grandchild {grandchild} outlived module teardown: the tree was not contained"
11995        );
11996    }
11997
11998    /// The mutation control: with containment withheld, the grandchild survives
11999    /// the same kill.
12000    ///
12001    /// This is the defect reproduction from #109 — a direct-child kill reaches
12002    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
12003    /// supervisor because `spawn_and_mark_running` now always contains on
12004    /// Windows, which is the point: there is no longer a path that spawns
12005    /// uncontained, so the control has to construct one.
12006    ///
12007    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
12008    /// grandchild ever dies here, that test is passing for a reason unrelated to
12009    /// the job object and the containment claim is unproven.
12010    #[test]
12011    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12012        let dir = TestTempDir::new("teardown-uncontained");
12013        let pid_file = dir.join("grandchild.pid");
12014        let mut child = std::process::Command::new(stub_path())
12015            .env("FAKE_AFT_NEVER_CONNECT", "1")
12016            .env(
12017                "FAKE_AFT_GRANDCHILD_PID_FILE",
12018                pid_file.display().to_string(),
12019            )
12020            .stdin(std::process::Stdio::null())
12021            .stdout(std::process::Stdio::null())
12022            .stderr(std::process::Stdio::null())
12023            .spawn()
12024            .expect("spawn the uncontained fixture");
12025        let grandchild = read_grandchild_pid(&pid_file);
12026
12027        // Exactly what the pre-fix teardown did: kill the direct child.
12028        child.kill().expect("kill the direct child");
12029        let _ = child.wait();
12030
12031        assert!(
12032            subc_jobobject::process_exists(grandchild),
12033            "grandchild {grandchild} died with the direct child, so this control no longer \
12034             distinguishes contained from uncontained teardown and the regression test is \
12035             passing vacuously"
12036        );
12037
12038        // The orphan this control demonstrates is the leak the fix prevents, so
12039        // the control must not leave one behind.
12040        kill_tree(grandchild);
12041    }
12042
12043    /// Crash durability: closing the containment handle reaps the tree with no
12044    /// teardown code running at all.
12045    ///
12046    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
12047    /// call anything — and it is why containment is a kernel property of the
12048    /// handle rather than a step in the drain. Discovered by getting the
12049    /// mutation control wrong: clearing `job` to "disable" containment instead
12050    /// killed the tree, which is the guarantee, not a mistake.
12051    #[tokio::test]
12052    async fn dropping_containment_reaps_the_grandchild() {
12053        let mut fixture = fixture("drop-containment", "tree-drop");
12054        let grandchild = fixture.grandchild;
12055
12056        assert!(subc_jobobject::process_exists(grandchild));
12057
12058        // No `drain` call, no kill: dropping the handle is the entire mechanism.
12059        fixture.child.as_mut().expect("child present").job = None;
12060
12061        assert!(
12062            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12063            "grandchild {grandchild} survived the containment handle closing, so a daemon \
12064             crash would leave the tree behind"
12065        );
12066    }
12067
12068    /// Kill a pid and its tree, then confirm it is gone.
12069    fn kill_tree(pid: u32) {
12070        let _ = std::process::Command::new("taskkill.exe")
12071            .args(["/PID", &pid.to_string(), "/T", "/F"])
12072            .stdin(std::process::Stdio::null())
12073            .stdout(std::process::Stdio::null())
12074            .stderr(std::process::Stdio::null())
12075            .status();
12076        assert!(
12077            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12078            "could not clean up grandchild {pid}"
12079        );
12080    }
12081}
12082
12083/// The daemon's real spawn path hands a subc-wire child its launch nonce on
12084/// descriptor 3, without an environment copy. The shell records the nonce
12085/// and its environment after exec so these tests observe the real handover.
12086#[cfg(all(test, unix))]
12087mod launch_nonce_descriptor_tests {
12088    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
12089    use crate::stderr_tail::{StderrRing, StderrTailConfig};
12090    use std::{
12091        path::PathBuf,
12092        sync::{Arc, Mutex},
12093        time::{Duration, Instant},
12094    };
12095    use subc_test_support::TestTempDir;
12096
12097    async fn probe(role: super::SpawnRole) {
12098        let scratch = TestTempDir::new("launch-nonce-descriptor");
12099        let fd_copy = scratch.join("from-descriptor");
12100        let env_copy = scratch.join("environment");
12101        let script = format!(
12102            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
12103            fd = fd_copy.display(), env = env_copy.display(),
12104        );
12105        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
12106        let spec = ModuleSpec {
12107            module_id: "nonce-descriptor-probe".to_string(),
12108            program: PathBuf::from("/bin/sh"),
12109            args: vec!["-c".to_string(), script],
12110            env: vec![
12111                xdg("XDG_DATA_HOME"),
12112                xdg("XDG_RUNTIME_DIR"),
12113                xdg("XDG_CONFIG_HOME"),
12114                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
12115            ],
12116            reserved: true,
12117            reserved_prefixes: Vec::new(),
12118            protocol: ModuleProtocol::Subc,
12119            overlap: Default::default(),
12120        };
12121        let handle = SupervisorHandle::new();
12122        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12123        let roster = ChildRoster::default();
12124        let child = super::spawn_child_in_slot(
12125            &spec,
12126            None,
12127            Some(&handle),
12128            &ring,
12129            None,
12130            &roster,
12131            #[cfg(target_os = "linux")]
12132            None,
12133            role,
12134            matches!(role, super::SpawnRole::SwapCandidate),
12135        )
12136        .expect("spawn probe");
12137        let deadline = Instant::now() + Duration::from_secs(10);
12138        while !(fd_copy.exists() && env_copy.exists()) {
12139            assert!(Instant::now() < deadline, "probe never wrote its copies");
12140            tokio::time::sleep(Duration::from_millis(20)).await;
12141        }
12142        let nonce = std::fs::read_to_string(fd_copy).unwrap();
12143        assert!(!nonce.is_empty());
12144        let environment = std::fs::read_to_string(env_copy).unwrap();
12145        assert!(environment
12146            .lines()
12147            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
12148        let copy = environment
12149            .lines()
12150            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
12151        assert_eq!(
12152            copy, None,
12153            "Unix children must never receive the environment nonce"
12154        );
12155        if matches!(role, super::SpawnRole::Plain) {
12156            assert_eq!(
12157                handle.spawn_nonce(&spec.module_id).as_deref(),
12158                Some(nonce.as_str())
12159            );
12160        }
12161        drop(child);
12162    }
12163
12164    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12165    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
12166        probe(super::SpawnRole::Plain).await;
12167    }
12168
12169    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12170    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
12171        probe(super::SpawnRole::SwapCandidate).await;
12172    }
12173}
12174
12175#[cfg(all(test, target_os = "linux"))]
12176mod cgroup_containment_tests {
12177    use super::*;
12178    use subc_test_support::TestTempDir;
12179
12180    fn running(pid: u32) -> bool {
12181        // An orphan can remain a zombie until the container init reaps it.
12182        std::fs::read_to_string(format!("/proc/{pid}/stat"))
12183            .ok()
12184            .and_then(|stat| {
12185                stat.rsplit_once(") ")
12186                    .map(|(_, rest)| rest.starts_with('Z'))
12187            })
12188            .is_some_and(|zombie| !zombie)
12189    }
12190
12191    #[tokio::test]
12192    async fn linux_teardown_reaps_the_grandchild() {
12193        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
12194    }
12195
12196    #[tokio::test]
12197    async fn linux_shutdown_straggler_reaps_the_grandchild() {
12198        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
12199    }
12200
12201    async fn teardown_tree(test_name: &str, shutdown: bool) {
12202        let dir = TestTempDir::new(test_name);
12203        let root = PathBuf::from(format!(
12204            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
12205            std::process::id(),
12206            unix_ms_now()
12207        ));
12208        if let Err(error) = std::fs::create_dir(&root) {
12209            assert!(
12210                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12211                "required cgroup test cannot execute: {error}"
12212            );
12213            eprintln!(
12214                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
12215                root.display()
12216            );
12217            return;
12218        }
12219        let placement = subc_cgroup::prepare_at(&root)
12220            .expect("prepare isolated kernel cgroup")
12221            .expect("isolated cgroup is delegated");
12222        let module_id = "tree-teardown";
12223        let module = placement
12224            .module_path(module_id)
12225            .expect("create isolated module cgroup");
12226        if !module.join("cgroup.kill").exists() {
12227            std::fs::remove_dir(&module).unwrap();
12228            std::fs::remove_dir(root.join("subc-modules")).unwrap();
12229            std::fs::remove_dir(&root).unwrap();
12230            assert!(
12231                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12232                "required cgroup.kill interface unavailable"
12233            );
12234            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
12235            return;
12236        }
12237        let supervisor = Supervisor::new(
12238            Arc::new(Registry::default()),
12239            RestartPolicy::new(3, Duration::ZERO),
12240        )
12241        .with_cgroup_placement(Some(placement));
12242        let mut runtime = supervisor.runtime_config();
12243        runtime.child_roster = runtime
12244            .child_roster
12245            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
12246        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12247        let pid_file = dir.join("grandchild.pid");
12248        let spec = ModuleSpec {
12249            module_id: module_id.to_string(),
12250            program: PathBuf::from("/bin/sh"),
12251            args: vec![
12252                "-c".into(),
12253                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
12254                "fixture".into(),
12255                pid_file.display().to_string(),
12256            ],
12257            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
12258                .into_iter()
12259                .map(|key| (key.to_string(), dir.display().to_string()))
12260                .collect(),
12261            reserved: false,
12262            reserved_prefixes: Vec::new(),
12263            protocol: ModuleProtocol::None,
12264            overlap: Default::default(),
12265        };
12266        let child =
12267            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
12268        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
12269        let grandchild: u32 = loop {
12270            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
12271                if let Ok(pid) = contents.trim().parse() {
12272                    break pid;
12273                }
12274            }
12275            assert!(
12276                tokio::time::Instant::now() < deadline,
12277                "grandchild pid was not recorded"
12278            );
12279            tokio::time::sleep(Duration::from_millis(10)).await;
12280        };
12281        assert!(
12282            running(grandchild),
12283            "grandchild must be alive before teardown"
12284        );
12285        if shutdown {
12286            let mut child = child;
12287            crate::child_roster::end_children_for_daemon_shutdown(
12288                &runtime.child_roster,
12289                false,
12290                std::future::pending(),
12291            )
12292            .await;
12293            child.wait().await.expect("reap shutdown straggler");
12294        } else {
12295            drain_child_to_state(
12296                module_id,
12297                ModuleProtocol::None,
12298                StopNotice::NotSent,
12299                &Registry::default(),
12300                None,
12301                &snapshot,
12302                &runtime.terminal_ring,
12303                &SpawnEventFeed::default(),
12304                child,
12305                Duration::from_millis(100),
12306                ModuleState::Stopped,
12307                Some(false),
12308            )
12309            .await
12310            .expect("real supervisor teardown");
12311        }
12312        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
12313        while running(grandchild) && tokio::time::Instant::now() < deadline {
12314            tokio::time::sleep(Duration::from_millis(10)).await;
12315        }
12316        let survived = running(grandchild);
12317        // Kill a surviving grandchild so a failed test does not leave it behind.
12318        if survived {
12319            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
12320            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
12321            tokio::time::sleep(Duration::from_millis(100)).await;
12322        }
12323        if module.exists() {
12324            std::fs::remove_dir(&module).expect("remove empty module cgroup");
12325        }
12326        std::fs::remove_dir(root.join("subc-modules")).unwrap();
12327        std::fs::remove_dir(&root).unwrap();
12328        assert!(
12329            !survived,
12330            "grandchild {grandchild} outlived module teardown"
12331        );
12332        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
12333    }
12334}