Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    let path = std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture");
134    // Without the fixture every macOS spawn is refused, and the tests that
135    // spawn fail later as a module in state Failed, which names the wrong
136    // cause. `cargo test -p subc-daemon --lib` alone does not build it.
137    assert!(
138        path.exists(),
139        "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140         --bins --features test-support` or `cargo test -p subc-daemon` first",
141        path.display()
142    );
143    path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148    use std::io::Read;
149    let mut probe = std::process::Command::new(path)
150        .args(["__disclaim-exec", "--probe"])
151        .stdin(Stdio::null())
152        .stdout(Stdio::piped())
153        .stderr(Stdio::piped())
154        .spawn()
155        .map_err(|error| {
156            format!(
157                "privacy trampoline probe failed for {}: {error}",
158                path.display()
159            )
160        })?;
161    let deadline = std::time::Instant::now() + Duration::from_secs(5);
162    let status = loop {
163        match probe.try_wait() {
164            Ok(Some(status)) => break status,
165            Ok(None) if std::time::Instant::now() < deadline => {
166                std::thread::sleep(Duration::from_millis(5))
167            }
168            result => {
169                let _ = probe.kill();
170                let _ = probe.wait();
171                return Err(format!(
172                    "privacy trampoline probe failed or timed out for {}: {result:?}",
173                    path.display()
174                ));
175            }
176        }
177    };
178    let mut answer = String::new();
179    if let Some(stdout) = probe.stdout.take() {
180        let _ = stdout.take(256).read_to_string(&mut answer);
181    }
182    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183        return Ok(());
184    }
185    let mut diagnostic = String::new();
186    if let Some(stderr) = probe.stderr.take() {
187        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188    }
189    let cause = diagnostic
190        .trim()
191        .strip_prefix("ck-subc: own privacy identity refused: ")
192        .unwrap_or("binary does not implement the privacy trampoline protocol");
193    Err(format!(
194        "{cause}: probe of {} exited {status}",
195        path.display()
196    ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201    spec: &ModuleSpec,
202    roster: &ChildRoster,
203) -> Result<
204    (
205        Command,
206        Option<PrivacyExec>,
207        subc_os::privacy_identity::ExecAcknowledgement,
208    ),
209    SuperviseError,
210> {
211    let failure = |cause: String| {
212        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213        SuperviseError::Spawn {
214            program: spec.program.clone(),
215            source: io::Error::other(cause),
216            cgroup_path: None,
217        }
218    };
219    let trampoline = roster.privacy_trampoline().map_err(failure)?;
220    // Resolve PATH with the same environment the Command will receive. For
221    // scripts retain the existing orphan-identity rule: the kernel chooses
222    // the interpreter, and its observed image is the one recorded. Do not
223    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
224    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225        let path = spec
226            .env
227            .iter()
228            .find(|(key, _)| key == "PATH")
229            .map(|(_, value)| std::ffi::OsString::from(value))
230            .or_else(|| std::env::var_os("PATH"))
231            .unwrap_or_else(|| "/usr/bin:/bin".into());
232        std::env::split_paths(&path)
233            .map(|dir| dir.join(&spec.program))
234            .find(|path| path.is_file())
235            .unwrap_or_else(|| spec.program.clone())
236    } else {
237        spec.program.clone()
238    };
239    let expected = subc_os::file_identity(&program);
240    let trampoline_image = subc_os::file_identity(&trampoline);
241    let script = {
242        use std::io::Read;
243        let mut prefix = [0u8; 2];
244        std::fs::File::open(&program)
245            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246    };
247    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248        return Err(failure(
249            "privacy identity module executable is missing or is the trampoline itself".to_string(),
250        ));
251    }
252    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253        .map_err(|error| failure(error.to_string()))?;
254    let reader =
255        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256    let mut command = Command::new(&trampoline);
257    command
258        .arg("__disclaim-exec")
259        .arg(ack.fd().to_string())
260        .arg(&program);
261    ack.install(command.as_std_mut());
262    Ok((
263        command,
264        Some(PrivacyExec {
265            reader,
266            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267            expected,
268            trampoline: trampoline_image,
269            script,
270            module_id: spec.module_id.clone(),
271        }),
272        ack,
273    ))
274}
275
276struct SupervisedChild {
277    child: Child,
278    #[cfg(target_os = "macos")]
279    privacy_exec: Option<PrivacyExec>,
280    /// Set once this launch's exec acknowledgement confirms the module image.
281    /// On macOS the pid first runs the `ck-subc` launch trampoline (see
282    /// `subc_os::privacy_identity`), which then replaces itself with the
283    /// module. The supervisor owns and can kill that pid from spawn, but
284    /// status readers report it only after this latch is set, so nothing
285    /// reports the trampoline's image as the module's.
286    #[cfg(target_os = "macos")]
287    report_ready: Arc<OnceLock<()>>,
288    /// Refusal before the module image was accepted, retained for terminal records.
289    spawn_failure: Option<String>,
290    /// The protocol this process was launched with. A reload can store a new
291    /// launch spec with a different protocol, but that takes effect only at the
292    /// next spawn, so this process keeps being handled by the protocol it
293    /// actually speaks.
294    protocol: ModuleProtocol,
295    /// This process's cgroup name: a bounded module/slot label followed by a
296    /// spawn suffix unique to this process (when cgroup placement is on). A
297    /// retired process in a slot may still be draining when a later one is
298    /// spawned into that slot, so the suffix keeps the later process out of
299    /// the retired one's cgroup, which is the domain a kill applies to.
300    #[cfg(target_os = "linux")]
301    module_id: String,
302    #[cfg(target_os = "linux")]
303    cgroup_placement: Option<subc_cgroup::Placement>,
304    /// The job that contains this child and every process it spawns (issue #109).
305    ///
306    /// Dropping this handle is what reaps a surviving tree when no supervisor
307    /// code runs — a daemon crash — because the job carries
308    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
309    ///
310    /// That limit is not crash-only, and the difference is worth knowing: a
311    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
312    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
313    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
314    /// module at once. Before this change they survived that, saw EOF on the
315    /// control socket, and ran their own teardown; Unix keeps that path
316    /// deliberately, so a module can seal a WAL or close a capture rather than
317    /// be killed mid-write. So this trades graceful teardown on every Windows
318    /// daemon stop for containment on a crash, which is the right way round
319    /// today: orphaned GPU workers are a reported, recurring problem, and the
320    /// modules that write most heavily do not run on Windows.
321    ///
322    /// The fix is a real Windows stop path — the daemon draining before it
323    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
324    /// reaches only what the drain left behind, which is what it should reach.
325    #[cfg(windows)]
326    job: Option<subc_jobobject::JobObject>,
327    stdout_pump: Option<JoinHandle<()>>,
328    stderr_pump: Option<StderrPump>,
329    stderr_ring: Arc<Mutex<StderrRing>>,
330    spawned_at_ms: u64,
331    spawned_from: PathBuf,
332    spawned_file_identity: Option<SpawnedFileIdentity>,
333    process_start_time: Option<u64>,
334    process_identity: Option<ProcessIdentity>,
335    pid: u32,
336    /// This process's entry in the daemon's child roster, released when the
337    /// process is reaped or this handle is dropped.
338    roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342    fn id(&self) -> Option<u32> {
343        Some(self.pid)
344    }
345
346    fn process_identity(&self) -> Option<ProcessIdentity> {
347        self.process_identity
348    }
349
350    async fn wait(&mut self) -> io::Result<ExitStatus> {
351        #[cfg(target_os = "macos")]
352        self.confirm_privacy_exec().await;
353        // The roster entry is NOT released here. A daemon shutdown waits for the
354        // roster to empty and then exits the process, so releasing at the reap
355        // let it exit before the exit handler wrote this child's terminal record
356        // (the stderr drain and snapshot update sit in between), and the
357        // shutdown's own `daemon_shutdown` record was intermittently lost. The
358        // caller releases it after recording the exit (`release_roster`), and
359        // dropping the handle releases it too.
360        let result = self.child.wait().await;
361        #[cfg(target_os = "linux")]
362        if result.is_ok() {
363            if let Some(placement) = self.cgroup_placement.as_ref() {
364                cleanup_reaped_cgroup(placement, &self.module_id).await;
365                // Keep ownership while awaiting kernel population changes: a
366                // drain timeout may cancel this wait and then escalate/reap.
367                self.cgroup_placement = None;
368            }
369        }
370        result
371    }
372
373    #[cfg(target_os = "macos")]
374    async fn confirm_privacy_exec(&mut self) {
375        let Some(pending) = &mut self.privacy_exec else {
376            return;
377        };
378        let result = tokio::time::timeout_at(pending.deadline, async {
379            let mut record = Vec::new();
380            loop {
381                let mut ready = pending.reader.readable().await?;
382                let read = ready.try_io(|reader| {
383                    use std::io::Read;
384                    let mut reader = reader.get_ref();
385                    let mut buffer = [0u8; 256];
386                    reader.read(&mut buffer).map(|count| (count, buffer))
387                });
388                match read {
389                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390                    Ok(Ok((count, buffer))) => {
391                        if record.len() + count > 1024 {
392                            return Err(io::Error::other(
393                                "privacy exec refusal record is too long",
394                            ));
395                        }
396                        record.extend_from_slice(&buffer[..count]);
397                    }
398                    Ok(Err(error)) => return Err(error),
399                    Err(_) => continue,
400                }
401            }
402        })
403        .await;
404        // Keep the reader in self across await: select cancellation must not
405        // discard the handshake or reset its original five-second deadline.
406        let pending = self.privacy_exec.as_ref().expect("pending exec");
407        let cause = match result {
408            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409            Ok(Err(error)) => Some(format!(
410                "privacy identity exec acknowledgement failed: {error}"
411            )),
412            Ok(Ok(record)) if !record.is_empty() => Some(
413                std::str::from_utf8(&record)
414                    .ok()
415                    .and_then(|record| {
416                        record
417                            .trim()
418                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419                    })
420                    .filter(|cause| !cause.is_empty())
421                    .unwrap_or("invalid privacy exec refusal record")
422                    .to_string(),
423            ),
424            Ok(Ok(_)) => match self.child.try_wait() {
425                // Empty EOF is the exec acknowledgement. A real module may exit
426                // immediately, including with a reserved trampoline status; no
427                // image is admitted, and its ordinary exit contract stays intact.
428                Ok(Some(_status)) => None,
429                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430                Ok(None) => {
431                    let image = observe_spawned_image(self.pid);
432                    if let Some(image) = image.filter(|image| {
433                        image.executable.is_some()
434                            && image.executable != pending.trampoline
435                            && (image.executable == pending.expected || pending.script)
436                    }) {
437                        if let Some(guard) = &self.roster_guard {
438                            guard.confirm_executable(image);
439                        }
440                        let _ = self.report_ready.set(());
441                        info!(module_id = %pending.module_id, pid = self.pid,
442                            "module spawned with own privacy identity (responsibility disclaimed)");
443                        None
444                    } else if image.is_none()
445                        || image.is_some_and(|image| image.executable.is_none())
446                    {
447                        // A process can exit between try_wait and the kernel
448                        // image read. Empty EOF already acknowledged exec, so
449                        // preserve that module's ordinary exit rather than
450                        // mislabel a disappearing image as trampoline refusal.
451                        // Keep pending in self across await so cancellation does
452                        // not discard validation or reset its original deadline.
453                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454                            Ok(Ok(_status)) => None,
455                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457                        }
458                    } else {
459                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460                    }
461                }
462            },
463        };
464        let pending = self.privacy_exec.take().expect("pending exec");
465        if let Some(cause) = cause {
466            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467            self.spawn_failure = Some(cause);
468            // No image is admitted on failure. Reach the entire fresh process
469            // group, including a module which spawned a helper before refusal.
470            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472            }
473            let _ = self.child.start_kill();
474        }
475    }
476
477    /// Releases this child's daemon-shutdown roster entry once its exit has
478    /// been recorded. The pid is already reaped and free for reuse, so the
479    /// entry must not outlive the record any longer than that.
480    fn release_roster(&mut self) {
481        self.roster_guard = None;
482    }
483
484    /// Kill the child and, where containment is available, its process tree.
485    ///
486    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
487    /// helper process leaked the helper — the Synapse embedding module's CUDA
488    /// worker holds the GPU allocation, so the leak cost VRAM until the next
489    /// restart of something else. Terminating the job reaches grandchildren that
490    /// a tree walk cannot, including one whose parent has already exited and
491    /// been reparented away.
492    ///
493    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
494    /// direct-child kill still decides the outcome, so containment can never
495    /// change whether a module is reported as stopped.
496    fn start_kill(&mut self) -> io::Result<()> {
497        #[cfg(windows)]
498        if let Some(job) = &self.job {
499            if let Err(error) = job.terminate() {
500                debug!(
501                    error = %error,
502                    "job termination failed; the direct-child kill still owns the outcome"
503                );
504            }
505        }
506        #[cfg(target_os = "linux")]
507        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508        self.child.start_kill()
509    }
510
511    async fn drain_stderr(&mut self, module_id: &str) {
512        if let Some(mut pump) = self.stdout_pump.take() {
513            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514                Ok(Ok(())) => {}
515                Ok(Err(error)) => {
516                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517                }
518                Err(_) => {
519                    pump.abort();
520                    warn!(
521                        module_id,
522                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523                        "stdout pump did not drain before restart; stopped it before the next process"
524                    );
525                }
526            }
527        }
528
529        let Some(pump) = self.stderr_pump.take() else {
530            return;
531        };
532        settle_stderr_pump(
533            module_id,
534            &self.stderr_ring,
535            pump,
536            STDERR_PUMP_DRAIN_TIMEOUT,
537        )
538        .await;
539    }
540}
541
542/// The reader task for one process's stderr, with the ring generation its
543/// lines are attributed to.
544struct StderrPump {
545    task: JoinHandle<()>,
546    generation: u64,
547}
548
549/// Retire an exited process's stderr reader and wait up to `bound` for it to
550/// reach EOF. A reader still running at the bound is detached, not stopped: it
551/// keeps filling the exited process's section of the ring until its pipe
552/// closes, and the tail reads `Incomplete` until then. See
553/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
554async fn settle_stderr_pump(
555    module_id: &str,
556    ring: &Arc<Mutex<StderrRing>>,
557    pump: StderrPump,
558    bound: Duration,
559) {
560    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561    let StderrPump {
562        mut task,
563        generation,
564    } = pump;
565    lock().retire_pump(generation);
566    match timeout(bound, &mut task).await {
567        Ok(Ok(())) => {}
568        Ok(Err(err)) => {
569            let mut ring = lock();
570            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571            ring.finish_pump(generation);
572            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573        }
574        Err(_) => {
575            // Dropping the handle detaches the task; it ends at EOF on its pipe.
576            drop(task);
577            lock().mark_pump_late(
578                generation,
579                format!(
580                    "stderr of the exited process had not reached EOF {bound:?} after it was \
581                     retired (a descendant may still hold the pipe open); lines it still \
582                     writes are kept in that process's section"
583                ),
584            );
585            warn!(
586                module_id,
587                waited = ?bound,
588                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589            );
590        }
591    }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596    EVENTS.get_or_init(|| {
597        let (sender, _receiver) = watch::channel(0);
598        sender
599    })
600}
601
602pub(crate) fn notify_registration_release() {
603    let events = registration_release_events();
604    let next_generation = (*events.borrow()).wrapping_add(1);
605    events.send_replace(next_generation);
606}
607
608/// How to launch one singleton module process.
609#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611    pub module_id: String,
612    pub program: PathBuf,
613    pub args: Vec<String>,
614    pub env: Vec<(String, String)>,
615    /// When true this is a reserved module: each spawn gets a fresh one-time launch
616    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
617    /// process can register this module_id (a security-boundary module like the
618    /// credential vault must not be impersonable while it is down/restarting).
619    pub reserved: bool,
620    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
621    /// Prefixes come from daemon config and must end in `:` before they reach the
622    /// supervisor; the owner module's current spawn nonce authorizes claims under
623    /// each prefix.
624    pub reserved_prefixes: Vec<String>,
625    /// The wire protocol this module speaks, as DECLARED in daemon config.
626    ///
627    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
628    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
629    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
630    /// and NO launch nonce, and a clean exit the daemon did not request is
631    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
632    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
633    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
634    /// because a process ignores an environment variable it does not read.
635    ///
636    /// The argument is the part that cannot be "harmless to a process that
637    /// ignores it": a stock binary exits on an unknown flag before it listens
638    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
639    /// first conformance run against this mode found it. The nonce is withheld
640    /// because a process that will never present it gains nothing from holding
641    /// it, and a secret in the environment of a process that does not need it is
642    /// a leak surface for no benefit.
643    pub protocol: ModuleProtocol,
644    /// Whether two processes of this module may run at once, which is what a
645    /// blue/green swap does for the length of its overlap. Declared in daemon
646    /// config because the daemon must be able to answer it while the module is
647    /// down, and so a module cannot talk itself into it after registering.
648    pub overlap: ModuleOverlap,
649}
650
651/// Whether a module tolerates a second process of itself running alongside.
652///
653/// Most modules are single-writer on their store (a WAL, a capture log, a
654/// resident index behind a writer barrier), and two processes on one store
655/// corrupt it. So a swap, which overlaps the old and new process by design,
656/// is refused unless the module's config opts in with `overlap: "safe"`.
657#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659    /// Never run two processes of this module at once. The default.
660    #[default]
661    Exclusive,
662    /// The module has said a second process of itself is harmless for the
663    /// length of a swap.
664    ///
665    /// Declare it only if a second instance can run for a few seconds without
666    /// touching ANY single-writer store: every database, WAL, index, projector
667    /// and scheduled job the module owns. A lease on part of that state is not
668    /// enough. broca's session lease guards WAL appends while its run index, its
669    /// store projector and its archive fold timer (which unlinks live WAL files)
670    /// stay single-writer, so broca is exclusive despite holding a lease. The
671    /// refusal only fires after this has been decided, so the decision is the
672    /// check.
673    Safe,
674}
675
676impl ModuleOverlap {
677    pub fn as_str(self) -> &'static str {
678        match self {
679            Self::Exclusive => "exclusive",
680            Self::Safe => "safe",
681        }
682    }
683}
684
685/// Environment variable telling a spawned module which case it was started
686/// for, before it sends HELLO. Only a swap candidate carries it, as
687/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
688///
689/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
690/// longer because nobody waits on it, while a plain restart must flip ready
691/// quickly because callers see `module_warming` until it does. Absence means
692/// plain restart, the safe reading. The daemon trusts nothing about it; the
693/// candidate is proven by its launch nonce at HELLO.
694pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
696pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697/// How long a swap waits for its candidate to register and declare itself
698/// ready when the operator does not say. A module warming as a swap candidate
699/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
700/// daemon allows that plus time to start the process and send HELLO.
701pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703/// Bounded restart policy for crash exits.
704///
705/// `max_restarts` is the number of replacement processes allowed after the
706/// initial spawn WITHIN `window`. After that many crash restarts inside one
707/// window the module enters [`ModuleState::Failed`] and the supervisor stops
708/// the crash loop.
709///
710/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
711/// and that only survived because crashes were rare: a module that crashed
712/// three times across a week was disabled forever by crashes that had nothing
713/// to do with each other. That stopped being survivable once modules began
714/// exiting non-zero whenever the daemon's connection to them drops, because
715/// then every daemon-side connection drop spends a unit of the same budget and
716/// one flappy hour permanently stops a healthy module. Restarts older than
717/// `window` release their slot, so a module that crashed twice yesterday has a
718/// full budget today, while a genuine crash loop -- which is fast by
719/// definition -- still reaches the cap and stops.
720#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722    pub max_restarts: u32,
723    /// Base delay before a crash replacement. The actual delay escalates with
724    /// the number of recent crash replacements and is capped by `max_backoff`.
725    pub backoff: Duration,
726    /// Maximum delay before a crash replacement.
727    pub max_backoff: Duration,
728    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
729    /// budget effectively infinite (nothing is ever in-window), which is why
730    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
731    pub window: Duration,
732}
733
734impl RestartPolicy {
735    /// A policy with the default crash window. Callers that care about the
736    /// window say so with [`Self::with_window`]; the ones that do not are
737    /// asking for the standard rate limit, not for no limit.
738    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739        Self {
740            max_restarts,
741            backoff,
742            max_backoff: DEFAULT_MAX_BACKOFF,
743            window: DEFAULT_RESTART_WINDOW,
744        }
745    }
746
747    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748        self.max_backoff = max_backoff;
749        self
750    }
751
752    pub fn with_window(mut self, window: Duration) -> Self {
753        self.window = window;
754        self
755    }
756
757    /// Calculate the capped exponential delay for the next crash replacement.
758    /// `restart_in_window` is zero for the first replacement after an operator
759    /// action (restart, reload, re-enable) cleared the crash ring, or after all
760    /// older crash replacements have aged out of the window.
761    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762        if self.backoff.is_zero() || self.max_backoff.is_zero() {
763            return Duration::ZERO;
764        }
765
766        let mut delay = self.backoff;
767        for _ in 0..restart_in_window {
768            if delay >= self.max_backoff {
769                return self.max_backoff;
770            }
771            delay = delay
772                .checked_mul(10)
773                .unwrap_or(self.max_backoff)
774                .min(self.max_backoff);
775        }
776        delay.min(self.max_backoff)
777    }
778
779    /// The one sentence that explains a budget-exhausted stop, used for both the
780    /// log line and the terminal record so the two cannot drift. It names the
781    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
782    /// exactly what this budget is not.
783    fn budget_exhausted_detail(&self) -> String {
784        format!(
785            "crash budget exhausted: max_restarts={} within window_secs={}",
786            self.max_restarts,
787            self.window.as_secs()
788        )
789    }
790}
791
792impl Default for RestartPolicy {
793    fn default() -> Self {
794        Self {
795            max_restarts: DEFAULT_MAX_RESTARTS,
796            backoff: DEFAULT_BACKOFF,
797            max_backoff: DEFAULT_MAX_BACKOFF,
798            window: DEFAULT_RESTART_WINDOW,
799        }
800    }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805    restart_in_window: u32,
806    delay: Duration,
807}
808
809/// Whether the daemon itself will bring this module back after the exit being
810/// handled: it is enabled AND its in-window crash restarts are below the cap.
811///
812/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
813/// the window are dropped here rather than by a timer, so the count is right
814/// the moment somebody asks and no bookkeeping runs for idle modules.
815fn daemon_will_restart(
816    state: &mut SupervisorSnapshot,
817    policy: &RestartPolicy,
818    now: Instant,
819) -> bool {
820    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830    Report,
831    Restart,
832    Alert,
833}
834
835impl fmt::Display for HealthAction {
836    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837        f.write_str(match self {
838            Self::Report => "report",
839            Self::Restart => "restart",
840            Self::Alert => "alert",
841        })
842    }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847    /// Optional loopback HTTP endpoint for a managed non-wire process.
848    /// Changing it applies live on rescan; the process protocol changes only
849    /// at its next spawn.
850    pub http: Option<String>,
851    pub cadence: Duration,
852    pub deadline: Duration,
853    pub failure_threshold: u32,
854    pub on_degraded: HealthAction,
855    pub on_failing: HealthAction,
856    pub critical: bool,
857}
858
859impl Default for HealthConfig {
860    fn default() -> Self {
861        Self {
862            http: None,
863            cadence: DEFAULT_HEALTH_CADENCE,
864            deadline: DEFAULT_HEALTH_DEADLINE,
865            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866            on_degraded: HealthAction::Report,
867            on_failing: HealthAction::Report,
868            critical: false,
869        }
870    }
871}
872
873/// The supervisor's view of one module's health, relayed to clients over
874/// channel-0 and rendered by `ck health`.
875///
876/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
877/// stated here rather than only at the wire type a consumer reads. A reader can
878/// look up what `None` means; only a writer can silently change it, and the
879/// writer has no reason to go looking at a downstream contract before editing.
880///
881/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
882/// back to `None` on re-registration precisely so a respawned module does not
883/// carry its predecessor's timestamp — so an old value and an absent one call for
884/// opposite readings, and anything that defaulted this to a number would make a
885/// never-probed module indistinguishable from one probed at the epoch.
886///
887/// `detail` and `metrics` are `None` when the module published none on this
888/// probe, which does not mean it reported nothing wrong — it is also the shape
889/// when the probe never reached it. `last_probe_ms` is what separates those.
890#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892    pub status: SupervisorHealthStatus,
893    pub last_probe_ms: Option<u64>,
894    pub detail: Option<String>,
895    pub metrics: Option<Value>,
896    pub consecutive_failures: u32,
897    /// Number of replies received after a recurring health probe's deadline.
898    /// Unlike a timeout, every increment proves the module was alive.
899    pub late_answer_count: u64,
900    /// End-to-end latency of the newest late reply, measured from probe start.
901    pub last_late_answer_latency_ms: Option<u64>,
902    pub last_action: Option<String>,
903    /// Set together with `last_action`; the pair moves as one, and both being
904    /// absent means no escalation has ever been taken rather than that the last
905    /// one succeeded.
906    pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910    fn default() -> Self {
911        Self {
912            status: SupervisorHealthStatus::Unknown,
913            last_probe_ms: None,
914            detail: None,
915            metrics: None,
916            consecutive_failures: 0,
917            late_answer_count: 0,
918            last_late_answer_latency_ms: None,
919            last_action: None,
920            last_action_ms: None,
921        }
922    }
923}
924
925/// Typed lifecycle state for a supervised module.
926#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928    Starting,
929    Running,
930    Unresponsive,
931    Restarting,
932    Draining,
933    Stopped,
934    Failed,
935    Disabled,
936}
937
938impl fmt::Display for ModuleState {
939    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940        f.write_str(match self {
941            Self::Starting => "starting",
942            Self::Running => "running",
943            Self::Unresponsive => "unresponsive",
944            Self::Restarting => "restarting",
945            Self::Draining => "draining",
946            Self::Stopped => "stopped",
947            Self::Failed => "failed",
948            Self::Disabled => "disabled",
949        })
950    }
951}
952
953/// Supervisor classification of a child-process exit.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956    Clean,
957    Crash,
958    DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962    fn from(kind: ExitKind) -> Self {
963        match kind {
964            ExitKind::Clean => Self::Clean,
965            ExitKind::Crash => Self::Crash,
966            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967        }
968    }
969}
970
971/// Exact process identity retained when a supervised module registers its
972/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
973#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975    pub(crate) pid: u32,
976    pub(crate) start_time: u64,
977}
978
979/// Last observed child exit, if any.
980#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982    pub kind: ExitKind,
983    pub code: Option<i32>,
984    pub signal: Option<i32>,
985    pub at_ms: u64,
986}
987
988/// Point-in-time module status answerable by subc without forwarding to the
989/// module process.
990#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992    pub module_id: String,
993    pub state: ModuleState,
994    pub enabled: bool,
995    pub process_alive: bool,
996    pub registration_active: bool,
997    /// The module's declared wire protocol, carried beside `live` because it is
998    /// what makes `live` readable: the two fields answer one question together.
999    /// While a process is alive this is its launch declaration, not a later
1000    /// pending-reload edit. When down it is the configured next launch protocol.
1001    pub protocol: ModuleProtocol,
1002    /// Whether the module is serving, under the strongest definition the daemon
1003    /// can assert for its protocol.
1004    ///
1005    /// A subc module must also be REGISTERED: its process being alive says
1006    /// nothing about whether it can take a request. A `protocol: "none"` module
1007    /// never registers, so that term is dropped and this falls back to "enabled,
1008    /// running, and the process the daemon launched is alive" -- which is all
1009    /// the daemon observes about a process that speaks no subc wire. It stays a
1010    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
1011    /// rather than printing it bare.
1012    pub live: bool,
1013    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
1014    /// restarts have already released their slot, so this count can go down
1015    /// without anybody touching the module.
1016    pub restart_count: u32,
1017    /// Replacement processes spawned over this module's entire supervisor lifetime;
1018    /// unlike `restart_count`, this value is never reset by an operator action
1019    /// and never falls out of a window.
1020    pub lifetime_restarts: u32,
1021    pub spawn_generation: u64,
1022    /// The budget `restart_count` is spent against. Carried alongside the count
1023    /// because the count alone does not say how close the module is to being
1024    /// disabled, and reporting one without the other is what makes an
1025    /// about-to-be-retired module look ordinary.
1026    pub max_restarts: u32,
1027    /// The span `restart_count` is counted over. Carried with the pair above for
1028    /// the same reason they are carried together: "2 of 3" means one thing for a
1029    /// ten-minute window and something else entirely for a lifetime.
1030    pub restart_window: Duration,
1031    /// Effective drain and restart timing policy used by this running module.
1032    /// These values are carried together with the restart budget so status
1033    /// readers can compare configured intent with what the supervisor applied.
1034    pub drain_timeout: Duration,
1035    pub restart_backoff: Duration,
1036    pub restart_max_backoff: Duration,
1037    /// The module's process. On macOS this stays absent while the `ck-subc`
1038    /// launch trampoline is still running in that pid, and appears once the
1039    /// exec acknowledgement confirms the module image has replaced it. Launch
1040    /// time and the supervisor's own hold on the process are unaffected.
1041    pub pid: Option<u32>,
1042    pub spawned_at_ms: Option<u64>,
1043    pub spawned_from: Option<PathBuf>,
1044    pub process_start_time: Option<u64>,
1045    pub last_exit: Option<ExitReport>,
1046    pub health: ModuleHealthStatus,
1047}
1048
1049#[derive(Debug, Clone, PartialEq)]
1050struct SupervisorSnapshot {
1051    state: ModuleState,
1052    enabled: bool,
1053    process_alive: bool,
1054    spawned_protocol: Option<ModuleProtocol>,
1055    spawn_failure: Option<String>,
1056    /// When each crash restart was spent, oldest first. This IS the crash
1057    /// budget: its in-window length is the count an operator sees and the count
1058    /// the restart decision is made against, so there is no second counter that
1059    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1060    /// operator actions that used to zero the old lifetime counter.
1061    crash_restarts: VecDeque<Instant>,
1062    lifetime_restarts: u32,
1063    /// Successful child spawns in this daemon incarnation.
1064    ///
1065    /// `lifetime_restarts` was considered and rejected: it starts at zero
1066    /// (line 640), successful initial/operator spawns in `set_running` do not
1067    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1068    /// increments before a successful replacement exists (lines 604, 3846,
1069    /// and 3921), so a failed spawn can consume it. This counter moves only
1070    /// when a live PID is accepted below.
1071    spawn_generation: u64,
1072    pid: Option<u32>,
1073    #[cfg(target_os = "macos")]
1074    report_ready: Option<Arc<OnceLock<()>>>,
1075    /// Last reaped child, retained after current process facts are cleared.
1076    reaped_pid: Option<u32>,
1077    /// Whether the command-serving supervision loop has a scheduled respawn.
1078    respawn_pending: bool,
1079    /// A second restart is waiting for the replacement already scheduled.
1080    coalesced_restart_pending: bool,
1081    spawned_at_ms: Option<u64>,
1082    spawned_from: Option<PathBuf>,
1083    spawned_file_identity: Option<SpawnedFileIdentity>,
1084    process_start_time: Option<u64>,
1085    deliberate_severance: Option<ProcessIdentity>,
1086    last_exit: Option<ExitReport>,
1087    /// Diagnostic attached to the next drain's terminal record, if any.
1088    drain_disposition_detail: Option<String>,
1089    health: ModuleHealthStatus,
1090    /// Whether the current process was started as a swap candidate and so
1091    /// lives in the module's alternate cgroup. The next swap's candidate takes
1092    /// the other one, so the two processes of a swap never share a cgroup. A
1093    /// plain spawn always uses the primary cgroup.
1094    in_alternate_slot: bool,
1095    /// Whether the current `Draining` state ends in a replacement process
1096    /// (restart, reload, health restart) rather than a stop. Only meaningful
1097    /// while `state` is `Draining`; every entry into that state rewrites it.
1098    /// It is what lets route.open answer the retryable `module_reloading` to a
1099    /// consumer that reaches a still-registered process mid-restart, instead of
1100    /// the `supervisor_not_live` a stop or disable deserves.
1101    draining_to_replace: bool,
1102    /// Whether a configuration update has been applied since the current
1103    /// process was spawned, so that process runs an older spec than the one
1104    /// the supervisor now holds. A queued restart is only coalesced into a
1105    /// fresher process when this is false: a restart requested to pick up a
1106    /// new configuration must not be satisfied by a process that predates it.
1107    configuration_updated_since_spawn: bool,
1108}
1109
1110impl SupervisorSnapshot {
1111    /// The pid that status, provenance and resource readings may report. While
1112    /// the launch trampoline still runs in the pid, reading its executable or
1113    /// resource use would describe `ck-subc`, not the module, so none is
1114    /// reported until the exec acknowledgement confirms the module image.
1115    fn reported_pid(&self) -> Option<u32> {
1116        #[cfg(target_os = "macos")]
1117        if self
1118            .report_ready
1119            .as_ref()
1120            .is_some_and(|ready| ready.get().is_none())
1121        {
1122            return None;
1123        }
1124        self.pid
1125    }
1126
1127    fn starting() -> Self {
1128        Self::new(ModuleState::Starting, true)
1129    }
1130
1131    fn disabled() -> Self {
1132        Self::new(ModuleState::Disabled, false)
1133    }
1134
1135    fn failed() -> Self {
1136        Self::new(ModuleState::Failed, true)
1137    }
1138
1139    /// Crash restarts still inside `window`, having dropped the ones that are
1140    /// not. Pruning on read is what makes the budget a rate: an instant older
1141    /// than the window stops holding a slot the moment anybody counts.
1142    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1143        while let Some(oldest) = self.crash_restarts.front() {
1144            if now.duration_since(*oldest) > window {
1145                self.crash_restarts.pop_front();
1146            } else {
1147                break;
1148            }
1149        }
1150        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1151    }
1152
1153    /// Spend one unit of the crash budget and record the restart in the ledger.
1154    ///
1155    /// The ring is bounded by the cap because more than `max_restarts` in-window
1156    /// instants can never be reached (the caller refuses the restart first), so
1157    /// anything beyond that is an unbounded queue waiting to happen.
1158    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1159        self.crash_restarts.push_back(now);
1160        while self.crash_restarts.len() > policy.max_restarts as usize {
1161            self.crash_restarts.pop_front();
1162        }
1163        self.lifetime_restarts += 1;
1164    }
1165
1166    /// Reserve one crash-restart slot and calculate the delay before respawning.
1167    /// The count is captured before recording this restart, so the first retry
1168    /// uses the base delay and each later in-window retry escalates once.
1169    fn next_crash_restart(
1170        &mut self,
1171        policy: &RestartPolicy,
1172        now: Instant,
1173    ) -> Option<CrashRestartSchedule> {
1174        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1175        if restart_in_window >= policy.max_restarts {
1176            return None;
1177        }
1178        self.record_crash_restart(policy, now);
1179        Some(CrashRestartSchedule {
1180            restart_in_window,
1181            delay: policy.delay_for_restart(restart_in_window),
1182        })
1183    }
1184
1185    /// Give the module its full budget back, as an operator restart, reload, or
1186    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1187    /// ledger of what actually happened, and an operator action does not unmake
1188    /// the crashes.
1189    fn clear_crash_restarts(&mut self) {
1190        self.crash_restarts.clear();
1191    }
1192
1193    fn new(state: ModuleState, enabled: bool) -> Self {
1194        Self {
1195            state,
1196            enabled,
1197            process_alive: false,
1198            spawned_protocol: None,
1199            spawn_failure: None,
1200            crash_restarts: VecDeque::new(),
1201            lifetime_restarts: 0,
1202            spawn_generation: 0,
1203            pid: None,
1204            #[cfg(target_os = "macos")]
1205            report_ready: None,
1206            reaped_pid: None,
1207            respawn_pending: false,
1208            coalesced_restart_pending: false,
1209            spawned_at_ms: None,
1210            spawned_from: None,
1211            spawned_file_identity: None,
1212            process_start_time: None,
1213            deliberate_severance: None,
1214            last_exit: None,
1215            drain_disposition_detail: None,
1216            health: ModuleHealthStatus::default(),
1217            in_alternate_slot: false,
1218            draining_to_replace: false,
1219            configuration_updated_since_spawn: false,
1220        }
1221    }
1222}
1223
1224type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1225
1226type SpawnSubscriberKey = (ConnectionId, u64);
1227
1228#[derive(Debug)]
1229struct SpawnSubscriber {
1230    version: u8,
1231    frames: mpsc::Sender<Frame>,
1232    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1233    /// from which event. The full frame channel cannot carry that news, so it
1234    /// travels beside it; see `SpawnEventFeed::subscribe`.
1235    lagged: Option<oneshot::Sender<SpawnCursor>>,
1236}
1237
1238#[derive(Debug)]
1239struct SpawnEventState {
1240    daemon_incarnation: String,
1241    seq: u64,
1242    capacity: usize,
1243    live: HashMap<String, LiveSpawn>,
1244    generations: HashMap<String, u64>,
1245    events: VecDeque<SpawnEvent>,
1246    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1247}
1248
1249impl Default for SpawnEventState {
1250    fn default() -> Self {
1251        Self {
1252            daemon_incarnation: "unconfigured".to_string(),
1253            seq: 0,
1254            capacity: SPAWN_EVENT_RING_CAPACITY,
1255            live: HashMap::new(),
1256            generations: HashMap::new(),
1257            events: VecDeque::new(),
1258            subscribers: HashMap::new(),
1259        }
1260    }
1261}
1262
1263#[derive(Debug, Clone, Default)]
1264struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1265
1266#[derive(Debug, Clone, PartialEq, Eq)]
1267pub(crate) enum SpawnSubscribeRefusal {
1268    ForeignIncarnation { current: String },
1269    TooOld { oldest: SpawnCursor },
1270    Frame(String),
1271}
1272
1273impl SpawnEventFeed {
1274    fn configure_incarnation(&self, daemon_incarnation: String) {
1275        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1276        state.daemon_incarnation = daemon_incarnation;
1277        state.seq = 0;
1278        state.live.clear();
1279        state.generations.clear();
1280        state.events.clear();
1281        state.subscribers.clear();
1282    }
1283
1284    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1285        SpawnCursor {
1286            daemon_incarnation: state.daemon_incarnation.clone(),
1287            seq: state.seq,
1288        }
1289    }
1290
1291    fn snapshot(&self) -> SpawnSnapshot {
1292        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1293        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1294        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1295        SpawnSnapshot {
1296            cursor: Self::cursor(&state),
1297            ring_bound: state.capacity as u64,
1298            live,
1299        }
1300    }
1301
1302    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1303        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1304        let generation = state
1305            .generations
1306            .get(module_id)
1307            .copied()
1308            .unwrap_or(0)
1309            .checked_add(1)
1310            .expect("spawn generation exhausted");
1311        state.generations.insert(module_id.to_string(), generation);
1312        let live = LiveSpawn {
1313            module_id: module_id.to_string(),
1314            spawn_generation: generation,
1315            pid,
1316            spawned_at_ms,
1317        };
1318        state.live.insert(module_id.to_string(), live);
1319        Self::emit_locked(
1320            &mut state,
1321            SpawnEventKind::Spawned,
1322            module_id.to_string(),
1323            generation,
1324            pid,
1325            None,
1326            None,
1327        );
1328        generation
1329    }
1330
1331    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1332        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1333        let Some(live) = state.live.remove(module_id) else {
1334            warn!(
1335                module_id,
1336                "terminal record had no live spawn event identity"
1337            );
1338            return;
1339        };
1340        Self::emit_locked(
1341            &mut state,
1342            SpawnEventKind::Exited,
1343            module_id.to_string(),
1344            live.spawn_generation,
1345            live.pid,
1346            exit_code,
1347            exit_signal,
1348        );
1349    }
1350
1351    /// Report the exit of a process that a swap has already replaced.
1352    ///
1353    /// `emit_exited` removes the module's live entry, which after a swap's
1354    /// cutover describes the promoted candidate, not the old process now
1355    /// exiting. This emits the old generation's exit and leaves the live entry
1356    /// alone unless it still names that generation.
1357    fn emit_superseded_exited(
1358        &self,
1359        module_id: &str,
1360        spawn_generation: u64,
1361        pid: u32,
1362        exit_code: Option<i32>,
1363        exit_signal: Option<i32>,
1364    ) {
1365        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1366        if state
1367            .live
1368            .get(module_id)
1369            .is_some_and(|live| live.spawn_generation == spawn_generation)
1370        {
1371            state.live.remove(module_id);
1372        }
1373        Self::emit_locked(
1374            &mut state,
1375            SpawnEventKind::Exited,
1376            module_id.to_string(),
1377            spawn_generation,
1378            pid,
1379            exit_code,
1380            exit_signal,
1381        );
1382    }
1383
1384    #[allow(clippy::too_many_arguments)]
1385    fn emit_locked(
1386        state: &mut SpawnEventState,
1387        kind: SpawnEventKind,
1388        module_id: String,
1389        spawn_generation: u64,
1390        pid: u32,
1391        exit_code: Option<i32>,
1392        exit_signal: Option<i32>,
1393    ) {
1394        state.seq = state
1395            .seq
1396            .checked_add(1)
1397            .expect("spawn event sequence exhausted");
1398        let event = SpawnEvent {
1399            cursor: Self::cursor(state),
1400            kind,
1401            module_id,
1402            spawn_generation,
1403            pid,
1404            exit_code,
1405            exit_signal,
1406        };
1407        state.events.push_back(event.clone());
1408        while state.events.len() > state.capacity {
1409            state.events.pop_front();
1410        }
1411        let body = match serde_json::to_vec(&event) {
1412            Ok(body) => body,
1413            Err(error) => {
1414                error!(%error, "failed to serialize supervisor spawn event");
1415                return;
1416            }
1417        };
1418        state.subscribers.retain(|(connection_id, corr), subscriber| {
1419            let frame = Frame::build_with_version(
1420                subscriber.version,
1421                FrameType::StreamData,
1422                control_flags(),
1423                0,
1424                0,
1425                *corr,
1426                body.clone(),
1427            );
1428            match frame {
1429                Ok(frame) => {
1430                    if subscriber.frames.try_send(frame).is_ok() {
1431                        true
1432                    } else {
1433                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1434                        if let Some(lagged) = subscriber.lagged.take() {
1435                            let _ = lagged.send(event.cursor.clone());
1436                        }
1437                        false
1438                    }
1439                }
1440                Err(error) => {
1441                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1442                    false
1443                }
1444            }
1445        });
1446    }
1447
1448    fn subscribe(
1449        &self,
1450        connection_id: ConnectionId,
1451        corr: u64,
1452        version: u8,
1453        since: Option<SpawnCursor>,
1454        sink: FrameSink,
1455    ) -> Result<(), SpawnSubscribeRefusal> {
1456        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1457        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1458        {
1459            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1460            let replay = if let Some(since) = since {
1461                if since.daemon_incarnation != state.daemon_incarnation {
1462                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1463                        current: state.daemon_incarnation.clone(),
1464                    });
1465                }
1466                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1467                    if since.seq < oldest.seq.saturating_sub(1) {
1468                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1469                    }
1470                }
1471                state
1472                    .events
1473                    .iter()
1474                    .filter(|event| event.cursor.seq > since.seq)
1475                    .cloned()
1476                    .collect::<Vec<_>>()
1477            } else {
1478                Vec::new()
1479            };
1480            for event in replay {
1481                let body = serde_json::to_vec(&event)
1482                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1483                let frame = Frame::build_with_version(
1484                    version,
1485                    FrameType::StreamData,
1486                    control_flags(),
1487                    0,
1488                    0,
1489                    corr,
1490                    body,
1491                )
1492                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1493                frames
1494                    .try_send(frame)
1495                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1496            }
1497            state.subscribers.insert(
1498                (connection_id, corr),
1499                SpawnSubscriber {
1500                    version,
1501                    frames,
1502                    lagged: Some(lagged),
1503                },
1504            );
1505        }
1506        // The lagged terminal is sent here, by the forwarder, rather than by
1507        // the emitter: at the moment of the drop the subscriber's own channel
1508        // is full, and writing to the connection sink directly from the emitter
1509        // would put the Error AHEAD of the events still queued in that channel
1510        // (and the emitter holds the feed lock, so it cannot await the sink).
1511        // Dropping the subscriber drops the only sender, so `recv` drains every
1512        // queued event and then returns `None`; only then is the Error sent, so
1513        // the client sees each event it can keep, then the reason it was cut.
1514        // Cancel and connection removal drop the oneshot unsent, so they end
1515        // the stream with no Error.
1516        tokio::spawn(async move {
1517            while let Some(frame) = receiver.recv().await {
1518                if sink.send(frame).await.is_err() {
1519                    return;
1520                }
1521            }
1522            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1523                return;
1524            };
1525            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1526                Ok(frame) => {
1527                    let _ = sink.send(frame).await;
1528                }
1529                Err(error) => {
1530                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1531                }
1532            }
1533        });
1534        Ok(())
1535    }
1536
1537    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1538        let Some(subscriber) = self
1539            .0
1540            .lock()
1541            .unwrap_or_else(|p| p.into_inner())
1542            .subscribers
1543            .remove(&(connection_id, corr))
1544        else {
1545            return false;
1546        };
1547        if let Ok(frame) = Frame::build_with_version(
1548            subscriber.version,
1549            FrameType::StreamEnd,
1550            control_flags(),
1551            0,
1552            0,
1553            corr,
1554            Vec::new(),
1555        ) {
1556            tokio::spawn(async move {
1557                let _ = subscriber.frames.send(frame).await;
1558            });
1559        }
1560        true
1561    }
1562
1563    fn remove_connection(&self, connection_id: ConnectionId) {
1564        self.0
1565            .lock()
1566            .unwrap_or_else(|p| p.into_inner())
1567            .subscribers
1568            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1569    }
1570
1571    #[cfg(any(test, feature = "test-support"))]
1572    fn set_capacity(&self, capacity: usize) {
1573        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1574    }
1575
1576    #[cfg(any(test, feature = "test-support"))]
1577    fn subscriber_count(&self) -> usize {
1578        self.0
1579            .lock()
1580            .unwrap_or_else(|p| p.into_inner())
1581            .subscribers
1582            .len()
1583    }
1584}
1585
1586/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1587/// The terminal Error a lagged spawn subscriber receives after its queued events.
1588fn spawn_subscriber_lagged_frame(
1589    version: u8,
1590    corr: u64,
1591    first_undelivered: SpawnCursor,
1592) -> Result<Frame, String> {
1593    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1594        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1595        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1596            .to_string(),
1597        detail: Some(serde_json::json!({
1598            "first_undelivered_cursor": first_undelivered
1599        })),
1600    })
1601    .map_err(|error| error.to_string())?;
1602    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1603        .map_err(|error| error.to_string())
1604}
1605
1606pub trait ModuleProcessLiveness: Send + Sync {
1607    fn process_live(&self, module_id: &str) -> Option<bool>;
1608
1609    /// Whether the supervisor is replacing this module's process right now: an
1610    /// operator restart or reload, a health restart, or a crash respawn whose
1611    /// backoff is running. A module in that state is not live, but a consumer
1612    /// refused now should retry shortly rather than treat the target as gone.
1613    /// Stopped, failed, and disabled modules are not replacing.
1614    fn process_replacing(&self, _module_id: &str) -> bool {
1615        false
1616    }
1617}
1618
1619/// Shared process-liveness registry keyed by supervised `module_id`.
1620#[derive(Debug, Clone, Default)]
1621pub struct SupervisorProcessLiveness {
1622    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1623}
1624
1625impl SupervisorProcessLiveness {
1626    pub fn new() -> Self {
1627        Self::default()
1628    }
1629
1630    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1631        let mut snapshots = self
1632            .snapshots
1633            .lock()
1634            .unwrap_or_else(|poisoned| poisoned.into_inner());
1635        snapshots.insert(module_id, snapshot);
1636    }
1637
1638    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1639        let mut snapshots = self
1640            .snapshots
1641            .lock()
1642            .unwrap_or_else(|poisoned| poisoned.into_inner());
1643        let is_current = snapshots
1644            .get(module_id)
1645            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1646            .unwrap_or(false);
1647        if is_current {
1648            snapshots.remove(module_id);
1649        }
1650    }
1651}
1652
1653impl ModuleProcessLiveness for SupervisorProcessLiveness {
1654    fn process_live(&self, module_id: &str) -> Option<bool> {
1655        let snapshot = {
1656            let snapshots = self
1657                .snapshots
1658                .lock()
1659                .unwrap_or_else(|poisoned| poisoned.into_inner());
1660            snapshots.get(module_id).cloned()
1661        }?;
1662        let snapshot = snapshot
1663            .lock()
1664            .unwrap_or_else(|poisoned| poisoned.into_inner());
1665        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1666    }
1667
1668    fn process_replacing(&self, module_id: &str) -> bool {
1669        let Some(snapshot) = self
1670            .snapshots
1671            .lock()
1672            .unwrap_or_else(|poisoned| poisoned.into_inner())
1673            .get(module_id)
1674            .cloned()
1675        else {
1676            return false;
1677        };
1678        let snapshot = snapshot
1679            .lock()
1680            .unwrap_or_else(|poisoned| poisoned.into_inner());
1681        snapshot.enabled
1682            && match snapshot.state {
1683                ModuleState::Restarting => true,
1684                ModuleState::Draining => snapshot.draining_to_replace,
1685                ModuleState::Starting
1686                | ModuleState::Running
1687                | ModuleState::Unresponsive
1688                | ModuleState::Stopped
1689                | ModuleState::Failed
1690                | ModuleState::Disabled => false,
1691            }
1692    }
1693}
1694
1695#[cfg(test)]
1696#[derive(Debug, Default)]
1697struct ReloadExitRecordGate {
1698    reached: tokio::sync::Notify,
1699    resume: tokio::sync::Notify,
1700}
1701
1702#[derive(Debug, Clone, Copy)]
1703enum RespawnKind {
1704    Spawn,
1705    Reload,
1706}
1707
1708#[derive(Debug, Clone, Copy)]
1709struct PendingRespawn {
1710    deadline: Instant,
1711    kind: RespawnKind,
1712}
1713
1714type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1715
1716#[derive(Debug, Clone)]
1717struct SupervisorRuntimeConfig {
1718    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1719    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1720    /// A reload acknowledges completion only after its replacement registers.
1721    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1722    restart_policy: RestartPolicy,
1723    /// This module's RESOLVED drain budget: per-module config when present,
1724    /// else `default_drain_timeout`.
1725    drain_timeout: Duration,
1726    /// Shared with the status handle so the attested value changes atomically
1727    /// when a rescan updates the running drain policy.
1728    effective_drain_timeout: Arc<Mutex<Duration>>,
1729    /// The supervisor-wide fallback, kept so a configuration update that
1730    /// REMOVES the per-module override can re-resolve to it.
1731    default_drain_timeout: Duration,
1732    health: HealthConfig,
1733    connection_file_path: Option<PathBuf>,
1734    capture_logs_dir: Option<PathBuf>,
1735    forwarding: Option<Arc<ForwardingTable>>,
1736    /// The shared handle, so every spawn path (initial, restart, reload) records the
1737    /// reserved-module launch nonce the HELLO verifier checks against.
1738    supervisor_handle: Option<SupervisorHandle>,
1739    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1740    /// status queries.
1741    ///
1742    /// One ring per module, held across every respawn. The lines explaining an exit
1743    /// are written BEFORE that exit, so a ring recreated per process would be empty
1744    /// exactly when it is asked for.
1745    stderr_ring: Arc<Mutex<StderrRing>>,
1746    terminal_ring: Arc<Mutex<TerminalRing>>,
1747    spawn_events: SpawnEventFeed,
1748    child_roster: ChildRoster,
1749    #[cfg(target_os = "linux")]
1750    cgroup_placement: Option<subc_cgroup::Placement>,
1751    #[cfg(test)]
1752    test_seed_stale_facts_before_enable_spawn: bool,
1753    #[cfg(test)]
1754    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1755}
1756
1757#[derive(Debug, Clone, PartialEq, Eq)]
1758struct SupervisedConfiguration {
1759    spec: ModuleSpec,
1760    health: HealthConfig,
1761}
1762
1763/// Shared daemon lookup table for supervised module handles.
1764///
1765/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1766/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1767/// launch nonces recorded at spawn are checked by the same daemon instance.
1768#[derive(Debug, Clone, Default)]
1769pub struct SupervisorHandle {
1770    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1771    /// Module ids the supervisor has taken on. An id is added BEFORE the
1772    /// module's first process is spawned and removed only when the module
1773    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1774    /// the keys of `modules`.
1775    ///
1776    /// `modules` cannot answer "is this module configured?" on its own: a
1777    /// [`SupervisedModule`] only exists once its process has been spawned, and
1778    /// a fast child can connect, register, sync its scopes and ask about them
1779    /// before the supervisor has inserted it. Answering "not configured" in that
1780    /// gap makes scope admission refuse with the terminal "will never sync"
1781    /// instead of the retryable "has not synced yet".
1782    configured_ids: Arc<Mutex<HashSet<String>>>,
1783    spawn_events: SpawnEventFeed,
1784    /// The current expected launch nonce for each reserved module_id. Set when the
1785    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1786    /// non-reserved module never has an entry here and is never nonce-checked.
1787    /// Reserved module ids and the nonce that authorizes their next HELLO.
1788    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1789    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1790    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1791    /// had NO entry and admitted anyone: the reservation protected the nonce
1792    /// holder, not the NAME (found live by CKCRED's canary probe registering
1793    /// against a reserved scratch id).
1794    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1795    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1796    ///
1797    /// This is deliberately in-memory only: subc is state-free across daemon
1798    /// restarts, and the tombstone only explains the hours-after-removal window
1799    /// while this executing daemon is still alive. Do not persist it in a store.
1800    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1801    /// The current launch nonce for every supervised spawn. This is separate from
1802    /// reserved_nonces because consumer route.open attestation applies to all spawned
1803    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1804    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1805    /// Reserved namespace prefixes mapped to the supervised owner module whose
1806    /// current spawn nonce authorizes HELLO claims below the prefix.
1807    ///
1808    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1809    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1810    /// accidental collisions and lower-trust processes from squatting protected
1811    /// namespaces.
1812    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1813    /// Blue/green swaps in progress, by module id. An entry exists from just
1814    /// before the candidate process is spawned until the swap has failed, or
1815    /// has cut over and the old process is gone. While it exists, HELLO for the
1816    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1817    /// consumer attestation accepts both processes' nonces.
1818    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1819    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1820    promotion_observer: PromotionObserverSlot,
1821    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1822    /// this daemon-wide ordering, a rescan could retire or update a module while a
1823    /// concurrent reload still held its old handle and launch specification.
1824    operation_lock: Arc<AsyncMutex<()>>,
1825}
1826
1827/// Told when a swap has promoted its candidate to be the module's active
1828/// registration.
1829///
1830/// An ordinary HELLO runs the control plane's registration side effects (the
1831/// capability cache, the deny census, the requirement recompute) as it
1832/// registers. A swap candidate's HELLO does not, because it is not routable;
1833/// promotion is when those must run instead, and promotion happens in the
1834/// supervisor, which has no other way into the control handler.
1835pub(crate) trait SwapPromotionObserver: Send + Sync {
1836    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1837}
1838
1839/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1840/// control handler) owns this handle, so a strong reference back would be a
1841/// cycle that keeps both alive.
1842#[derive(Clone, Default)]
1843struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1844
1845impl fmt::Debug for PromotionObserverSlot {
1846    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1847        f.write_str("PromotionObserverSlot")
1848    }
1849}
1850
1851/// The nonces of one open swap.
1852#[derive(Debug, Clone)]
1853struct OpenSwap {
1854    /// The launch nonce minted for the candidate process. It is the swap
1855    /// token: the only thing that admits a HELLO into the candidate slot.
1856    candidate_nonce: String,
1857    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1858    /// here because cutover moves the module's recorded spawn nonce to the
1859    /// candidate while the incumbent is still draining and its consumers are
1860    /// still attesting with this one.
1861    incumbent_nonce: Option<String>,
1862    /// Set once a HELLO has been admitted with the swap token, so the token
1863    /// admits one registration and cannot be replayed after cutover empties
1864    /// the candidate slot.
1865    candidate_admitted: bool,
1866}
1867
1868/// What the swap gate says about a HELLO. See
1869/// [`SupervisorHandle::swap_hello_admission`].
1870#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1871pub(crate) enum SwapHelloAdmission {
1872    /// No swap is open for the id (or the HELLO carries the incumbent's own
1873    /// nonce); the ordinary gates decide.
1874    NotSwapping,
1875    /// The HELLO carries the swap token: register it into the candidate slot.
1876    Candidate,
1877    /// A swap is open and the HELLO carries a nonce the supervisor did not
1878    /// mint for this id, no nonce, or a token already used.
1879    Refused,
1880}
1881
1882#[derive(Debug, Clone, PartialEq, Eq)]
1883pub(crate) enum ReservedHelloRejection {
1884    Exact {
1885        module_id: String,
1886    },
1887    Prefix {
1888        prefix: String,
1889        owner_module_id: String,
1890    },
1891}
1892
1893impl SupervisorHandle {
1894    pub fn new() -> Self {
1895        Self::default()
1896    }
1897
1898    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1899        self.spawn_events.snapshot()
1900    }
1901
1902    pub(crate) fn subscribe_spawns(
1903        &self,
1904        connection_id: ConnectionId,
1905        corr: u64,
1906        version: u8,
1907        since: Option<SpawnCursor>,
1908        sink: FrameSink,
1909    ) -> Result<(), SpawnSubscribeRefusal> {
1910        self.spawn_events
1911            .subscribe(connection_id, corr, version, since, sink)
1912    }
1913
1914    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1915        self.spawn_events.cancel(connection_id, corr)
1916    }
1917
1918    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1919        self.spawn_events.remove_connection(connection_id);
1920    }
1921
1922    #[cfg(any(test, feature = "test-support"))]
1923    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1924        assert!(capacity > 0, "spawn event capacity must be non-zero");
1925        self.spawn_events.set_capacity(capacity);
1926    }
1927
1928    #[cfg(any(test, feature = "test-support"))]
1929    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1930        self.spawn_events.subscriber_count()
1931    }
1932
1933    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1934    /// a respawn invalidates stale consumer identities.
1935    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1936        self.spawn_nonces
1937            .lock()
1938            .unwrap_or_else(|poisoned| poisoned.into_inner())
1939            .insert(module_id.to_string(), nonce);
1940    }
1941
1942    /// Record the launch nonce expected from the next HELLO for a reserved module,
1943    /// replacing any prior nonce (a respawn invalidates the previous one).
1944    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1945        self.reserved_nonces
1946            .lock()
1947            .unwrap_or_else(|poisoned| poisoned.into_inner())
1948            .insert(module_id.to_string(), Some(nonce));
1949    }
1950
1951    /// Record namespace prefixes owned by a supervised module.
1952    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1953        let mut owners = self
1954            .reserved_prefix_owners
1955            .lock()
1956            .unwrap_or_else(|poisoned| poisoned.into_inner());
1957        owners.retain(|_, owner| owner != owner_module_id);
1958        for prefix in prefixes {
1959            owners.insert(prefix.clone(), owner_module_id.to_string());
1960        }
1961    }
1962
1963    /// The launch nonce most recently minted for a module's spawn, if any.
1964    #[cfg(test)]
1965    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1966        self.spawn_nonces
1967            .lock()
1968            .unwrap_or_else(|poisoned| poisoned.into_inner())
1969            .get(module_id)
1970            .cloned()
1971    }
1972
1973    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1974        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1975        let spawn_nonce = self
1976            .spawn_nonces
1977            .lock()
1978            .unwrap_or_else(|poisoned| poisoned.into_inner())
1979            .get(&spec.module_id)
1980            .cloned();
1981        let mut reserved_nonces = self
1982            .reserved_nonces
1983            .lock()
1984            .unwrap_or_else(|poisoned| poisoned.into_inner());
1985        if spec.reserved {
1986            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1987            // reserved name whose module has never spawned has no legitimate
1988            // holder, and the entry's absence is what used to leave the name
1989            // open to the first claimant.
1990            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1991        }
1992        drop(reserved_nonces);
1993        // A later unreserved declaration must not silently unreserve an id that
1994        // was retained after its reserved configuration was removed. The explicit
1995        // release ceremony is the only operation that retires that gate.
1996        self.removal_tombstones
1997            .lock()
1998            .unwrap_or_else(|poisoned| poisoned.into_inner())
1999            .remove(&spec.module_id);
2000    }
2001
2002    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
2003    /// authorized only by its expected nonce; otherwise a matching reserved prefix
2004    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
2005    /// with no matching prefix are always authorized.
2006    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2007        self.reserved_hello_rejection(module_id, presented)
2008            .is_none()
2009    }
2010
2011    pub(crate) fn reserved_hello_rejection(
2012        &self,
2013        module_id: &str,
2014        presented: Option<&str>,
2015    ) -> Option<ReservedHelloRejection> {
2016        let nonces = self
2017            .reserved_nonces
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner());
2020        if let Some(expected) = nonces.get(module_id) {
2021            // `None` = reserved with no legitimate holder: refuse every
2022            // presentation, because no process can hold a nonce that was never
2023            // minted. Only a real minted nonce admits, in constant time.
2024            let authorized = match expected {
2025                Some(expected) => {
2026                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2027                }
2028                None => false,
2029            };
2030            if authorized {
2031                return None;
2032            }
2033            return Some(ReservedHelloRejection::Exact {
2034                module_id: module_id.to_string(),
2035            });
2036        }
2037        drop(nonces);
2038
2039        let matched_prefix = self
2040            .reserved_prefix_owners
2041            .lock()
2042            .unwrap_or_else(|poisoned| poisoned.into_inner())
2043            .iter()
2044            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2045            .max_by_key(|(prefix, _)| prefix.len())
2046            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2047        let (prefix, owner_module_id) = matched_prefix?;
2048
2049        let authorized = presented.is_some_and(|presented| {
2050            self.spawn_nonces
2051                .lock()
2052                .unwrap_or_else(|poisoned| poisoned.into_inner())
2053                .get(&owner_module_id)
2054                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2055                // While the owner is being swapped, children started by
2056                // either of its two processes hold that process's nonce.
2057                || self.swap_nonce_matches(&owner_module_id, presented)
2058        });
2059        if authorized {
2060            None
2061        } else {
2062            Some(ReservedHelloRejection::Prefix {
2063                prefix,
2064                owner_module_id,
2065            })
2066        }
2067    }
2068
2069    /// Whether a consumer connection proved it came from a daemon-spawned module.
2070    ///
2071    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2072    /// accepted only for module ids the supervisor has spawned.
2073    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2074        if presented.is_empty() {
2075            return false;
2076        }
2077        let nonces = self
2078            .spawn_nonces
2079            .lock()
2080            .unwrap_or_else(|poisoned| poisoned.into_inner());
2081        let current = nonces
2082            .get(module_id)
2083            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2084        drop(nonces);
2085        // During a swap two processes of the module are alive, and a consumer
2086        // started by either one presents that process's nonce. Accepting only
2087        // the recorded one would fail the incumbent's consumers for the whole
2088        // overlap once cutover moves the record to the candidate.
2089        current || self.swap_nonce_matches(module_id, presented)
2090    }
2091
2092    /// Whether `presented` is either nonce of an open swap for `module_id`.
2093    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2094        let swaps = self
2095            .swaps
2096            .lock()
2097            .unwrap_or_else(|poisoned| poisoned.into_inner());
2098        swaps.get(module_id).is_some_and(|swap| {
2099            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2100                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2101                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2102                })
2103        })
2104    }
2105
2106    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2107    /// Called before the candidate process exists.
2108    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2109        let incumbent_nonce = self
2110            .spawn_nonces
2111            .lock()
2112            .unwrap_or_else(|poisoned| poisoned.into_inner())
2113            .get(module_id)
2114            .cloned();
2115        self.swaps
2116            .lock()
2117            .unwrap_or_else(|poisoned| poisoned.into_inner())
2118            .insert(
2119                module_id.to_string(),
2120                OpenSwap {
2121                    candidate_nonce,
2122                    incumbent_nonce,
2123                    candidate_admitted: false,
2124                },
2125            );
2126    }
2127
2128    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2129    /// the module's recorded one.
2130    pub(crate) fn close_swap(&self, module_id: &str) {
2131        self.swaps
2132            .lock()
2133            .unwrap_or_else(|poisoned| poisoned.into_inner())
2134            .remove(module_id);
2135    }
2136
2137    /// Install the observer told about swap promotions, replacing any earlier
2138    /// one.
2139    pub(crate) fn set_swap_promotion_observer(
2140        &self,
2141        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2142    ) {
2143        *self
2144            .promotion_observer
2145            .0
2146            .lock()
2147            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2148    }
2149
2150    /// Tell the installed observer, if it is still alive, that a swap promoted
2151    /// `registration`.
2152    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2153        let observer = self
2154            .promotion_observer
2155            .0
2156            .lock()
2157            .unwrap_or_else(|poisoned| poisoned.into_inner())
2158            .as_ref()
2159            .and_then(std::sync::Weak::upgrade);
2160        if let Some(observer) = observer {
2161            observer.swap_promoted(registration);
2162        }
2163    }
2164
2165    /// Whether a swap is open for `module_id`.
2166    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2167        self.swaps
2168            .lock()
2169            .unwrap_or_else(|poisoned| poisoned.into_inner())
2170            .contains_key(module_id)
2171    }
2172
2173    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2174    /// respawn would, once cutover has made the candidate the module's process.
2175    /// The swap stays open so the incumbent's nonce keeps attesting until the
2176    /// incumbent has drained and exited.
2177    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2178        let candidate_nonce = self
2179            .swaps
2180            .lock()
2181            .unwrap_or_else(|poisoned| poisoned.into_inner())
2182            .get(module_id)
2183            .map(|swap| swap.candidate_nonce.clone());
2184        let Some(nonce) = candidate_nonce else {
2185            return;
2186        };
2187        self.set_spawn_nonce(module_id, nonce.clone());
2188        if reserved {
2189            self.set_reserved_nonce(module_id, nonce);
2190        }
2191    }
2192
2193    /// The swap gate for a HELLO claiming `module_id`.
2194    ///
2195    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2196    /// presents the candidate nonce, which the reserved gate (holding the
2197    /// incumbent's nonce) would refuse as `reserved_module` before swap
2198    /// admission was ever reached. And it applies to unreserved ids too: for an
2199    /// unreserved id the only thing that ever stopped a second process claiming
2200    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2201    /// refusal a swap lifts for its candidate.
2202    ///
2203    /// The incumbent's own nonce falls through to the ordinary gates, which
2204    /// treat it as they always have (a live incumbent is refused as a
2205    /// duplicate). Anything else while a swap is open is refused, including an
2206    /// absent nonce.
2207    pub(crate) fn swap_hello_admission(
2208        &self,
2209        module_id: &str,
2210        presented: Option<&str>,
2211    ) -> SwapHelloAdmission {
2212        let swaps = self
2213            .swaps
2214            .lock()
2215            .unwrap_or_else(|poisoned| poisoned.into_inner());
2216        let Some(swap) = swaps.get(module_id) else {
2217            return SwapHelloAdmission::NotSwapping;
2218        };
2219        let Some(presented) = presented else {
2220            return SwapHelloAdmission::Refused;
2221        };
2222        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2223            return if swap.candidate_admitted {
2224                SwapHelloAdmission::Refused
2225            } else {
2226                SwapHelloAdmission::Candidate
2227            };
2228        }
2229        if swap
2230            .incumbent_nonce
2231            .as_deref()
2232            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2233        {
2234            return SwapHelloAdmission::NotSwapping;
2235        }
2236        SwapHelloAdmission::Refused
2237    }
2238
2239    /// Record that the swap token has registered a candidate, so it admits no
2240    /// second HELLO.
2241    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2242        if let Some(swap) = self
2243            .swaps
2244            .lock()
2245            .unwrap_or_else(|poisoned| poisoned.into_inner())
2246            .get_mut(module_id)
2247        {
2248            swap.candidate_admitted = true;
2249        }
2250    }
2251
2252    /// Test/support lookup for the current launch nonce of a supervised spawn.
2253    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2254        self.spawn_nonces
2255            .lock()
2256            .unwrap_or_else(|poisoned| poisoned.into_inner())
2257            .get(module_id)
2258            .cloned()
2259    }
2260
2261    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2262    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2263        self.reserved_nonces
2264            .lock()
2265            .unwrap_or_else(|poisoned| poisoned.into_inner())
2266            .get(module_id)
2267            .cloned()
2268            .flatten()
2269    }
2270
2271    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2272        // Normally already marked before the process was spawned; marking here
2273        // too keeps `configured_ids` a superset of the roster for any caller
2274        // that inserts a module directly.
2275        self.mark_configured(module.module_id());
2276        let mut modules = self
2277            .modules
2278            .lock()
2279            .unwrap_or_else(|poisoned| poisoned.into_inner());
2280        modules.insert(module.module_id().to_string(), module)
2281    }
2282
2283    /// Record that the supervisor has taken on `module_id`. Called before the
2284    /// module's first process is spawned, so that by the time that process can
2285    /// register, [`Self::is_configured`] already answers true.
2286    fn mark_configured(&self, module_id: &str) {
2287        self.configured_ids
2288            .lock()
2289            .unwrap_or_else(|poisoned| poisoned.into_inner())
2290            .insert(module_id.to_string());
2291    }
2292
2293    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2294    /// before it was ever put on the roster. A module already on the roster
2295    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2296    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2297        let modules = self
2298            .modules
2299            .lock()
2300            .unwrap_or_else(|poisoned| poisoned.into_inner());
2301        if !modules.contains_key(module_id) {
2302            self.configured_ids
2303                .lock()
2304                .unwrap_or_else(|poisoned| poisoned.into_inner())
2305                .remove(module_id);
2306        }
2307    }
2308
2309    /// Whether `module_id` is a module this daemon supervises: on the roster,
2310    /// or about to be (its process is being spawned right now).
2311    ///
2312    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2313    /// for scopes: a supervised module's process can register and sync before
2314    /// [`Self::get`] can return it, and in that window it is still a module
2315    /// that will sync, not one that never will.
2316    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2317        self.configured_ids
2318            .lock()
2319            .unwrap_or_else(|poisoned| poisoned.into_inner())
2320            .contains(module_id)
2321    }
2322
2323    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2324        let modules = self
2325            .modules
2326            .lock()
2327            .unwrap_or_else(|poisoned| poisoned.into_inner());
2328        modules.get(module_id).cloned()
2329    }
2330
2331    pub(crate) fn record_late_health_answer(
2332        &self,
2333        module_id: &str,
2334        latency_ms: u64,
2335    ) -> Result<bool, SuperviseError> {
2336        let Some(module) = self.get(module_id) else {
2337            return Ok(false);
2338        };
2339        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2340            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2341            state.health.last_late_answer_latency_ms = Some(latency_ms);
2342            // A late answer is an answer: the module served the probe, just past
2343            // the deadline. Leaving the miss streak in place while logging
2344            // "proves the module is alive" is how a CPU-starved module that
2345            // answers every probe a few seconds late still marches to the
2346            // threshold and gets killed — the exact kill class `NoAnswer` is
2347            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2348            // is degradation, and degradation reports; it does not restart.
2349            state.health.consecutive_failures = 0;
2350        })?;
2351        Ok(true)
2352    }
2353
2354    /// Arm the one-shot marker for the module process that this caller
2355    /// deliberately initiated severance against. Generic connection teardown
2356    /// must not call this:
2357    /// a surviving process would otherwise retain an exemption for a later
2358    /// genuine crash.
2359    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2360        let Some(module) = self.get(module_id) else {
2361            return Ok(false);
2362        };
2363        let snapshot = lock_snapshot(&module.inner.snapshot)?;
2364        let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2365            return Ok(false);
2366        };
2367        drop(snapshot);
2368        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2369    }
2370
2371    pub fn list(&self) -> Vec<SupervisedModule> {
2372        let modules = self
2373            .modules
2374            .lock()
2375            .unwrap_or_else(|poisoned| poisoned.into_inner());
2376        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2377        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2378        modules
2379    }
2380
2381    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2382        self.spawn_nonces
2383            .lock()
2384            .unwrap_or_else(|poisoned| poisoned.into_inner())
2385            .remove(module_id);
2386        self.close_swap(module_id);
2387        let mut reserved_nonces = self
2388            .reserved_nonces
2389            .lock()
2390            .unwrap_or_else(|poisoned| poisoned.into_inner());
2391        if reserved_nonces.contains_key(module_id) {
2392            // The old nonce must die with the removed process, but the exact-id
2393            // gate remains until an operator explicitly releases it.
2394            reserved_nonces.insert(module_id.to_string(), None);
2395        }
2396        drop(reserved_nonces);
2397        self.reserved_prefix_owners
2398            .lock()
2399            .unwrap_or_else(|poisoned| poisoned.into_inner())
2400            .retain(|_, owner| owner != module_id);
2401        let removed = self
2402            .modules
2403            .lock()
2404            .unwrap_or_else(|poisoned| poisoned.into_inner())
2405            .remove(module_id);
2406        self.configured_ids
2407            .lock()
2408            .unwrap_or_else(|poisoned| poisoned.into_inner())
2409            .remove(module_id);
2410        removed
2411    }
2412
2413    /// Remember a module removed by a non-preview rescan so route.open can
2414    /// distinguish that intentional removal from an unknown id.
2415    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2416        self.removal_tombstones
2417            .lock()
2418            .unwrap_or_else(|poisoned| poisoned.into_inner())
2419            .insert(module_id.to_string(), unix_ms_now());
2420    }
2421
2422    /// Return how long ago a rescan removed this module in milliseconds.
2423    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2424        self.removal_tombstones
2425            .lock()
2426            .unwrap_or_else(|poisoned| poisoned.into_inner())
2427            .get(module_id)
2428            .copied()
2429            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2430    }
2431
2432    /// Retire a reserved-id gate only after its module has left supervision.
2433    ///
2434    /// A retained gate has no live nonce (`None`), so releasing any other entry
2435    /// would weaken a currently configured or otherwise active reservation.
2436    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2437        if self.get(module_id).is_some() {
2438            return false;
2439        }
2440        let mut reserved_nonces = self
2441            .reserved_nonces
2442            .lock()
2443            .unwrap_or_else(|poisoned| poisoned.into_inner());
2444        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2445            return false;
2446        }
2447        reserved_nonces.remove(module_id);
2448        true
2449    }
2450
2451    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2452        Arc::clone(&self.operation_lock)
2453    }
2454}
2455
2456/// Process supervisor for subc-owned singleton modules.
2457#[derive(Debug, Clone)]
2458pub struct Supervisor {
2459    registry: Arc<Registry>,
2460    restart_policy: RestartPolicy,
2461    drain_timeout: Duration,
2462    connection_file_path: Option<PathBuf>,
2463    capture_logs_dir: Option<PathBuf>,
2464    forwarding: Option<Arc<ForwardingTable>>,
2465    process_liveness: Arc<SupervisorProcessLiveness>,
2466    supervisor_handle: Option<SupervisorHandle>,
2467    health: HealthConfig,
2468    daemon_start_clock: crate::clock::StartClock,
2469    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2470    spawn_events: SpawnEventFeed,
2471    provenance_probe: ExecutableIdentityProbe,
2472    /// Every process spawned through this supervisor (and its clones) and not
2473    /// yet reaped, so daemon shutdown can end them.
2474    child_roster: ChildRoster,
2475    #[cfg(target_os = "linux")]
2476    cgroup_placement: Option<subc_cgroup::Placement>,
2477    #[cfg(test)]
2478    test_after_first_spawn: AfterFirstSpawnHook,
2479}
2480
2481/// Test-only hook run on the path that takes on a new module, right after its
2482/// first `spawn_child` returns (the process exists and could already be
2483/// registering) and before that process is handed to the module's supervise
2484/// loop and put on the roster. Lets a test observe what a fast child would see
2485/// in that window without racing a real one.
2486#[cfg(test)]
2487#[derive(Clone, Default)]
2488struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2489
2490#[cfg(test)]
2491type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2492
2493#[cfg(test)]
2494impl fmt::Debug for AfterFirstSpawnHook {
2495    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2496        f.write_str("AfterFirstSpawnHook")
2497    }
2498}
2499
2500#[cfg(test)]
2501impl AfterFirstSpawnHook {
2502    fn run(&self, module_id: &str) {
2503        if let Some(hook) = &self.0 {
2504            hook(module_id);
2505        }
2506    }
2507}
2508
2509impl Supervisor {
2510    #[cfg(test)]
2511    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2512        let supervisor = Self::new(registry, policy);
2513        #[cfg(target_os = "macos")]
2514        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2515        supervisor
2516    }
2517    /// Verify the trampoline once when configured. Missing private OS support
2518    /// refuses every macOS launch by name but does not stop the daemon's control
2519    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2520    /// entry point; the library must not exec an arbitrary hosting program.
2521    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2522        let path = path.into();
2523        #[cfg(target_os = "macos")]
2524        {
2525            let result = probe_privacy_trampoline(&path).map(|()| path);
2526            if let Err(cause) = &result {
2527                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2528            }
2529            self.child_roster.set_privacy_trampoline(result);
2530        }
2531        #[cfg(not(target_os = "macos"))]
2532        let _ = path;
2533        self
2534    }
2535    /// The first step of an announced daemon shutdown, before the notice and
2536    /// before any connection is closed.
2537    ///
2538    /// Sets the daemon-shutdown flag first: from here on no module is
2539    /// respawned (crash restart, operator restart, or swap), and every child
2540    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2541    /// the module exits on the EOF this shutdown gives it or is signalled by a
2542    /// service manager that kills the whole cgroup. Then writes the journal's
2543    /// shutdown marker, which records the instant and closes this daemon
2544    /// incarnation's stretch of the journal.
2545    #[cfg(unix)]
2546    pub(crate) fn begin_daemon_shutdown(&self) {
2547        self.child_roster.close();
2548        if let Some(journal) = &self.terminal_journal {
2549            journal.stamp_shutdown();
2550        }
2551    }
2552
2553    /// Announce a cut while established connections can still carry replies.
2554    /// These budgets promise notice and a bounded wait, not child completion;
2555    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2556    #[cfg(unix)]
2557    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2558        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2559        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2560        let Some(forwarding) = &self.forwarding else {
2561            return Ok(());
2562        };
2563        let module_ids = forwarding
2564            .begin_daemon_drain()
2565            .map_err(SuperviseError::Forwarding)?;
2566        let deadline_ms =
2567            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2568        let mut notices = tokio::task::JoinSet::new();
2569        let mut drains = Vec::new();
2570        for module_id in module_ids {
2571            let Some(target) = forwarding
2572                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2573                .map_err(SuperviseError::Forwarding)?
2574            else {
2575                continue;
2576            };
2577            let routes = forwarding
2578                .endpoint_routes(target.endpoint)
2579                .map_err(SuperviseError::Forwarding)?;
2580            // Restart allows deployed consumers to reopen after the new daemon
2581            // appears. The wire reason stays `restart`; what tells a daemon cut
2582            // apart from a module restart afterwards is the terminal record
2583            // itself, whose disposition is `daemon_shutdown` for every exit
2584            // observed once `begin_daemon_shutdown` has run.
2585            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2586                reason: RouteCloseReason::Restart,
2587                deadline_ms,
2588            })
2589            .expect("module draining serializes");
2590            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2591            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2592            for route in routes {
2593                let client = route.goodbye_target;
2594                if let Some((_, channels)) = clients
2595                    .iter_mut()
2596                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2597                {
2598                    channels.push(client.channel);
2599                } else {
2600                    let channel = client.channel;
2601                    clients.push((client, vec![channel]));
2602                }
2603            }
2604            for (client, mut channels) in clients {
2605                channels.sort_unstable();
2606                channels.dedup();
2607                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2608                    module_id: module_id.clone(),
2609                    channels,
2610                    reason: RouteCloseReason::Restart,
2611                })
2612                .expect("route closing serializes");
2613                recipients.push((client.sink, client.negotiated_ver, closing));
2614            }
2615            for (sink, version, body) in recipients {
2616                notices.spawn(async move {
2617                    let frame = Frame::build_with_version(
2618                        version,
2619                        FrameType::Push,
2620                        control_flags(),
2621                        0,
2622                        0,
2623                        0,
2624                        body,
2625                    )
2626                    .expect("bounded lifecycle notice frame builds");
2627                    sink.send_flushed(frame).await
2628                });
2629            }
2630            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2631            drains.push((module_id, target.endpoint, gauges));
2632        }
2633        // A quiet forwarding table is not proof that queued notices reached the
2634        // socket. Wait for writer flush acknowledgements before testing quiescence.
2635        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2636        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2637            if !matches!(result, Ok(Ok(()))) {
2638                warn!(?result, "daemon shutdown notice delivery failed");
2639            }
2640        }
2641        notices.abort_all();
2642        let deadline = Instant::now() + DRAIN_BUDGET;
2643        let mut waits = tokio::task::JoinSet::new();
2644        for (module_id, endpoint, gauges) in drains {
2645            let forwarding = Arc::clone(forwarding);
2646            let mut runtime = self.runtime_config();
2647            runtime.health.cadence = Duration::from_millis(100);
2648            waits.spawn(async move {
2649                wait_for_forwarding_quiescence(
2650                    &forwarding,
2651                    &module_id,
2652                    &runtime,
2653                    endpoint,
2654                    deadline,
2655                    &gauges,
2656                    DrainScope::Active,
2657                )
2658                .await
2659            });
2660        }
2661        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2662            if !matches!(result, Ok(Ok(true))) {
2663                warn!(?result, "daemon shutdown drain did not reach quiescence");
2664            }
2665        }
2666        Ok(())
2667    }
2668
2669    /// The last step of an announced daemon shutdown, after the notice and the
2670    /// drain: send every registered module a module GOODBYE, the same planned
2671    /// stop signal `ck module stop` gives, then close every connection so each
2672    /// subc module sees EOF and starts its own teardown, then end every
2673    /// supervised child that has not exited
2674    /// by its own deadline (its drain budget, capped). Modules lead their own
2675    /// process groups, so a
2676    /// service manager's group kill no longer reaches them; without this a
2677    /// child that does not stop on EOF (every `protocol: "none"` child, which
2678    /// has no connection) would outlive the daemon. Every wait is bounded (see
2679    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2680    #[cfg(unix)]
2681    pub(crate) async fn end_children_for_daemon_shutdown(
2682        &self,
2683        already_escalated: bool,
2684        escalate: impl std::future::Future<Output = ()>,
2685    ) {
2686        tokio::pin!(escalate);
2687        let mut escalated = already_escalated;
2688        if let Some(forwarding) = &self.forwarding {
2689            let reason = CloseReason::new(
2690                "daemon_shutdown",
2691                "the daemon is exiting after its shutdown notice and drain",
2692            );
2693            if escalated {
2694                // The operator asked to stop waiting: queue the GOODBYEs but
2695                // do not wait for them to be written.
2696                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2697            } else {
2698                tokio::select! {
2699                    biased;
2700                    _ = escalate.as_mut() => {
2701                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2702                        escalated = true;
2703                    }
2704                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2705                }
2706            }
2707            let closed = forwarding.close_all_connections(&reason);
2708            debug!(closed, "closed established connections for daemon shutdown");
2709        }
2710        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2711        // already completed and must not be polled again; the child shutdown
2712        // wait is told it is escalated and gets a future that never fires.
2713        let escalated_here = escalated && !already_escalated;
2714        let remaining_escalate = async move {
2715            if escalated_here {
2716                std::future::pending::<()>().await;
2717            } else {
2718                escalate.await;
2719            }
2720        };
2721        crate::child_roster::end_children_for_daemon_shutdown(
2722            &self.child_roster,
2723            escalated,
2724            remaining_escalate,
2725        )
2726        .await;
2727    }
2728
2729    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2730        Self {
2731            registry,
2732            restart_policy,
2733            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2734            connection_file_path: None,
2735            capture_logs_dir: None,
2736            forwarding: None,
2737            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2738            supervisor_handle: None,
2739            health: HealthConfig::default(),
2740            daemon_start_clock: crate::clock::StartClock::capture(),
2741            terminal_journal: None,
2742            spawn_events: SpawnEventFeed::default(),
2743            provenance_probe: ExecutableIdentityProbe::default(),
2744            child_roster: ChildRoster::default(),
2745            #[cfg(target_os = "linux")]
2746            cgroup_placement: None,
2747            #[cfg(test)]
2748            test_after_first_spawn: AfterFirstSpawnHook::default(),
2749        }
2750    }
2751
2752    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2753        self.drain_timeout = drain_timeout;
2754        self
2755    }
2756
2757    pub fn with_process_liveness(
2758        mut self,
2759        process_liveness: Arc<SupervisorProcessLiveness>,
2760    ) -> Self {
2761        self.process_liveness = process_liveness;
2762        self
2763    }
2764
2765    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2766        self.connection_file_path = Some(connection_file_path.into());
2767        self
2768    }
2769
2770    /// Enables daemon-owned capture files for supervised stdout and stderr.
2771    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2772        self.capture_logs_dir = Some(logs_dir.into());
2773        self
2774    }
2775
2776    /// Names this daemon lifetime in spawn events, independently of whether a
2777    /// terminal journal is configured.
2778    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2779        // A millisecond start stamp can repeat after clock rollback or a rapid
2780        // restart. Use the connection file's random daemon_id instead: it already
2781        // identifies this daemon lifetime independently of the wall clock.
2782        self.spawn_events.configure_incarnation(daemon_incarnation);
2783        self
2784    }
2785
2786    /// Enables best-effort history shared by every supervised module. Without
2787    /// it, terminal history is kept only in each module's in-memory ring.
2788    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2789        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2790        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2791            path,
2792            daemon_incarnation,
2793        )));
2794        this
2795    }
2796
2797    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2798        self.forwarding = Some(forwarding);
2799        self
2800    }
2801
2802    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2803        self.spawn_events = supervisor_handle.spawn_events.clone();
2804        self.supervisor_handle = Some(supervisor_handle);
2805        self
2806    }
2807
2808    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2809        self.health = health;
2810        self
2811    }
2812
2813    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2814    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2815    /// record is kept.
2816    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2817        self.child_roster.record_to(path.into());
2818        self
2819    }
2820
2821    #[cfg(target_os = "linux")]
2822    pub fn with_cgroup_placement(
2823        mut self,
2824        cgroup_placement: Option<subc_cgroup::Placement>,
2825    ) -> Self {
2826        self.cgroup_placement = cgroup_placement;
2827        self
2828    }
2829
2830    /// Spawn `spec.program` and start monitoring it.
2831    ///
2832    /// The child is expected to parse `--subc <connection-file-path>`, read the
2833    /// TCP+key connection file, authenticate to the already-running listener, and
2834    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2835    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2836        validate_spec(&spec)?;
2837        self.establish_identity(&spec);
2838
2839        let runtime = self.runtime_config();
2840        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2841        let spawned = spawn_child(
2842            &spec,
2843            runtime.connection_file_path.as_deref(),
2844            self.supervisor_handle.as_ref(),
2845            &runtime.stderr_ring,
2846            runtime.capture_logs_dir.as_deref(),
2847            &runtime.child_roster,
2848            #[cfg(target_os = "linux")]
2849            runtime.cgroup_placement.as_ref(),
2850        );
2851        #[cfg(test)]
2852        self.test_after_first_spawn.run(&spec.module_id);
2853        let child = match spawned {
2854            Ok(child) => child,
2855            Err(err) => {
2856                // Unlike the configured paths, a failed `spawn` leaves nothing
2857                // on the roster, so the module must not stay marked configured.
2858                self.abandon_unrostered(&spec.module_id);
2859                return Err(err);
2860            }
2861        };
2862        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2863
2864        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2865    }
2866
2867    /// Make `spec`'s module count as configured, with its identity gates
2868    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2869    /// exists.
2870    ///
2871    /// Every path that takes on a new module calls this before `spawn_child`.
2872    /// The order is the point: the child can connect, register, sync its
2873    /// scopes and ask about them as soon as it is spawned, and the module is
2874    /// only put on the roster after `spawn_child` returns. Were the mark set
2875    /// with the roster entry, a fast child would see its own owner reported
2876    /// as not configured, and a scoped `route.open` in that window would be
2877    /// refused as terminal `scope_not_live` ("will never sync") instead of
2878    /// retryable `scope_not_synced`.
2879    fn establish_identity(&self, spec: &ModuleSpec) {
2880        if let Some(supervisor_handle) = &self.supervisor_handle {
2881            supervisor_handle.apply_identity_configuration(spec);
2882            supervisor_handle.mark_configured(&spec.module_id);
2883        }
2884    }
2885
2886    /// Take back [`Self::establish_identity`]'s configured mark when the
2887    /// module will not be put on the roster after all.
2888    fn abandon_unrostered(&self, module_id: &str) {
2889        if let Some(supervisor_handle) = &self.supervisor_handle {
2890            supervisor_handle.unmark_configured_unless_rostered(module_id);
2891        }
2892    }
2893
2894    /// Record a freshly spawned first process as running. On failure the
2895    /// module never reaches the roster, so its configured mark is taken back.
2896    fn mark_first_process_running(
2897        &self,
2898        spec: &ModuleSpec,
2899        runtime: &SupervisorRuntimeConfig,
2900        snapshot: &SharedSnapshot,
2901        child: &SupervisedChild,
2902    ) -> Result<(), SuperviseError> {
2903        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2904            self.abandon_unrostered(&spec.module_id);
2905            return Err(err);
2906        }
2907        self.process_liveness
2908            .track(spec.module_id.clone(), Arc::clone(snapshot));
2909        Ok(())
2910    }
2911
2912    /// Start supervising a module declared in daemon configuration.
2913    ///
2914    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2915    /// failures in the supervisor handle so operator-facing `supervisor.list`
2916    /// reflects every configured module while daemon startup continues.
2917    pub fn supervise_configured(
2918        &self,
2919        spec: ModuleSpec,
2920        enabled: bool,
2921    ) -> Result<SupervisedModule, SuperviseError> {
2922        validate_spec(&spec)?;
2923        self.establish_identity(&spec);
2924
2925        let runtime = self.runtime_config();
2926        if !enabled {
2927            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2928            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2929        }
2930
2931        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2932        let spawned = spawn_child(
2933            &spec,
2934            runtime.connection_file_path.as_deref(),
2935            self.supervisor_handle.as_ref(),
2936            &runtime.stderr_ring,
2937            runtime.capture_logs_dir.as_deref(),
2938            &runtime.child_roster,
2939            #[cfg(target_os = "linux")]
2940            runtime.cgroup_placement.as_ref(),
2941        );
2942        #[cfg(test)]
2943        self.test_after_first_spawn.run(&spec.module_id);
2944        match spawned {
2945            Ok(child) => {
2946                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2947                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2948            }
2949            Err(err) => {
2950                error!(
2951                    module_id = %spec.module_id,
2952                    program = %spec.program.display(),
2953                    error = %err,
2954                    "configured module failed to spawn; marking failed and continuing"
2955                );
2956                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2957                Ok(self.supervised_module(spec, runtime, snapshot, None))
2958            }
2959        }
2960    }
2961
2962    /// Supervise a configured module with its own health, drain, and crash
2963    /// budget. The restart policy is per-module because the config file is:
2964    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2965    /// module that is expensive to restart should not be forced onto the same
2966    /// budget as one that is cheap.
2967    pub fn supervise_configured_with_health(
2968        &self,
2969        spec: ModuleSpec,
2970        enabled: bool,
2971        health: HealthConfig,
2972        drain_timeout_ms: Option<u64>,
2973        restart_policy: RestartPolicy,
2974    ) -> Result<SupervisedModule, SuperviseError> {
2975        validate_spec(&spec)?;
2976        self.establish_identity(&spec);
2977
2978        let mut runtime = self.runtime_config();
2979        runtime.health = health.clone();
2980        runtime.restart_policy = restart_policy;
2981        if let Some(ms) = drain_timeout_ms {
2982            runtime.drain_timeout = Duration::from_millis(ms);
2983            *runtime
2984                .effective_drain_timeout
2985                .lock()
2986                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2987        }
2988        if !enabled {
2989            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2990            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2991        }
2992
2993        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2994        let spawned = spawn_child(
2995            &spec,
2996            runtime.connection_file_path.as_deref(),
2997            self.supervisor_handle.as_ref(),
2998            &runtime.stderr_ring,
2999            runtime.capture_logs_dir.as_deref(),
3000            &runtime.child_roster,
3001            #[cfg(target_os = "linux")]
3002            runtime.cgroup_placement.as_ref(),
3003        );
3004        #[cfg(test)]
3005        self.test_after_first_spawn.run(&spec.module_id);
3006        match spawned {
3007            Ok(child) => {
3008                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3009                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3010            }
3011            Err(err) => {
3012                if health.critical {
3013                    error!(
3014                        module_id = %spec.module_id,
3015                        program = %spec.program.display(),
3016                        error = %err,
3017                        "critical configured module failed to spawn; marking failed and alerting"
3018                    );
3019                } else {
3020                    error!(
3021                        module_id = %spec.module_id,
3022                        program = %spec.program.display(),
3023                        error = %err,
3024                        "configured module failed to spawn; marking failed and continuing"
3025                    );
3026                }
3027                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3028                Ok(self.supervised_module(spec, runtime, snapshot, None))
3029            }
3030        }
3031    }
3032
3033    fn runtime_config(&self) -> SupervisorRuntimeConfig {
3034        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3035        SupervisorRuntimeConfig {
3036            scheduled_respawn: Arc::default(),
3037            deferred_reload_reply: Arc::default(),
3038            restart_policy: self.restart_policy,
3039            drain_timeout: self.drain_timeout,
3040            // Shared with this module's roster copy: daemon shutdown waits on
3041            // each child for the module's own drain budget, as resolved now.
3042            child_roster: self
3043                .child_roster
3044                .for_module(Arc::clone(&effective_drain_timeout)),
3045            effective_drain_timeout,
3046            default_drain_timeout: self.drain_timeout,
3047            health: self.health.clone(),
3048            connection_file_path: self.connection_file_path.clone(),
3049            capture_logs_dir: self.capture_logs_dir.clone(),
3050            forwarding: self.forwarding.clone(),
3051            supervisor_handle: self.supervisor_handle.clone(),
3052            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3053            terminal_ring: Arc::new(Mutex::new(
3054                TerminalRing::new(
3055                    TerminalRingConfig::default(),
3056                    self.daemon_start_clock.started_at_ms(),
3057                )
3058                .with_start_clock(self.daemon_start_clock)
3059                .with_journal(self.terminal_journal.clone())
3060                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3061            )),
3062            spawn_events: self.spawn_events.clone(),
3063            #[cfg(target_os = "linux")]
3064            cgroup_placement: self.cgroup_placement.clone(),
3065            #[cfg(test)]
3066            test_seed_stale_facts_before_enable_spawn: false,
3067            #[cfg(test)]
3068            test_reload_exit_record_gate: None,
3069        }
3070    }
3071
3072    fn supervised_module(
3073        &self,
3074        spec: ModuleSpec,
3075        runtime: SupervisorRuntimeConfig,
3076        snapshot: SharedSnapshot,
3077        child: Option<SupervisedChild>,
3078    ) -> SupervisedModule {
3079        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3080            spec: spec.clone(),
3081            health: runtime.health.clone(),
3082        }));
3083        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3084        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3085        // The module's OWN policy, which may be its per-module config rather than
3086        // the supervisor-wide one; status must report the budget the supervise
3087        // loop actually enforces.
3088        let restart_policy = runtime.restart_policy;
3089        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3090        let (tx, rx) = mpsc::channel(4);
3091        let monitor = tokio::spawn(supervise_loop(
3092            spec.clone(),
3093            runtime,
3094            Arc::clone(&self.registry),
3095            Arc::clone(&self.process_liveness),
3096            Arc::clone(&snapshot),
3097            child,
3098            rx,
3099        ));
3100
3101        let module_id = spec.module_id.clone();
3102        let module = SupervisedModule {
3103            inner: Arc::new(SupervisedModuleInner {
3104                module_id: module_id.clone(),
3105                registry: Arc::clone(&self.registry),
3106                snapshot,
3107                configuration,
3108                stderr_ring,
3109                terminal_ring,
3110                commands: tx,
3111                monitor: Mutex::new(Some(monitor)),
3112                restart_policy,
3113                effective_drain_timeout,
3114                provenance_probe: self.provenance_probe.clone(),
3115            }),
3116        };
3117        // The identity gates and the configured mark were set by
3118        // `establish_identity` before any process was spawned; only the roster
3119        // entry waits for the module handle, which needs the spawned child.
3120        if let Some(supervisor_handle) = &self.supervisor_handle {
3121            supervisor_handle.insert(module.clone());
3122        }
3123        module
3124    }
3125}
3126
3127impl Default for Supervisor {
3128    fn default() -> Self {
3129        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3130    }
3131}
3132
3133/// Handle to one supervised child process.
3134#[derive(Clone)]
3135pub struct SupervisedModule {
3136    inner: Arc<SupervisedModuleInner>,
3137}
3138
3139struct SupervisedModuleInner {
3140    module_id: String,
3141    registry: Arc<Registry>,
3142    snapshot: SharedSnapshot,
3143    configuration: Arc<Mutex<SupervisedConfiguration>>,
3144    stderr_ring: Arc<Mutex<StderrRing>>,
3145    terminal_ring: Arc<Mutex<TerminalRing>>,
3146    commands: mpsc::Sender<SupervisorCommand>,
3147    monitor: Mutex<Option<JoinHandle<()>>>,
3148    /// Copied from the supervisor's runtime config at spawn so `status()` can
3149    /// report the restart budget without reaching back into the supervisor. The
3150    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3151    restart_policy: RestartPolicy,
3152    effective_drain_timeout: Arc<Mutex<Duration>>,
3153    provenance_probe: ExecutableIdentityProbe,
3154}
3155
3156impl fmt::Debug for SupervisedModule {
3157    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3158        f.debug_struct("SupervisedModule")
3159            .field("module_id", &self.inner.module_id)
3160            .field("status", &self.status())
3161            .finish_non_exhaustive()
3162    }
3163}
3164
3165impl SupervisedModule {
3166    pub fn module_id(&self) -> &str {
3167        &self.inner.module_id
3168    }
3169
3170    /// Test-only: put one probe miss on the streak, the way
3171    /// `handle_health_probe_failure` does, so tests can assert what a later
3172    /// event does to the streak without driving the whole probe loop.
3173    #[cfg(test)]
3174    pub(crate) fn record_health_probe_failure_for_test(
3175        &self,
3176        detail: &str,
3177    ) -> Result<(), SuperviseError> {
3178        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3179            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3180            state.health.detail = Some(detail.to_string());
3181        })
3182    }
3183
3184    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3185        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3186    }
3187
3188    /// The module's retained stderr, newest lines last.
3189    ///
3190    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3191    /// module, `supervisor.list` renders every module, and putting it in the
3192    /// shared snapshot would make each status read carry a payload almost nobody
3193    /// asked for. Callers that want the text ask for it.
3194    pub fn stderr_tail(
3195        &self,
3196        max_lines: Option<usize>,
3197        max_bytes: Option<usize>,
3198    ) -> StderrTailSnapshot {
3199        self.inner
3200            .stderr_ring
3201            .lock()
3202            .unwrap_or_else(|poisoned| poisoned.into_inner())
3203            .snapshot(max_lines, max_bytes)
3204    }
3205
3206    /// The module's bounded terminal history, oldest retained exit first.
3207    ///
3208    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3209    /// daemon whose in-memory history was necessarily reset.
3210    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3211        self.inner
3212            .terminal_ring
3213            .lock()
3214            .unwrap_or_else(|poisoned| poisoned.into_inner())
3215            .snapshot()
3216    }
3217
3218    /// Retained observations from the current ring and all journal generations.
3219    ///
3220    /// Blocking: this reads the journal files. Async callers use
3221    /// [`Self::read_durable_terminal_history`].
3222    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3223        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3224    }
3225
3226    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3227    /// read (up to every retained generation) never occupies a runtime worker.
3228    /// Fails only if the blocking task could not finish (runtime shutdown or a
3229    /// panic in the read).
3230    pub(crate) async fn read_durable_terminal_history(
3231        &self,
3232    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3233        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3234        let module_id = self.inner.module_id.clone();
3235        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3236            .await
3237    }
3238
3239    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3240        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3241            .map(|(status, _)| status)
3242    }
3243
3244    pub(crate) fn record_deliberate_severance(
3245        &self,
3246        identity: ProcessIdentity,
3247    ) -> Result<bool, SuperviseError> {
3248        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3249        if snapshot.pid != Some(identity.pid)
3250            || snapshot.process_start_time != Some(identity.start_time)
3251        {
3252            return Ok(false);
3253        }
3254        snapshot.deliberate_severance = Some(identity);
3255        Ok(true)
3256    }
3257
3258    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3259    ///
3260    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3261    /// does not produce reader-observability logs.
3262    pub(crate) fn status_for_control(
3263        &self,
3264        caller: &'static str,
3265    ) -> Result<ModuleStatus, SuperviseError> {
3266        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3267            .map(|(status, _)| status)
3268    }
3269
3270    fn status_with_snapshot_lock(
3271        &self,
3272        snapshot: &SharedSnapshot,
3273        caller: Option<&'static str>,
3274    ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3275        let mut guard = match caller {
3276            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3277            None => lock_snapshot(snapshot)?,
3278        };
3279        // Read the budget through the pruning path so a reader sees the same
3280        // in-window count the restart decision would use, not a stale total.
3281        let restart_count =
3282            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3283        let snapshot = guard.clone();
3284        drop(guard);
3285        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3286            SuperviseError::StatePoisoned {
3287                module_id: Some(self.inner.module_id.clone()),
3288            }
3289        })?;
3290        let registration_active = self
3291            .inner
3292            .registry
3293            .get_module(&self.inner.module_id)
3294            .map_err(SuperviseError::Registry)?
3295            .is_some();
3296        let protocol = snapshot
3297            .spawned_protocol
3298            .unwrap_or(self.declared_protocol()?);
3299        let running_process =
3300            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3301        // Registration is the difference between the two protocols and the only
3302        // one: a subc module that has not registered cannot serve a request even
3303        // though its process is up, and a `none` module never registers at all,
3304        // so requiring it there would pin `live` to false for the whole life of
3305        // a perfectly healthy process.
3306        let live = match protocol {
3307            ModuleProtocol::Subc => running_process && registration_active,
3308            ModuleProtocol::None => running_process,
3309        };
3310
3311        Ok((
3312            ModuleStatus {
3313                module_id: self.inner.module_id.clone(),
3314                state: snapshot.state,
3315                enabled: snapshot.enabled,
3316                process_alive: snapshot.process_alive,
3317                registration_active,
3318                protocol,
3319                live,
3320                restart_count,
3321                lifetime_restarts: snapshot.lifetime_restarts,
3322                spawn_generation: snapshot.spawn_generation,
3323                max_restarts: self.inner.restart_policy.max_restarts,
3324                restart_window: self.inner.restart_policy.window,
3325                drain_timeout,
3326                restart_backoff: self.inner.restart_policy.backoff,
3327                restart_max_backoff: self.inner.restart_policy.max_backoff,
3328                pid: snapshot.reported_pid(),
3329                spawned_at_ms: snapshot.spawned_at_ms,
3330                spawned_from: snapshot.spawned_from,
3331                process_start_time: snapshot.process_start_time,
3332                last_exit: snapshot.last_exit,
3333                health: snapshot.health,
3334            },
3335            snapshot.spawned_file_identity,
3336        ))
3337    }
3338
3339    #[cfg(test)]
3340    pub(crate) fn hold_snapshot_for_test(
3341        &self,
3342        acquired: std::sync::mpsc::Sender<()>,
3343        hold: Duration,
3344    ) -> std::thread::JoinHandle<()> {
3345        let snapshot = Arc::clone(&self.inner.snapshot);
3346        std::thread::spawn(move || {
3347            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3348            acquired
3349                .send(())
3350                .expect("test receiver waits for snapshot lock");
3351            std::thread::sleep(hold);
3352        })
3353    }
3354
3355    /// The status and the running-image check for `supervisor.provenance`,
3356    /// taken from one status read. The exec acknowledgement can land between
3357    /// two separate reads, and the reply would then pair "no pid yet" with an
3358    /// image observed after the module started, which describes no single
3359    /// moment.
3360    pub(crate) async fn status_and_running_image_agreement(
3361        &self,
3362    ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3363        let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3364        let image = self
3365            .inner
3366            .provenance_probe
3367            .observe(
3368                status.pid,
3369                status.spawned_from.as_deref(),
3370                identity,
3371                status.process_start_time,
3372            )
3373            .await;
3374        Ok((status, image))
3375    }
3376
3377    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3378        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3379            Ok(snapshot) => snapshot.clone(),
3380            Err(_) => {
3381                return subc_control::RunningImageAgreement::Unavailable {
3382                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3383                };
3384            }
3385        };
3386        self.inner
3387            .provenance_probe
3388            .observe(
3389                snapshot.reported_pid(),
3390                snapshot.spawned_from.as_deref(),
3391                snapshot.spawned_file_identity,
3392                snapshot.process_start_time,
3393            )
3394            .await
3395    }
3396
3397    /// Memory and CPU time of the module's current process, read now. Only the
3398    /// process the supervisor spawned is read, not processes it has started.
3399    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3400        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3401            Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3402            Err(_) => {
3403                return subc_control::ChildResourceUsage::Unavailable {
3404                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3405                }
3406            }
3407        };
3408        crate::child_resources::read(pid, start_time)
3409    }
3410
3411    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3412        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3413        Ok(match snapshot.state {
3414            ModuleState::Restarting => true,
3415            ModuleState::Failed | ModuleState::Disabled => false,
3416            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3417        })
3418    }
3419
3420    #[cfg(test)]
3421    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3422        self.is_warming_with_snapshot_lock(None)
3423    }
3424
3425    pub(crate) fn is_warming_for_control(
3426        &self,
3427        caller: &'static str,
3428    ) -> Result<bool, SuperviseError> {
3429        self.is_warming_with_snapshot_lock(Some(caller))
3430    }
3431
3432    fn is_warming_with_snapshot_lock(
3433        &self,
3434        caller: Option<&'static str>,
3435    ) -> Result<bool, SuperviseError> {
3436        let snapshot = match caller {
3437            Some(caller) => {
3438                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3439            }
3440            None => lock_snapshot(&self.inner.snapshot)?,
3441        }
3442        .clone();
3443        Ok(matches!(
3444            snapshot.state,
3445            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3446        ))
3447    }
3448
3449    /// Drain the module and stop monitoring it.
3450    pub async fn drain(&self) -> Result<(), SuperviseError> {
3451        self.stop().await
3452    }
3453
3454    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3455        match self.state()? {
3456            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3457            ModuleState::Starting
3458            | ModuleState::Running
3459            | ModuleState::Unresponsive
3460            | ModuleState::Restarting
3461            | ModuleState::Draining
3462            | ModuleState::Disabled => {}
3463        }
3464
3465        let (reply_tx, reply_rx) = oneshot::channel();
3466        self.inner
3467            .commands
3468            .send(SupervisorCommand::Retire { reply: reply_tx })
3469            .await
3470            .map_err(|_| SuperviseError::CommandClosed {
3471                module_id: self.inner.module_id.clone(),
3472            })?;
3473        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3474            module_id: self.inner.module_id.clone(),
3475        })?
3476    }
3477
3478    pub async fn stop(&self) -> Result<(), SuperviseError> {
3479        match self.state()? {
3480            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3481            ModuleState::Starting
3482            | ModuleState::Running
3483            | ModuleState::Unresponsive
3484            | ModuleState::Restarting
3485            | ModuleState::Draining
3486            | ModuleState::Disabled => {}
3487        }
3488
3489        let (reply_tx, reply_rx) = oneshot::channel();
3490        self.inner
3491            .commands
3492            .send(SupervisorCommand::Drain { reply: reply_tx })
3493            .await
3494            .map_err(|_| SuperviseError::CommandClosed {
3495                module_id: self.inner.module_id.clone(),
3496            })?;
3497        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3498            module_id: self.inner.module_id.clone(),
3499        })?
3500    }
3501
3502    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3503        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3504        let (reply_tx, reply_rx) = oneshot::channel();
3505        self.inner
3506            .commands
3507            .send(SupervisorCommand::Restart {
3508                drain_timeout_ms,
3509                received_at_generation,
3510                queued_at: Instant::now(),
3511                reply: reply_tx,
3512            })
3513            .await
3514            .map_err(|_| SuperviseError::CommandClosed {
3515                module_id: self.inner.module_id.clone(),
3516            })?;
3517        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3518            module_id: self.inner.module_id.clone(),
3519        })?
3520    }
3521
3522    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3523    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3524    /// process then drains in the background of the supervise loop) or has
3525    /// failed, leaving the old process serving.
3526    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3527        let (reply_tx, reply_rx) = oneshot::channel();
3528        self.inner
3529            .commands
3530            .send(SupervisorCommand::Swap {
3531                ready_timeout,
3532                reply: reply_tx,
3533            })
3534            .await
3535            .map_err(|_| SuperviseError::CommandClosed {
3536                module_id: self.inner.module_id.clone(),
3537            })?;
3538        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3539            module_id: self.inner.module_id.clone(),
3540        })?
3541    }
3542
3543    pub async fn reload(&self) -> Result<(), SuperviseError> {
3544        let (reply_tx, reply_rx) = oneshot::channel();
3545        self.inner
3546            .commands
3547            .send(SupervisorCommand::Reload { reply: reply_tx })
3548            .await
3549            .map_err(|_| SuperviseError::CommandClosed {
3550                module_id: self.inner.module_id.clone(),
3551            })?;
3552        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3553            module_id: self.inner.module_id.clone(),
3554        })?
3555    }
3556
3557    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3558        let (reply_tx, reply_rx) = oneshot::channel();
3559        self.inner
3560            .commands
3561            .send(SupervisorCommand::SetEnabled {
3562                enabled,
3563                reply: reply_tx,
3564            })
3565            .await
3566            .map_err(|_| SuperviseError::CommandClosed {
3567                module_id: self.inner.module_id.clone(),
3568            })?;
3569        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3570            module_id: self.inner.module_id.clone(),
3571        })?
3572    }
3573
3574    /// The current process's protocol, or the configured protocol when down.
3575    /// A rescan stores the next launch spec without changing how an existing
3576    /// process registers, serves routes, is probed, or exits.
3577    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3578        let configured = self
3579            .inner
3580            .configuration
3581            .lock()
3582            .map_err(|_| SuperviseError::StatePoisoned {
3583                module_id: Some(self.inner.module_id.clone()),
3584            })?
3585            .spec
3586            .protocol;
3587        let state = lock_snapshot(&self.inner.snapshot)?;
3588        Ok(state.spawned_protocol.unwrap_or(configured))
3589    }
3590
3591    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3592        let configuration =
3593            self.inner
3594                .configuration
3595                .lock()
3596                .map_err(|_| SuperviseError::StatePoisoned {
3597                    module_id: Some(self.inner.module_id.clone()),
3598                })?;
3599        Ok((configuration.spec.clone(), configuration.health.clone()))
3600    }
3601
3602    /// Replace this module's launch spec, keeping its health and drain policy,
3603    /// the way a rescan does for a changed config entry. The running process is
3604    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3605    #[cfg(any(test, feature = "test-support"))]
3606    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3607        let (_, health) = self.configuration()?;
3608        let drain_timeout_ms = u64::try_from(
3609            self.inner
3610                .effective_drain_timeout
3611                .lock()
3612                .unwrap_or_else(|poisoned| poisoned.into_inner())
3613                .as_millis(),
3614        )
3615        .ok();
3616        self.update_configuration(spec, health, drain_timeout_ms)
3617            .await
3618    }
3619
3620    pub(crate) async fn update_configuration(
3621        &self,
3622        spec: ModuleSpec,
3623        health: HealthConfig,
3624        drain_timeout_ms: Option<u64>,
3625    ) -> Result<(), SuperviseError> {
3626        if spec.module_id != self.inner.module_id {
3627            return Err(SuperviseError::InvalidSpec {
3628                reason: "a supervised module's module_id cannot be changed".to_string(),
3629            });
3630        }
3631        validate_spec(&spec)?;
3632        let (reply_tx, reply_rx) = oneshot::channel();
3633        self.inner
3634            .commands
3635            .send(SupervisorCommand::UpdateConfiguration {
3636                spec: spec.clone(),
3637                health: health.clone(),
3638                drain_timeout_ms,
3639                reply: reply_tx,
3640            })
3641            .await
3642            .map_err(|_| SuperviseError::CommandClosed {
3643                module_id: self.inner.module_id.clone(),
3644            })?;
3645        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3646            module_id: self.inner.module_id.clone(),
3647        })?;
3648        let mut configuration =
3649            self.inner
3650                .configuration
3651                .lock()
3652                .map_err(|_| SuperviseError::StatePoisoned {
3653                    module_id: Some(self.inner.module_id.clone()),
3654                })?;
3655        configuration.spec = spec;
3656        configuration.health = health;
3657        Ok(())
3658    }
3659}
3660
3661impl Drop for SupervisedModuleInner {
3662    fn drop(&mut self) {
3663        let Ok(mut monitor) = self.monitor.lock() else {
3664            return;
3665        };
3666        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3667            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3668                state.state = ModuleState::Stopped;
3669                clear_current_process_facts(state);
3670            });
3671            monitor.abort();
3672        }
3673        let _ = monitor.take();
3674    }
3675}
3676
3677#[derive(Debug)]
3678enum SupervisorCommand {
3679    Drain {
3680        reply: oneshot::Sender<Result<(), SuperviseError>>,
3681    },
3682    Retire {
3683        reply: oneshot::Sender<Result<(), SuperviseError>>,
3684    },
3685    Restart {
3686        /// Operator override for this one restart's drain budget, in ms. `None`
3687        /// uses the module's configured/default budget; `Some(0)` cuts
3688        /// immediately (wedge bounce: a stuck request never settles, so
3689        /// waiting only delays recovery).
3690        drain_timeout_ms: Option<u64>,
3691        /// The module's `spawn_generation` when the request was received, before
3692        /// it waited in the command queue. A queued restart whose module has
3693        /// since spawned a newer process is already satisfied (see the handler).
3694        received_at_generation: u64,
3695        /// When the request entered the command queue, so the handler can log
3696        /// how long it waited behind the loop's other work.
3697        queued_at: Instant,
3698        reply: oneshot::Sender<Result<(), SuperviseError>>,
3699    },
3700    Reload {
3701        reply: oneshot::Sender<Result<(), SuperviseError>>,
3702    },
3703    SetEnabled {
3704        enabled: bool,
3705        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3706    },
3707    UpdateConfiguration {
3708        spec: ModuleSpec,
3709        health: HealthConfig,
3710        /// Per-module drain override from the new config; `None` re-resolves to
3711        /// the supervisor-wide default.
3712        drain_timeout_ms: Option<u64>,
3713        reply: oneshot::Sender<()>,
3714    },
3715    Swap {
3716        /// How long the candidate may take to register and declare itself
3717        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3718        ready_timeout: Option<Duration>,
3719        /// Answered at cutover or failure; the incumbent's drain follows.
3720        reply: oneshot::Sender<Result<(), SuperviseError>>,
3721    },
3722}
3723
3724#[derive(Debug)]
3725pub enum SuperviseError {
3726    InvalidSpec {
3727        reason: String,
3728    },
3729    Spawn {
3730        program: PathBuf,
3731        source: io::Error,
3732        cgroup_path: Option<PathBuf>,
3733    },
3734    Cgroup {
3735        module_id: String,
3736        source: io::Error,
3737    },
3738    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3739    /// than spawn a reserved module without its identity binding.
3740    LaunchNonce {
3741        reason: String,
3742    },
3743    Wait {
3744        module_id: String,
3745        source: io::Error,
3746    },
3747    Kill {
3748        module_id: String,
3749        source: io::Error,
3750    },
3751    Forwarding(ForwardingError),
3752    Registry(RegistryError),
3753    ReloadUnavailable {
3754        module_id: String,
3755        reason: String,
3756    },
3757    /// An operator restart/reload was requested for a module that is currently
3758    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3759    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3760    /// by a restart, so these commands are rejected instead of re-enabling it.
3761    Disabled {
3762        module_id: String,
3763    },
3764    ReloadFailed {
3765        module_id: String,
3766        reason: String,
3767    },
3768    RegistrationStillActive {
3769        module_id: String,
3770        waited: Duration,
3771    },
3772    StatePoisoned {
3773        module_id: Option<String>,
3774    },
3775    CommandClosed {
3776        module_id: String,
3777    },
3778    /// A restart or reload arrived while a swap's candidate was warming. The
3779    /// swap owns the module until it cuts over or fails; a stop or disable
3780    /// would have aborted it instead.
3781    SwapInProgress {
3782        module_id: String,
3783    },
3784    /// A swap was refused before anything was spawned.
3785    SwapRefused {
3786        module_id: String,
3787        reason: SwapRefusal,
3788    },
3789    /// A swap spawned a candidate and gave up on it. The candidate has been
3790    /// killed and its slot freed; the incumbent was left serving and was never
3791    /// drained, except in the one `CutoverLost` case described on that arm.
3792    SwapFailed {
3793        module_id: String,
3794        arm: SwapFailureArm,
3795        detail: String,
3796        /// How the candidate exited, when it exited on its own before the
3797        /// supervisor gave up on it.
3798        candidate_exit: Option<ExitReport>,
3799    },
3800}
3801
3802/// Why a swap was refused before a candidate was spawned.
3803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3804pub enum SwapRefusal {
3805    /// The module's config does not declare `overlap: "safe"`.
3806    OverlapExclusive,
3807    /// The module is not registered, so there is no incumbent to keep serving
3808    /// and nothing a swap would improve on; a plain restart is the tool.
3809    NotRegistered,
3810    /// The module does not speak the subc wire, so a candidate could never
3811    /// register or declare itself ready.
3812    ProtocolNone,
3813    /// The supervisor lacks the forwarding table (to cut routes over) or the
3814    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3815    NotConfigured,
3816    /// A swap is already open for this module.
3817    AlreadySwapping,
3818}
3819
3820impl SwapRefusal {
3821    pub fn as_str(self) -> &'static str {
3822        match self {
3823            Self::OverlapExclusive => "overlap_exclusive",
3824            Self::NotRegistered => "not_registered",
3825            Self::ProtocolNone => "protocol_none",
3826            Self::NotConfigured => "not_configured",
3827            Self::AlreadySwapping => "already_swapping",
3828        }
3829    }
3830}
3831
3832/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3833/// serving and undrained; see `CutoverLost`.
3834#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3835pub enum SwapFailureArm {
3836    /// The candidate process could not be started.
3837    SpawnFailed,
3838    /// The candidate did not register within the readiness budget.
3839    NeverRegistered,
3840    /// The candidate registered but did not declare itself ready in time.
3841    NeverReady,
3842    /// The candidate exited before cutover.
3843    CandidateExited,
3844    /// The candidate declared itself ready but failed its health probe.
3845    CandidateUnhealthy,
3846    /// An operator stop, disable or retire arrived while the candidate warmed.
3847    /// The candidate was killed and the operator's command then carried out on
3848    /// the incumbent.
3849    Interrupted,
3850    /// The candidate's connection closed at the moment of cutover. If it
3851    /// closed before forwarding moved, the incumbent is untouched. If it closed
3852    /// between the forwarding and registry halves of cutover, forwarding can no
3853    /// longer route to the incumbent, so the module is restarted plainly.
3854    CutoverLost,
3855}
3856
3857impl SwapFailureArm {
3858    pub fn as_str(self) -> &'static str {
3859        match self {
3860            Self::SpawnFailed => "spawn_failed",
3861            Self::NeverRegistered => "never_registered",
3862            Self::NeverReady => "never_ready",
3863            Self::CandidateExited => "candidate_exited",
3864            Self::CandidateUnhealthy => "candidate_unhealthy",
3865            Self::Interrupted => "interrupted",
3866            Self::CutoverLost => "cutover_lost",
3867        }
3868    }
3869}
3870
3871impl fmt::Display for SuperviseError {
3872    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3873        match self {
3874            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3875            Self::Spawn {
3876                program,
3877                source,
3878                cgroup_path: Some(cgroup_path),
3879            } => write!(
3880                f,
3881                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3882                cgroup_path.display(),
3883                program.display()
3884            ),
3885            Self::Spawn {
3886                program,
3887                source,
3888                cgroup_path: None,
3889            } => write!(
3890                f,
3891                "failed to spawn module '{}': {source}",
3892                program.display()
3893            ),
3894            Self::Cgroup { module_id, source } => {
3895                write!(
3896                    f,
3897                    "failed to prepare cgroup for module '{module_id}': {source}"
3898                )
3899            }
3900            Self::LaunchNonce { reason } => {
3901                write!(
3902                    f,
3903                    "failed to generate reserved-module launch nonce: {reason}"
3904                )
3905            }
3906            Self::Wait { module_id, source } => {
3907                write!(f, "failed to wait for module '{module_id}': {source}")
3908            }
3909            Self::Kill { module_id, source } => {
3910                write!(f, "failed to kill module '{module_id}': {source}")
3911            }
3912            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3913            Self::Registry(err) => write!(f, "registry error: {err}"),
3914            Self::ReloadUnavailable { module_id, reason } => {
3915                write!(f, "reload unavailable for module '{module_id}': {reason}")
3916            }
3917            Self::Disabled { module_id } => {
3918                write!(
3919                    f,
3920                    "module '{module_id}' is disabled; enable it before restart or reload"
3921                )
3922            }
3923            Self::ReloadFailed { module_id, reason } => {
3924                write!(f, "reload failed for module '{module_id}': {reason}")
3925            }
3926            Self::RegistrationStillActive { module_id, waited } => write!(
3927                f,
3928                "module '{module_id}' registration remained active after waiting {waited:?}"
3929            ),
3930            Self::StatePoisoned { module_id } => match module_id {
3931                Some(module_id) => {
3932                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3933                }
3934                None => write!(f, "supervisor state was poisoned"),
3935            },
3936            Self::CommandClosed { module_id } => {
3937                write!(
3938                    f,
3939                    "supervisor command channel for module '{module_id}' is closed"
3940                )
3941            }
3942            Self::SwapInProgress { module_id } => write!(
3943                f,
3944                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3945            ),
3946            Self::SwapRefused { module_id, reason } => match reason {
3947                SwapRefusal::OverlapExclusive => write!(
3948                    f,
3949                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3950                ),
3951                SwapRefusal::NotRegistered => write!(
3952                    f,
3953                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3954                ),
3955                SwapRefusal::ProtocolNone => write!(
3956                    f,
3957                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3958                ),
3959                SwapRefusal::NotConfigured => write!(
3960                    f,
3961                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3962                ),
3963                SwapRefusal::AlreadySwapping => {
3964                    write!(f, "module '{module_id}' is already being swapped")
3965                }
3966            },
3967            Self::SwapFailed {
3968                module_id,
3969                arm,
3970                detail,
3971                ..
3972            } => write!(
3973                f,
3974                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3975                arm.as_str()
3976            ),
3977        }
3978    }
3979}
3980
3981impl Error for SuperviseError {
3982    fn source(&self) -> Option<&(dyn Error + 'static)> {
3983        match self {
3984            Self::Spawn { source, .. }
3985            | Self::Cgroup { source, .. }
3986            | Self::Wait { source, .. }
3987            | Self::Kill { source, .. } => Some(source),
3988            Self::Forwarding(err) => Some(err),
3989            Self::Registry(err) => Some(err),
3990            Self::LaunchNonce { .. }
3991            | Self::InvalidSpec { .. }
3992            | Self::ReloadUnavailable { .. }
3993            | Self::Disabled { .. }
3994            | Self::ReloadFailed { .. }
3995            | Self::RegistrationStillActive { .. }
3996            | Self::StatePoisoned { .. }
3997            | Self::CommandClosed { .. }
3998            | Self::SwapInProgress { .. }
3999            | Self::SwapRefused { .. }
4000            | Self::SwapFailed { .. } => None,
4001        }
4002    }
4003}
4004
4005pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4006    if spec.module_id.trim().is_empty() {
4007        return Err(SuperviseError::InvalidSpec {
4008            reason: "module_id must not be empty".to_string(),
4009        });
4010    }
4011
4012    Ok(())
4013}
4014
4015#[derive(Debug, Default)]
4016struct HealthProbeRuntime {
4017    configured_health: Option<HealthConfig>,
4018    registered_connection: Option<crate::ConnectionId>,
4019    advertised: bool,
4020    next_probe_at: Option<Instant>,
4021    probe_index: u64,
4022}
4023
4024fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4025    lock_snapshot(snapshot)
4026        .ok()
4027        .and_then(|state| state.spawned_protocol)
4028        .unwrap_or(spec.protocol)
4029}
4030
4031impl HealthProbeRuntime {
4032    fn refresh_registration(
4033        &mut self,
4034        spec: &ModuleSpec,
4035        runtime: &SupervisorRuntimeConfig,
4036        registry: &Registry,
4037        snapshot: &SharedSnapshot,
4038    ) {
4039        if self.configured_health.as_ref() != Some(&runtime.health) {
4040            self.configured_health = Some(runtime.health.clone());
4041            self.next_probe_at = None;
4042            self.registered_connection = None;
4043            self.probe_index = 0;
4044        }
4045        // A non-wire process never registers. Only an explicitly configured
4046        // HTTP endpoint can arm its health probe; an absent HELLO is not a
4047        // health failure for that kind of process.
4048        if running_protocol(spec, snapshot) == ModuleProtocol::None {
4049            self.registered_connection = None;
4050            self.advertised = runtime.health.http.is_some();
4051            if !self.advertised {
4052                self.next_probe_at = None;
4053                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4054                    state.health = ModuleHealthStatus::default();
4055                });
4056            } else if self.next_probe_at.is_none() {
4057                self.next_probe_at = Some(
4058                    Instant::now()
4059                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4060                );
4061            }
4062            return;
4063        }
4064
4065        let registration = match registry.get_module(&spec.module_id) {
4066            Ok(registration) => registration,
4067            Err(err) => {
4068                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4069                self.advertised = false;
4070                self.next_probe_at = None;
4071                return;
4072            }
4073        };
4074
4075        let Some(registration) = registration else {
4076            self.registered_connection = None;
4077            self.advertised = false;
4078            self.next_probe_at = None;
4079            return;
4080        };
4081
4082        let advertised = registration
4083            .control_ops
4084            .iter()
4085            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4086        if !advertised {
4087            self.registered_connection = Some(registration.connection_id);
4088            self.advertised = false;
4089            self.next_probe_at = None;
4090            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4091                state.health.status = SupervisorHealthStatus::Unknown;
4092                state.health.consecutive_failures = 0;
4093                state.health.last_probe_ms = None;
4094                state.health.detail = None;
4095                state.health.metrics = None;
4096            });
4097            return;
4098        }
4099
4100        let reregistered = self.registered_connection != Some(registration.connection_id);
4101        self.registered_connection = Some(registration.connection_id);
4102        self.advertised = true;
4103        if reregistered || self.next_probe_at.is_none() {
4104            self.probe_index = 0;
4105            self.next_probe_at = Some(
4106                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4107            );
4108            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4109                state.health.status = SupervisorHealthStatus::Unknown;
4110                state.health.consecutive_failures = 0;
4111                state.health.detail = None;
4112                state.health.metrics = None;
4113            });
4114        }
4115    }
4116
4117    fn wake_after(&self) -> Duration {
4118        if !self.advertised {
4119            return REGISTRY_RELEASE_POLL;
4120        }
4121        self.next_probe_at
4122            .map(|next| next.saturating_duration_since(Instant::now()))
4123            .unwrap_or(REGISTRY_RELEASE_POLL)
4124    }
4125
4126    fn due(&self) -> bool {
4127        self.advertised
4128            && self
4129                .next_probe_at
4130                .is_some_and(|next| Instant::now() >= next)
4131    }
4132
4133    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4134        self.probe_index = self.probe_index.wrapping_add(1);
4135        self.next_probe_at = Some(
4136            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4137        );
4138    }
4139}
4140
4141/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4142///
4143/// This was a struct with a single `message: String`, and every one of the
4144/// fifteen construction sites collapsed into it. Each site knows exactly what it
4145/// saw -- the lane is gone, the module did not answer in time, the module
4146/// answered with the wrong thing -- and `handle_health_probe_failure` then
4147/// treated all of them identically: increment a counter, compare to a threshold,
4148/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4149/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4150///
4151/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4152///
4153/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4154///   answer on it again.
4155/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4156///   AND with a perfectly healthy one that lost a CPU race -- which is what
4157///   happens under machine load, and is how this supervisor killed a healthy
4158///   module three times in one day.
4159/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4160///   Restarting on it is defensible, but it is not the silence case and should
4161///   never be counted as one.
4162/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4163///   anything, so it cannot be evidence about the module at all.
4164///
4165/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4166/// one that fires most often, and while every variant collapsed into one string
4167/// it carried the same weight as the strongest.
4168///
4169/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4170/// DESIGN and a reader stopping at it gets the build backwards: the restart
4171/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4172/// probes still increment the failure streak and drive escalation at the
4173/// threshold (see `is_proof_of_death` below for why that is deliberate and
4174/// what gates the change). Absence of evidence restarts modules today.
4175#[derive(Debug)]
4176enum HealthProbeEvidence {
4177    /// The module's control lane is gone. Proof of death.
4178    LaneDead,
4179    /// No reply within the deadline. Proves nothing about the module's state.
4180    NoAnswer,
4181    /// The module replied, but not with a usable health report. Proves it is alive.
4182    BadAnswer,
4183    /// The daemon could not ask. Says nothing about the module.
4184    Misconfigured,
4185}
4186
4187#[derive(Debug)]
4188struct HealthProbeError {
4189    evidence: HealthProbeEvidence,
4190    message: String,
4191}
4192
4193impl HealthProbeError {
4194    fn lane_dead(message: impl Into<String>) -> Self {
4195        Self::with(HealthProbeEvidence::LaneDead, message)
4196    }
4197
4198    fn no_answer(message: impl Into<String>) -> Self {
4199        Self::with(HealthProbeEvidence::NoAnswer, message)
4200    }
4201
4202    fn bad_answer(message: impl Into<String>) -> Self {
4203        Self::with(HealthProbeEvidence::BadAnswer, message)
4204    }
4205
4206    fn misconfigured(message: impl Into<String>) -> Self {
4207        Self::with(HealthProbeEvidence::Misconfigured, message)
4208    }
4209
4210    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4211        Self {
4212            evidence,
4213            message: message.into(),
4214        }
4215    }
4216
4217    /// Whether this observation is proof the module cannot serve.
4218    ///
4219    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4220    /// variant that fires under CPU starvation, and treating it as proof is the
4221    /// defect this enum exists to make impossible to reintroduce silently.
4222    ///
4223    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4224    /// to restart also needs a bound for the case it excludes -- a genuinely
4225    /// wedged module, alive but never answering -- and that bound must come from
4226    /// the distribution of real late-answer latencies, which nothing measures
4227    /// yet. Landing the classification first makes the later change a one-line
4228    /// decision against evidence that already exists, rather than two unproven
4229    /// changes at once.
4230    #[allow(dead_code)]
4231    fn is_proof_of_death(&self) -> bool {
4232        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4233    }
4234
4235    /// Short stable label for logs and the health snapshot.
4236    ///
4237    /// An operator reading `ck health` currently cannot tell "the module is gone"
4238    /// from "the module did not answer in five seconds", because both render as
4239    /// prose in the same field. These labels are what make the two
4240    /// distinguishable at a glance, and they are what a later restart-policy
4241    /// change will be argued from.
4242    fn label(&self) -> &'static str {
4243        match self.evidence {
4244            HealthProbeEvidence::LaneDead => "lane-dead",
4245            HealthProbeEvidence::NoAnswer => "no-answer",
4246            HealthProbeEvidence::BadAnswer => "bad-answer",
4247            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4248        }
4249    }
4250}
4251
4252impl fmt::Display for HealthProbeError {
4253    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4254        f.write_str(&self.message)
4255    }
4256}
4257
4258async fn run_health_probe_cycle(
4259    spec: &ModuleSpec,
4260    runtime: &SupervisorRuntimeConfig,
4261    registry: &Registry,
4262    process_liveness: &SupervisorProcessLiveness,
4263    snapshot: &SharedSnapshot,
4264    child: &mut Option<SupervisedChild>,
4265) {
4266    let now_ms = unix_ms_now();
4267    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4268        .then_some(runtime.health.http.as_deref())
4269        .flatten();
4270    let result = match http {
4271        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4272        None => probe_module_health(&spec.module_id, runtime, None).await,
4273    };
4274    match result {
4275        Ok(report) => {
4276            handle_health_report(
4277                spec,
4278                runtime,
4279                registry,
4280                process_liveness,
4281                snapshot,
4282                child,
4283                report,
4284                now_ms,
4285            )
4286            .await;
4287        }
4288        Err(err) => {
4289            if http.is_some() {
4290                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4291                    state.health.status = SupervisorHealthStatus::Failing;
4292                });
4293            }
4294            handle_health_probe_failure(
4295                spec,
4296                runtime,
4297                registry,
4298                process_liveness,
4299                snapshot,
4300                child,
4301                err,
4302                now_ms,
4303            )
4304            .await;
4305        }
4306    }
4307}
4308
4309pub(crate) struct HttpProbeTarget<'a> {
4310    address: std::net::SocketAddr,
4311    localhost: bool,
4312    authority: &'a str,
4313    path: String,
4314}
4315
4316/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4317/// or TLS. A URL cannot turn a local health check into an outbound connection.
4318pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4319    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4320        return Err("must not contain whitespace, controls, or a fragment".into());
4321    }
4322    let rest = url
4323        .strip_prefix("http://")
4324        .ok_or("must use plain http://")?;
4325    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4326    let (authority, suffix) = rest.split_at(split);
4327    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4328        ("::1", rest)
4329    } else {
4330        let split = authority.find(':').unwrap_or(authority.len());
4331        authority.split_at(split)
4332    };
4333    let ip: std::net::IpAddr = match host {
4334        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4335        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4336        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4337    };
4338    let port = if port.is_empty() {
4339        80
4340    } else {
4341        port.strip_prefix(':')
4342            .and_then(|p| p.parse::<u16>().ok())
4343            .filter(|p| *p > 0)
4344            .ok_or("must have a valid nonzero TCP port")?
4345    };
4346    let path = if suffix.is_empty() {
4347        "/".into()
4348    } else if suffix.starts_with('?') {
4349        format!("/{suffix}")
4350    } else {
4351        suffix.into()
4352    };
4353    Ok(HttpProbeTarget {
4354        address: std::net::SocketAddr::new(ip, port),
4355        localhost: host == "localhost",
4356        authority,
4357        path,
4358    })
4359}
4360
4361async fn probe_http_health(
4362    url: &str,
4363    deadline: Duration,
4364) -> Result<HealthReport, HealthProbeError> {
4365    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4366    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4367    // Keep partial diagnostics outside the timed future so cancellation does
4368    // not discard a status line or body bytes already received.
4369    let mut response_status = String::new();
4370    let mut body = Vec::new();
4371    let probe = async {
4372        // Resolve localhost ourselves so a hosts-file override cannot turn
4373        // this into an outbound request, while IPv6-only local servers work.
4374        let connection = match tokio::net::TcpStream::connect(target.address).await {
4375            Err(_) if target.localhost => {
4376                tokio::net::TcpStream::connect((
4377                    std::net::Ipv6Addr::LOCALHOST,
4378                    target.address.port(),
4379                ))
4380                .await
4381            }
4382            result => result,
4383        };
4384        let mut stream = connection.map_err(|error| {
4385            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4386        })?;
4387        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4388            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4389        let mut reader = BufReader::new(stream);
4390        let mut budget = 16 * 1024;
4391        let status = http_line(&mut reader, &mut budget).await?;
4392        let mut words = status.split_ascii_whitespace();
4393        let version = words.next();
4394        let code = words
4395            .next()
4396            .filter(|word| word.len() == 3)
4397            .and_then(|word| word.parse::<u16>().ok());
4398        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4399            || !code.is_some_and(|code| (100..600).contains(&code))
4400        {
4401            return Err(HealthProbeError::bad_answer(format!(
4402                "invalid HTTP status: {status}"
4403            )));
4404        }
4405        let code = code.expect("validated status code");
4406        response_status = status.clone();
4407        let mut length = None;
4408        let mut chunked = false;
4409        loop {
4410            let line = http_line(&mut reader, &mut budget).await?;
4411            if line.is_empty() {
4412                break;
4413            }
4414            if let Some((name, value)) = line.split_once(':') {
4415                if name.eq_ignore_ascii_case("content-length") {
4416                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4417                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4418                    })?);
4419                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4420                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4421                }
4422            }
4423        }
4424        if chunked {
4425            while body.len() < 200 {
4426                let line = http_line(&mut reader, &mut budget).await?;
4427                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4428                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4429                if size == 0 {
4430                    break;
4431                }
4432                let count = size.min((200 - body.len()) as u64) as usize;
4433                let start = body.len();
4434                (&mut reader)
4435                    .take(count as u64)
4436                    .read_to_end(&mut body)
4437                    .await
4438                    .map_err(|error| {
4439                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4440                    })?;
4441                if body.len() - start != count {
4442                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4443                }
4444                if size > count as u64 || body.len() == 200 {
4445                    break;
4446                }
4447                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4448                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4449                }
4450            }
4451        } else {
4452            reader
4453                .take(length.unwrap_or(200).min(200))
4454                .read_to_end(&mut body)
4455                .await
4456                .map_err(|error| {
4457                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4458                })?;
4459        }
4460        if (200..300).contains(&code) {
4461            Ok(HealthReport::ok())
4462        } else {
4463            Err(HealthProbeError::bad_answer(
4464                "HTTP health endpoint returned non-2xx",
4465            ))
4466        }
4467    };
4468    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4469        Err(HealthProbeError::no_answer(format!(
4470            "HTTP probe timed out after {deadline:?}"
4471        )))
4472    });
4473    if let Err(error) = &mut result {
4474        if !response_status.is_empty() {
4475            error.message = format!(
4476                "{}; {response_status}: {}",
4477                error.message,
4478                String::from_utf8_lossy(&body)
4479            );
4480        }
4481    }
4482    result
4483}
4484
4485async fn http_line(
4486    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4487    remaining: &mut usize,
4488) -> Result<String, HealthProbeError> {
4489    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4490    let mut line = Vec::new();
4491    (&mut *reader)
4492        .take(*remaining as u64)
4493        .read_until(b'\n', &mut line)
4494        .await
4495        .map_err(|error| {
4496            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4497        })?;
4498    *remaining -= line.len();
4499    if !line.ends_with(b"\r\n") {
4500        return Err(HealthProbeError::bad_answer(
4501            "HTTP headers are incomplete or exceed 16 KiB",
4502        ));
4503    }
4504    line.truncate(line.len() - 2);
4505    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4506}
4507
4508async fn probe_module_health(
4509    module_id: &str,
4510    runtime: &SupervisorRuntimeConfig,
4511    drain_deadline: Option<Instant>,
4512) -> Result<HealthReport, HealthProbeError> {
4513    let Some(forwarding) = runtime.forwarding.as_ref() else {
4514        return Err(HealthProbeError::misconfigured(
4515            "supervisor was not configured with a forwarding table",
4516        ));
4517    };
4518    let probe_started_at = Instant::now();
4519    let mut deadline = probe_started_at + runtime.health.deadline;
4520    if let Some(drain_deadline) = drain_deadline {
4521        deadline = deadline.min(drain_deadline);
4522    }
4523    let pending = if drain_deadline.is_some() {
4524        forwarding.begin_drain_health_probe_rpc_for(
4525            module_id,
4526            MODULE_CONTROL_OP_HEALTH_CHECK,
4527            probe_started_at,
4528            deadline,
4529        )
4530    } else {
4531        forwarding.begin_health_probe_rpc_for(
4532            module_id,
4533            MODULE_CONTROL_OP_HEALTH_CHECK,
4534            probe_started_at,
4535            deadline,
4536        )
4537    }
4538    .map_err(|err| {
4539        // The endpoint is not registered, so there is no live control lane to
4540        // ask. That is the module being absent, not slow.
4541        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4542    })?;
4543    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4544}
4545
4546/// [`probe_module_health`] for one endpoint rather than the id's active one.
4547///
4548/// A swap probes two processes that no by-id lookup reaches: its candidate
4549/// before cutover, and its superseded incumbent (for busy gauges) while the
4550/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4551/// bounds the by-id drain probe.
4552async fn probe_endpoint_health(
4553    endpoint: crate::ModuleEndpointId,
4554    runtime: &SupervisorRuntimeConfig,
4555    deadline_cap: Option<Instant>,
4556) -> Result<HealthReport, HealthProbeError> {
4557    let Some(forwarding) = runtime.forwarding.as_ref() else {
4558        return Err(HealthProbeError::misconfigured(
4559            "supervisor was not configured with a forwarding table",
4560        ));
4561    };
4562    let probe_started_at = Instant::now();
4563    let mut deadline = probe_started_at + runtime.health.deadline;
4564    if let Some(cap) = deadline_cap {
4565        deadline = deadline.min(cap);
4566    }
4567    let pending = forwarding
4568        .begin_endpoint_health_probe_rpc_for(
4569            endpoint,
4570            MODULE_CONTROL_OP_HEALTH_CHECK,
4571            probe_started_at,
4572            deadline,
4573        )
4574        .map_err(|err| {
4575            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4576        })?;
4577    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4578}
4579
4580/// Send a begun health probe and classify its answer.
4581async fn await_health_probe(
4582    forwarding: &ForwardingTable,
4583    pending: PendingModuleControlRpc,
4584    deadline: Instant,
4585    probe_budget: Duration,
4586) -> Result<HealthReport, HealthProbeError> {
4587    let PendingModuleControlRpc {
4588        endpoint,
4589        module_sink,
4590        negotiated_ver,
4591        corr,
4592        receiver,
4593    } = pending;
4594    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4595        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4596    })?;
4597    let frame = Frame::build_with_version(
4598        negotiated_ver,
4599        FrameType::Request,
4600        control_flags(),
4601        0,
4602        0,
4603        corr,
4604        body,
4605    )
4606    .map_err(|err| {
4607        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4608    })?;
4609
4610    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4611    // blocks waiting for capacity when the module's egress queue is full, and an
4612    // unbounded await here freezes the whole supervision actor (it stops polling
4613    // Child::wait and supervisor commands), making the module unrecoverable
4614    // in-band. On timeout the probe fails like any transport failure.
4615    match timeout_at(deadline, module_sink.send(frame)).await {
4616        Ok(Ok(())) => {}
4617        Ok(Err(err)) => {
4618            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4619            // A closed sink means the module's egress channel is gone -- the
4620            // receiving half is dropped when its connection tears down. Proof.
4621            return Err(HealthProbeError::lane_dead(format!(
4622                "failed to send health.check: {err}"
4623            )));
4624        }
4625        Err(_elapsed) => {
4626            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4627            // A full egress queue means the module is not draining its socket, which
4628            // is consistent with a wedged module AND with one whose reader is merely
4629            // starved. Silence, not proof.
4630            return Err(HealthProbeError::no_answer(
4631                "health.check send timed out before enqueue (module egress full)",
4632            ));
4633        }
4634    }
4635
4636    match timeout_at(deadline, receiver).await {
4637        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4638        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4639        // and those prove it is alive even though the probe failed.
4640        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4641            response.health_report().ok_or_else(|| {
4642                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4643            })
4644        }
4645        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4646            format!("health.check rejected: {}", body.message),
4647        )),
4648        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4649            Err(HealthProbeError::lane_dead(message))
4650        }
4651        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4652            Err(HealthProbeError::bad_answer(message))
4653        }
4654        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4655            Err(HealthProbeError::bad_answer(format!(
4656                "expected module-control op '{expected}', got '{actual}'"
4657            )))
4658        }
4659        // A reply that crosses the deadline before this waiter observes it is
4660        // still proof of life. The forwarding path records its end-to-end latency
4661        // before delivering this classification.
4662        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4663            "module answered health.check after its daemon deadline",
4664        )),
4665        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4666            "health.check waiter was canceled before the module responded",
4667        )),
4668        Err(_) => {
4669            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4670            Err(HealthProbeError::no_answer(format!(
4671                "module did not answer health.check within {probe_budget:?}"
4672            )))
4673        }
4674    }
4675}
4676
4677#[allow(clippy::too_many_arguments)]
4678async fn handle_health_report(
4679    spec: &ModuleSpec,
4680    runtime: &SupervisorRuntimeConfig,
4681    registry: &Registry,
4682    process_liveness: &SupervisorProcessLiveness,
4683    snapshot: &SharedSnapshot,
4684    child: &mut Option<SupervisedChild>,
4685    report: HealthReport,
4686    now_ms: u64,
4687) {
4688    let status = supervisor_health_status(report.status);
4689    let detail = report.detail.clone();
4690    let metrics = truncate_health_metrics(report.metrics);
4691    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4692        state.health.status = status;
4693        state.health.last_probe_ms = Some(now_ms);
4694        state.health.detail = detail.clone();
4695        state.health.metrics = metrics.clone();
4696        state.health.consecutive_failures = 0;
4697    });
4698
4699    let action = match report.status {
4700        HealthStatus::Ok => return,
4701        HealthStatus::Degraded => runtime.health.on_degraded,
4702        HealthStatus::Failing => runtime.health.on_failing,
4703    };
4704    apply_l3_health_action(
4705        spec,
4706        runtime,
4707        registry,
4708        process_liveness,
4709        snapshot,
4710        child,
4711        status,
4712        detail.as_deref(),
4713        action,
4714        now_ms,
4715    )
4716    .await;
4717}
4718
4719#[allow(clippy::too_many_arguments)]
4720async fn handle_health_probe_failure(
4721    spec: &ModuleSpec,
4722    runtime: &SupervisorRuntimeConfig,
4723    registry: &Registry,
4724    process_liveness: &SupervisorProcessLiveness,
4725    snapshot: &SharedSnapshot,
4726    child: &mut Option<SupervisedChild>,
4727    err: HealthProbeError,
4728    now_ms: u64,
4729) {
4730    let threshold = runtime.health.failure_threshold.max(1);
4731    let mut failures = 0;
4732    // Carry the evidence class into the operator-visible detail. Without it,
4733    // "module did not answer within 5s" and "the control lane is gone" are two
4734    // prose strings in the same field, and the reader has to know the codebase to
4735    // tell which one is proof of anything.
4736    let detail = format!("[{}] {err}", err.label());
4737    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4738        // A failed wire probe invalidates the last report, even before the
4739        // restart threshold. HTTP probes already mark failures as Failing.
4740        if state.spawned_protocol.unwrap_or(spec.protocol) == ModuleProtocol::Subc {
4741            state.health.status = SupervisorHealthStatus::Unknown;
4742        }
4743        state.health.last_probe_ms = Some(now_ms);
4744        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4745        state.health.detail = Some(detail.clone());
4746        state.health.metrics = None;
4747        failures = state.health.consecutive_failures;
4748    });
4749
4750    if failures < threshold {
4751        warn!(
4752            module_id = %spec.module_id,
4753            consecutive_failures = failures,
4754            threshold,
4755            evidence = err.label(),
4756            detail = %detail,
4757            "health.check probe failed"
4758        );
4759        return;
4760    }
4761
4762    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4763        state.state = ModuleState::Unresponsive;
4764        state.health.status = SupervisorHealthStatus::Unresponsive;
4765    });
4766    // The evidence class is logged at the kill site because this is the line an
4767    // operator reads after an unexplained restart. A streak of `no-answer` under
4768    // machine load is the known false-positive shape; a `lane-dead` is not.
4769    if runtime.health.critical {
4770        error!(
4771            module_id = %spec.module_id,
4772            status = "unresponsive",
4773            evidence = err.label(),
4774            detail = %detail,
4775            "critical module health alert"
4776        );
4777    } else {
4778        warn!(
4779            module_id = %spec.module_id,
4780            status = "unresponsive",
4781            evidence = err.label(),
4782            detail = %detail,
4783            "module health threshold breached"
4784        );
4785    }
4786    if let Err(err) = health_restart_child(
4787        spec,
4788        runtime,
4789        registry,
4790        process_liveness,
4791        snapshot,
4792        child,
4793        SupervisorHealthStatus::Unresponsive,
4794        Some(&detail),
4795        now_ms,
4796    )
4797    .await
4798    {
4799        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4800    }
4801}
4802
4803#[allow(clippy::too_many_arguments)]
4804async fn apply_l3_health_action(
4805    spec: &ModuleSpec,
4806    runtime: &SupervisorRuntimeConfig,
4807    registry: &Registry,
4808    process_liveness: &SupervisorProcessLiveness,
4809    snapshot: &SharedSnapshot,
4810    child: &mut Option<SupervisedChild>,
4811    status: SupervisorHealthStatus,
4812    detail: Option<&str>,
4813    action: HealthAction,
4814    now_ms: u64,
4815) {
4816    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4817    match action {
4818        HealthAction::Report => {
4819            info!(
4820                module_id = %spec.module_id,
4821                status = ?status,
4822                detail,
4823                "module reported non-ok health"
4824            );
4825        }
4826        HealthAction::Alert => {
4827            error!(
4828                module_id = %spec.module_id,
4829                status = ?status,
4830                detail,
4831                "module health alert"
4832            );
4833        }
4834        HealthAction::Restart => {
4835            if let Err(err) = health_restart_child(
4836                spec,
4837                runtime,
4838                registry,
4839                process_liveness,
4840                snapshot,
4841                child,
4842                status,
4843                detail,
4844                now_ms,
4845            )
4846            .await
4847            {
4848                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4849            }
4850        }
4851    }
4852}
4853
4854#[allow(clippy::too_many_arguments)]
4855async fn health_restart_child(
4856    spec: &ModuleSpec,
4857    runtime: &SupervisorRuntimeConfig,
4858    registry: &Registry,
4859    process_liveness: &SupervisorProcessLiveness,
4860    snapshot: &SharedSnapshot,
4861    child: &mut Option<SupervisedChild>,
4862    status: SupervisorHealthStatus,
4863    detail: Option<&str>,
4864    now_ms: u64,
4865) -> Result<(), SuperviseError> {
4866    let (enabled, schedule) = {
4867        let mut state = lock_snapshot(snapshot)?;
4868        let enabled = state.enabled;
4869        let schedule = if enabled {
4870            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4871        } else {
4872            None
4873        };
4874        (enabled, schedule)
4875    };
4876
4877    if !enabled {
4878        return Err(SuperviseError::Disabled {
4879            module_id: spec.module_id.clone(),
4880        });
4881    }
4882
4883    if schedule.is_none() {
4884        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4885        error!(
4886            module_id = %spec.module_id,
4887            status = ?status,
4888            detail,
4889            max_restarts = runtime.restart_policy.max_restarts,
4890            window_secs = runtime.restart_policy.window.as_secs(),
4891            reason = %runtime.restart_policy.budget_exhausted_detail(),
4892            "health restart budget exhausted; marking module failed"
4893        );
4894        let stop_notice = begin_forwarding_drain_if_configured(
4895            spec,
4896            runtime,
4897            registry,
4898            snapshot,
4899            Some(true),
4900            RouteCloseReason::Disable,
4901        )
4902        .await?;
4903        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4904            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4905        })?;
4906        drain_optional_child(
4907            &spec.module_id,
4908            spec.protocol,
4909            stop_notice,
4910            registry,
4911            runtime.forwarding.as_deref(),
4912            snapshot,
4913            &runtime.terminal_ring,
4914            &runtime.spawn_events,
4915            child,
4916            runtime.drain_timeout,
4917            ModuleState::Failed,
4918            Some(true),
4919        )
4920        .await?;
4921        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4922        return Ok(());
4923    }
4924
4925    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4926    let mut restart_count = 0;
4927    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4928        restart_count = state.crash_restarts.len();
4929        state.state = ModuleState::Unresponsive;
4930        state.health.status = status;
4931        state.health.last_action = Some(HealthAction::Restart.to_string());
4932        state.health.last_action_ms = Some(now_ms);
4933    })?;
4934    warn!(
4935        module_id = %spec.module_id,
4936        status = ?status,
4937        detail,
4938        restart_count,
4939        restart_in_window = schedule.restart_in_window,
4940        delay_ms = schedule.delay.as_millis() as u64,
4941        "health-triggered module restart"
4942    );
4943
4944    let stop_notice = begin_forwarding_drain_if_configured(
4945        spec,
4946        runtime,
4947        registry,
4948        snapshot,
4949        Some(true),
4950        RouteCloseReason::Restart,
4951    )
4952    .await?;
4953    drain_optional_child(
4954        &spec.module_id,
4955        spec.protocol,
4956        stop_notice,
4957        registry,
4958        runtime.forwarding.as_deref(),
4959        snapshot,
4960        &runtime.terminal_ring,
4961        &runtime.spawn_events,
4962        child,
4963        runtime.drain_timeout,
4964        ModuleState::Restarting,
4965        Some(true),
4966    )
4967    .await?;
4968    schedule_respawn(
4969        runtime,
4970        snapshot,
4971        &spec.module_id,
4972        schedule.delay,
4973        RespawnKind::Spawn,
4974    )
4975}
4976
4977fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4978    if let Some(reply) = runtime
4979        .deferred_reload_reply
4980        .lock()
4981        .unwrap_or_else(|p| p.into_inner())
4982        .take()
4983    {
4984        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4985            module_id: module_id.to_string(),
4986            reason: reason.to_string(),
4987        }));
4988    }
4989}
4990
4991fn schedule_respawn(
4992    runtime: &SupervisorRuntimeConfig,
4993    snapshot: &SharedSnapshot,
4994    module_id: &str,
4995    delay: Duration,
4996    kind: RespawnKind,
4997) -> Result<(), SuperviseError> {
4998    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4999    update_snapshot(snapshot, Some(module_id), |state| {
5000        state.respawn_pending = true
5001    })?;
5002    *runtime
5003        .scheduled_respawn
5004        .lock()
5005        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5006        deadline: Instant::now() + delay,
5007        kind,
5008    });
5009    Ok(())
5010}
5011
5012fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5013    let _ = update_snapshot(snapshot, Some(module_id), |state| {
5014        state.health.last_action = Some(action);
5015        state.health.last_action_ms = Some(now_ms);
5016    });
5017}
5018
5019fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5020    match status {
5021        HealthStatus::Ok => SupervisorHealthStatus::Ok,
5022        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5023        HealthStatus::Failing => SupervisorHealthStatus::Failing,
5024    }
5025}
5026
5027/// Caps the metrics blob stored in the cached supervisor snapshot, which is
5028/// returned to every `supervisor.list` and `supervisor.health` caller.
5029///
5030/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
5031/// path: that request exists to return a module's complete metrics object, and
5032/// `ck health <module-id>` documents it as the way to see what the cached view
5033/// truncates. The asymmetry is the feature.
5034///
5035/// So a new caller must decide which side it is on rather than assume the cap is
5036/// universal. Reaching for it on a fresh-probe path would silently reintroduce
5037/// the truncation that path exists to avoid.
5038fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5039    let metrics = metrics?;
5040    match serde_json::to_vec(&metrics) {
5041        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5042            "truncated": true,
5043            "original_bytes": encoded.len(),
5044        })),
5045        Ok(_) | Err(_) => Some(metrics),
5046    }
5047}
5048
5049/// Spread health probes so a fleet-wide restart does not converge them.
5050///
5051/// The delay is derived from the module id and probe index rather than a random
5052/// source, so it is deterministic per module: a module keeps its own offset
5053/// across daemon restarts instead of re-rolling into a collision.
5054fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5055    if cadence.is_zero() {
5056        return Duration::ZERO;
5057    }
5058    let cadence_ms = cadence.as_millis() as u64;
5059    // This early return is REDUNDANT, deliberately, and a mutation run will show
5060    // it surviving removal. Recording why here so the next person to notice does
5061    // not have to re-derive it:
5062    //
5063    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
5064    //   a zero cadence and builds the Duration from whole milliseconds, so a
5065    //   sub-millisecond cadence cannot come from config.
5066    // - Even if reached it changes no answer. The `.max(1)` below makes the span
5067    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
5068    //   -- exactly what this returns.
5069    //
5070    // Kept as a guard against a future widening of the config parser (accepting
5071    // microseconds, say), which would make the sub-millisecond case reachable.
5072    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
5073    // divides by zero. Remove this and nothing changes.
5074    if cadence_ms == 0 {
5075        return cadence;
5076    }
5077    // Note that this never returns less than one cadence, including for the FIRST
5078    // probe. So a freshly registered module reports health `unknown` for a full
5079    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5080    // ready to answer.
5081    //
5082    // That is a property of the supervisor's schedule, not of any module: an
5083    // operator watching a restart sees `unknown` and cannot tell it from a module
5084    // that is slow to warm. Measured on two unrelated modules, both flipping to
5085    // `ok` between 22s and 32s after restart.
5086    //
5087    // Left as-is because spreading the first probe is what keeps a fleet-wide
5088    // restart from firing fourteen simultaneous probes into a cold machine. The
5089    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5090    // that thundering herd for a faster first reading.
5091    let jitter_span = (cadence_ms / 10).max(1);
5092    let hash = module_id.as_bytes().iter().fold(
5093        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5094        |acc, byte| {
5095            acc.wrapping_mul(1099511628211)
5096                .wrapping_add(u64::from(*byte))
5097        },
5098    );
5099    cadence + Duration::from_millis(hash % jitter_span)
5100}
5101
5102#[cfg(test)]
5103mod tests {
5104    use super::*;
5105
5106    #[test]
5107    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5108        let handle = SupervisorHandle::new();
5109        let module_id = "readded-tombstone";
5110        handle.record_rescan_removal(module_id);
5111        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5112
5113        handle.apply_identity_configuration(&ModuleSpec {
5114            module_id: module_id.to_string(),
5115            program: PathBuf::from("/test/module"),
5116            args: Vec::new(),
5117            env: Vec::new(),
5118            reserved: false,
5119            reserved_prefixes: Vec::new(),
5120            protocol: ModuleProtocol::Subc,
5121            overlap: Default::default(),
5122        });
5123
5124        assert!(
5125            handle.removal_tombstone_age_ms(module_id).is_none(),
5126            "a re-added module must not retain a stale removal tombstone"
5127        );
5128    }
5129
5130    /// What one module's owner looked like from the control plane at the
5131    /// instant after its first process was spawned.
5132    #[derive(Debug, PartialEq, Eq)]
5133    struct OwnerInSpawnWindow {
5134        module_id: String,
5135        configured: bool,
5136        on_roster: bool,
5137        admission_refusal: Option<&'static str>,
5138    }
5139
5140    /// A supervised module's process can connect, register, sync its scopes
5141    /// and describe them as soon as it is spawned, which is BEFORE the
5142    /// supervisor puts the module on the roster. In that window the owner must
5143    /// already read as configured, so a scoped `route.open` against it is
5144    /// refused as retryable `scope_not_synced` and not as terminal
5145    /// `scope_not_live` ("will never sync").
5146    ///
5147    /// The hook runs in exactly that window on every path that takes on a new
5148    /// module, so no race with a real child is needed: `on_roster: false`
5149    /// proves each observation was taken before the roster insert.
5150    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5151    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5152        use crate::scopes::ScopeTable;
5153        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5154
5155        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5156        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5157            module_id: module_id.to_string(),
5158            program,
5159            args: Vec::new(),
5160            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5161                .into_iter()
5162                .map(|key| (key.to_string(), dir.path().display().to_string()))
5163                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5164                .collect(),
5165            reserved: false,
5166            reserved_prefixes: Vec::new(),
5167            protocol: ModuleProtocol::Subc,
5168            overlap: Default::default(),
5169        };
5170        let live = super::terminal_history_tests::fake_aft_stub_path();
5171        let missing = dir.path().join("definitely-missing-module");
5172
5173        let handle = SupervisorHandle::new();
5174        let mut supervisor =
5175            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5176                .with_handle(handle.clone());
5177        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5178        let hook_handle = handle.clone();
5179        let hook_observed = Arc::clone(&observed);
5180        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5181            // Exactly what the control plane computes for a scoped route.open
5182            // naming this module as the owner of a scope it has not synced.
5183            let configured = hook_handle.is_configured(module_id);
5184            let selector = ScopeSelector {
5185                owner: Principal::Reserved {
5186                    module_id: module_id.to_string(),
5187                },
5188                scope_ref: "s".to_string(),
5189                scope_epoch: Some(1),
5190            };
5191            let carrier = Principal::Reserved {
5192                module_id: "carrier".to_string(),
5193            };
5194            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5195                .admit(&carrier, module_id, &selector, configured)
5196            {
5197                Ok(_) => None,
5198                Err(refusal) => Some(refusal.code),
5199            };
5200            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5201                module_id: module_id.to_string(),
5202                configured,
5203                on_roster: hook_handle.get(module_id).is_some(),
5204                admission_refusal,
5205            });
5206        })));
5207
5208        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5209        let configured = supervisor
5210            .supervise_configured(stub("configured", live.clone()), true)
5211            .unwrap();
5212        let with_health = supervisor
5213            .supervise_configured_with_health(
5214                stub("with-health", live.clone()),
5215                true,
5216                HealthConfig::default(),
5217                None,
5218                RestartPolicy::default(),
5219            )
5220            .unwrap();
5221        // The failed-spawn path still puts the module on the roster (as
5222        // failed), so it is configured throughout.
5223        let failed = supervisor
5224            .supervise_configured_with_health(
5225                stub("failed-spawn", missing.clone()),
5226                true,
5227                HealthConfig::default(),
5228                None,
5229                RestartPolicy::default(),
5230            )
5231            .unwrap();
5232        // A failed plain `spawn` puts nothing on the roster, so its mark is
5233        // taken back once the spawn has failed.
5234        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5235
5236        let expected = [
5237            "plain",
5238            "configured",
5239            "with-health",
5240            "failed-spawn",
5241            "spawn-error",
5242        ]
5243        .into_iter()
5244        .map(|module_id| OwnerInSpawnWindow {
5245            module_id: module_id.to_string(),
5246            configured: true,
5247            on_roster: false,
5248            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5249        })
5250        .collect::<Vec<_>>();
5251        assert_eq!(*observed.lock().unwrap(), expected);
5252
5253        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5254            assert!(
5255                handle.get(module_id).is_some(),
5256                "{module_id} is on the roster"
5257            );
5258            assert!(
5259                handle.is_configured(module_id),
5260                "{module_id} stays configured"
5261            );
5262        }
5263        assert!(handle.get("spawn-error").is_none());
5264        assert!(
5265            !handle.is_configured("spawn-error"),
5266            "a plain spawn that failed must not leave its module marked configured"
5267        );
5268
5269        // Leaving the roster clears the mark with it.
5270        handle.retire("failed-spawn");
5271        assert!(!handle.is_configured("failed-spawn"));
5272
5273        for module in [plain, configured, with_health] {
5274            module.stop().await.unwrap();
5275        }
5276        drop(failed);
5277    }
5278
5279    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5280        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5281        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5282            snapshot.process_alive = true;
5283            snapshot.pid = Some(41);
5284            snapshot.spawned_at_ms = Some(42);
5285            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5286            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5287                device: 43,
5288                inode: 44,
5289            });
5290        })
5291        .unwrap();
5292        snapshot
5293    }
5294
5295    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5296        let snapshot = lock_snapshot(snapshot).unwrap();
5297        assert!(!snapshot.process_alive);
5298        assert_eq!(snapshot.pid, None);
5299        assert_eq!(snapshot.spawned_at_ms, None);
5300        assert_eq!(snapshot.spawned_from, None);
5301        assert_eq!(snapshot.spawned_file_identity, None);
5302    }
5303
5304    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5305    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5306        let supervisor =
5307            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5308        let mut runtime = supervisor.runtime_config();
5309        runtime.test_seed_stale_facts_before_enable_spawn = true;
5310        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5311        let mut child = None;
5312        let spec = ModuleSpec {
5313            module_id: "failed-enable-clears-facts".to_string(),
5314            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5315            args: Vec::new(),
5316            env: Vec::new(),
5317            reserved: false,
5318            reserved_prefixes: Vec::new(),
5319            protocol: ModuleProtocol::Subc,
5320            overlap: Default::default(),
5321        };
5322
5323        let result = set_child_enabled(
5324            &spec,
5325            &runtime,
5326            &supervisor.registry,
5327            &supervisor.process_liveness,
5328            &snapshot,
5329            &mut child,
5330            true,
5331        )
5332        .await;
5333
5334        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5335        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5336        assert_snapshot_process_facts_cleared(&snapshot);
5337    }
5338
5339    #[tokio::test]
5340    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5341        let supervisor =
5342            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5343        let runtime = supervisor.runtime_config();
5344        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5345            ModuleState::Restarting,
5346            true,
5347        )));
5348        let spec = ModuleSpec {
5349            module_id: "start-stranded-restarting".to_string(),
5350            program: super::terminal_history_tests::fake_aft_stub_path(),
5351            args: Vec::new(),
5352            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5353            reserved: false,
5354            reserved_prefixes: Vec::new(),
5355            protocol: ModuleProtocol::None,
5356            overlap: Default::default(),
5357        };
5358        let mut child = None;
5359        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5360        assert!(!super::set_child_enabled(
5361            &spec,
5362            &runtime,
5363            &Registry::default(),
5364            &supervisor.process_liveness,
5365            &snapshot,
5366            &mut child,
5367            true
5368        )
5369        .await
5370        .unwrap());
5371        assert!(child.is_none());
5372        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5373        assert!(super::set_child_enabled(
5374            &spec,
5375            &runtime,
5376            &Registry::default(),
5377            &supervisor.process_liveness,
5378            &snapshot,
5379            &mut child,
5380            true
5381        )
5382        .await
5383        .unwrap());
5384        assert_eq!(
5385            lock_snapshot(&snapshot).unwrap().state,
5386            ModuleState::Running
5387        );
5388        let mut child = child.unwrap();
5389        child.start_kill().unwrap();
5390        child.wait().await.unwrap();
5391    }
5392
5393    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5394    async fn failed_reload_spawn_clears_current_process_facts() {
5395        let supervisor =
5396            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5397        let mut runtime = supervisor.runtime_config();
5398        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5399        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5400        let mut child = None;
5401        let spec = ModuleSpec {
5402            module_id: "failed-reload-clears-facts".to_string(),
5403            program: PathBuf::from("/unused/failed-reload-module"),
5404            args: Vec::new(),
5405            env: Vec::new(),
5406            reserved: false,
5407            reserved_prefixes: Vec::new(),
5408            protocol: ModuleProtocol::Subc,
5409            overlap: Default::default(),
5410        };
5411
5412        let result = handle_reload_spawn_failure(
5413            &spec,
5414            &runtime,
5415            &supervisor.process_liveness,
5416            &snapshot,
5417            &mut child,
5418            "forced reload spawn failure".to_string(),
5419        )
5420        .await;
5421
5422        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5423        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5424        assert_snapshot_process_facts_cleared(&snapshot);
5425    }
5426
5427    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5428    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5429        let supervisor =
5430            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5431        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5432        let module = supervisor.supervised_module(
5433            ModuleSpec {
5434                module_id: "drop-clears-facts".to_string(),
5435                program: PathBuf::from("/unused/drop-module"),
5436                args: Vec::new(),
5437                env: Vec::new(),
5438                reserved: false,
5439                reserved_prefixes: Vec::new(),
5440                protocol: ModuleProtocol::Subc,
5441                overlap: Default::default(),
5442            },
5443            supervisor.runtime_config(),
5444            Arc::clone(&snapshot),
5445            None,
5446        );
5447        assert!(!module
5448            .inner
5449            .monitor
5450            .lock()
5451            .unwrap()
5452            .as_ref()
5453            .unwrap()
5454            .is_finished());
5455
5456        drop(module);
5457
5458        assert_eq!(
5459            lock_snapshot(&snapshot).unwrap().state,
5460            ModuleState::Stopped
5461        );
5462        assert_snapshot_process_facts_cleared(&snapshot);
5463    }
5464
5465    #[cfg(unix)]
5466    #[tokio::test]
5467    async fn rescan_preserves_running_protocol_until_respawn() {
5468        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5469        let initial = ModuleSpec {
5470            module_id: "rescan-protocol".into(),
5471            program: PathBuf::from("/bin/sleep"),
5472            args: vec!["60".into()],
5473            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5474                .into_iter()
5475                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5476                .collect(),
5477            reserved: false,
5478            reserved_prefixes: vec![],
5479            protocol: ModuleProtocol::None,
5480            overlap: Default::default(),
5481        };
5482        let supervisor =
5483            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5484        let module = supervisor.spawn(initial.clone()).unwrap();
5485        assert!(module.status().unwrap().live);
5486        let mut next = initial;
5487        next.protocol = ModuleProtocol::Subc;
5488        module
5489            .update_configuration(next.clone(), HealthConfig::default(), None)
5490            .await
5491            .unwrap();
5492        assert!(
5493            module.status().unwrap().live,
5494            "rescan must not require HELLO from the old non-wire process"
5495        );
5496        let runtime = supervisor.runtime_config();
5497        let action = on_child_exit(
5498            &next,
5499            RestartPolicy::default(),
5500            &supervisor.registry,
5501            &module.inner.snapshot,
5502            &runtime.terminal_ring,
5503            &runtime.spawn_events,
5504            &runtime.child_roster,
5505            ExitReport {
5506                kind: ExitKind::Clean,
5507                code: Some(0),
5508                signal: None,
5509                at_ms: unix_ms_now(),
5510            },
5511        )
5512        .await;
5513        assert!(
5514            matches!(action, NextAction::Restart { .. }),
5515            "the old non-wire process's clean exit must restart"
5516        );
5517        module.drain().await.unwrap();
5518    }
5519
5520    #[cfg(unix)]
5521    #[tokio::test]
5522    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5523        use std::os::unix::fs::PermissionsExt;
5524        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5525        let script = dir.join("module.sh");
5526        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5527        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5528        let record_path = dir.join("live-children.json");
5529        let supervisor =
5530            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5531                .with_live_children_record(&record_path);
5532        for (program, args) in [
5533            (PathBuf::from("sleep"), vec!["60".into()]),
5534            (script, vec![]),
5535        ] {
5536            let spec = ModuleSpec {
5537                module_id: "image-identity".into(),
5538                program: program.clone(),
5539                args,
5540                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5541                    .into_iter()
5542                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5543                    .collect(),
5544                reserved: false,
5545                reserved_prefixes: vec![],
5546                protocol: ModuleProtocol::None,
5547                overlap: Default::default(),
5548            };
5549            let module = supervisor.spawn(spec).unwrap();
5550            #[cfg(target_os = "macos")]
5551            {
5552                // SETEXEC confirmation is asynchronous; the orphan record must
5553                // identify the final image, never the intermediate trampoline.
5554                let deadline = Instant::now() + Duration::from_secs(5);
5555                while crate::live_children::read_record(&record_path)
5556                    .unwrap()
5557                    .iter()
5558                    .all(|entry| entry.executable.is_none())
5559                {
5560                    assert!(Instant::now() < deadline, "module image was not confirmed");
5561                    tokio::time::sleep(Duration::from_millis(5)).await;
5562                }
5563            }
5564            let entry = crate::live_children::read_record(&record_path)
5565                .unwrap()
5566                .pop()
5567                .unwrap();
5568            let observed = subc_os::Process::open(entry.pid)
5569                .unwrap()
5570                .unwrap()
5571                .observe()
5572                .unwrap();
5573            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5574            module.drain().await.unwrap();
5575            assert_eq!(
5576                verdict,
5577                crate::live_children::IdentityVerdict::Matches,
5578                "program {program:?}: recorded {entry:?}, observed {observed:?}"
5579            );
5580        }
5581    }
5582
5583    #[cfg(unix)]
5584    fn http_fixture(
5585        dir: &std::path::Path,
5586        url: &str,
5587        threshold: u32,
5588    ) -> crate::daemon_config::ConfiguredModule {
5589        let path = dir.join("subc.jsonc");
5590        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5591            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5592            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5593            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5594            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5595        }}}).to_string()).unwrap();
5596        crate::daemon_config::load(&path)
5597            .unwrap()
5598            .unwrap()
5599            .modules
5600            .pop()
5601            .unwrap()
5602    }
5603
5604    #[cfg(unix)]
5605    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5606        timeout(Duration::from_secs(5), async {
5607            loop {
5608                if module.status().unwrap().health.status == status {
5609                    break;
5610                }
5611                sleep(Duration::from_millis(5)).await;
5612            }
5613        })
5614        .await
5615        .unwrap_or_else(|_| {
5616            panic!(
5617                "expected {status:?}, got {:?}",
5618                module.status().unwrap().health
5619            )
5620        });
5621    }
5622
5623    #[cfg(unix)]
5624    #[tokio::test]
5625    async fn http_health_status_flips_ok_failing_ok() {
5626        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5627        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5628        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5629        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5630        let serving_status = status.clone();
5631        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5632        let server = tokio::spawn(async move {
5633            loop {
5634                let (mut stream, _) = listener.accept().await.unwrap();
5635                let mut request = [0u8; 2048];
5636                let count = stream.read(&mut request).await.unwrap();
5637                assert!(count > 0, "a probe must send an HTTP request");
5638                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5639                let body = if code == 200 {
5640                    "ready"
5641                } else {
5642                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5643                };
5644                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5645                let _ = stream.write_all(response.as_bytes()).await;
5646            }
5647        });
5648        let configured = http_fixture(&dir, &url, 1000);
5649        let module =
5650            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5651                .supervise_configured_with_health(
5652                    configured.module_spec(),
5653                    true,
5654                    configured.health,
5655                    configured.drain_timeout_ms,
5656                    configured.restart,
5657                )
5658                .unwrap();
5659        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5660        status.store(503, std::sync::atomic::Ordering::SeqCst);
5661        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5662        assert!(module
5663            .status()
5664            .unwrap()
5665            .health
5666            .detail
5667            .unwrap()
5668            .contains("scratch failure"));
5669        status.store(200, std::sync::atomic::Ordering::SeqCst);
5670        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5671        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5672        let before = module.status().unwrap();
5673        let (spec, mut health) = module.configuration().unwrap();
5674        health.http = None;
5675        module
5676            .update_configuration(spec.clone(), health.clone(), Some(10))
5677            .await
5678            .unwrap();
5679        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5680        health.http = Some(url);
5681        module
5682            .update_configuration(spec, health, Some(10))
5683            .await
5684            .unwrap();
5685        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5686        assert_eq!(
5687            module.status().unwrap().pid,
5688            before.pid,
5689            "changing a probe must apply live, not restart its process"
5690        );
5691        let (spec, mut health) = module.configuration().unwrap();
5692        health.failure_threshold = 2;
5693        module
5694            .update_configuration(spec, health, Some(10))
5695            .await
5696            .unwrap();
5697        status.store(503, std::sync::atomic::Ordering::SeqCst);
5698        timeout(Duration::from_secs(5), async {
5699            while module.status().unwrap().spawn_generation == before.spawn_generation {
5700                sleep(Duration::from_millis(5)).await;
5701            }
5702        })
5703        .await
5704        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5705        module.drain().await.unwrap();
5706        server.abort();
5707    }
5708
5709    #[cfg(unix)]
5710    #[tokio::test]
5711    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5712        let dir = subc_test_support::TestTempDir::new("http-refused");
5713        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5714        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5715        drop(unused);
5716        let configured = http_fixture(&dir, &url, 2);
5717        let module =
5718            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5719                .supervise_configured_with_health(
5720                    configured.module_spec(),
5721                    true,
5722                    configured.health,
5723                    configured.drain_timeout_ms,
5724                    configured.restart,
5725                )
5726                .unwrap();
5727        let before = module.status().unwrap().spawn_generation;
5728        timeout(Duration::from_secs(5), async {
5729            loop {
5730                let status = module.status().unwrap();
5731                if status.spawn_generation > before {
5732                    assert!(status.lifetime_restarts > 0);
5733                    break;
5734                }
5735                sleep(Duration::from_millis(5)).await;
5736            }
5737        })
5738        .await
5739        .expect("sustained HTTP refusal must trigger the health restart policy");
5740        module.drain().await.unwrap();
5741    }
5742
5743    #[cfg(unix)]
5744    #[tokio::test]
5745    async fn http_health_timeout_honours_deadline() {
5746        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5747        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5748        let server = tokio::spawn(async move {
5749            let _held = listener.accept().await.unwrap();
5750            std::future::pending::<()>().await;
5751        });
5752        let error = timeout(
5753            Duration::from_secs(1),
5754            probe_http_health(&url, Duration::from_millis(10)),
5755        )
5756        .await
5757        .expect("the probe must enforce its own deadline")
5758        .unwrap_err();
5759        server.abort();
5760        assert!(error.to_string().contains("timed out"));
5761    }
5762
5763    #[cfg(unix)]
5764    #[tokio::test]
5765    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5766        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5767        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5768        let url = format!(
5769            "http://localhost:{}/healthz",
5770            listener.local_addr().unwrap().port()
5771        );
5772        let server = tokio::spawn(async move {
5773            let (mut stream, _) = listener.accept().await.unwrap();
5774            let mut request = [0u8; 2048];
5775            assert!(stream.read(&mut request).await.unwrap() > 0);
5776            stream
5777                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5778                .await
5779                .unwrap();
5780        });
5781        // The deadline only bounds a hang. A probe that never tried the IPv6
5782        // address would be refused on 127.0.0.1 and fail at once, so a longer
5783        // deadline does not weaken the assertion; one second timed out under a
5784        // loaded parallel test run.
5785        assert_eq!(
5786            probe_http_health(&url, Duration::from_secs(10))
5787                .await
5788                .unwrap()
5789                .status,
5790            HealthStatus::Ok
5791        );
5792        server.await.unwrap();
5793    }
5794
5795    #[cfg(unix)]
5796    #[tokio::test]
5797    async fn http_health_timeout_keeps_partial_status_and_body() {
5798        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5799        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5800        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5801        let server = tokio::spawn(async move {
5802            let (mut stream, _) = listener.accept().await.unwrap();
5803            let mut request = [0u8; 2048];
5804            assert!(stream.read(&mut request).await.unwrap() > 0);
5805            stream
5806                .write_all(
5807                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5808                )
5809                .await
5810                .unwrap();
5811            std::future::pending::<()>().await;
5812        });
5813        let error = probe_http_health(&url, Duration::from_secs(1))
5814            .await
5815            .unwrap_err()
5816            .to_string();
5817        server.abort();
5818        assert!(
5819            error.contains("timed out")
5820                && error.contains("503 Unavailable")
5821                && error.contains("partial diagnostic"),
5822            "{error}"
5823        );
5824    }
5825
5826    #[cfg(unix)]
5827    #[tokio::test]
5828    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5829        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5830        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5831        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5832        let server = tokio::spawn(async move {
5833            let (mut stream, _) = listener.accept().await.unwrap();
5834            let mut request = [0u8; 2048];
5835            assert!(stream.read(&mut request).await.unwrap() > 0);
5836            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5837            let response = format!(
5838                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5839                body.len()
5840            );
5841            stream.write_all(response.as_bytes()).await.unwrap();
5842        });
5843        let error = probe_http_health(&url, Duration::from_secs(1))
5844            .await
5845            .unwrap_err()
5846            .to_string();
5847        server.await.unwrap();
5848        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5849        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5850        assert!(!error.contains("not-in-diagnostic"));
5851    }
5852
5853    #[cfg(unix)]
5854    #[tokio::test]
5855    async fn http_health_real_nats_server_monitoring() {
5856        if std::process::Command::new("nats-server")
5857            .arg("--version")
5858            .env("XDG_DATA_HOME", std::env::temp_dir())
5859            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5860            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5861            .output()
5862            .is_err()
5863        {
5864            eprintln!(
5865                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5866            );
5867            return;
5868        }
5869        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5870        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5871        let port = monitor.local_addr().unwrap().port();
5872        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5873        let client_port = client.local_addr().unwrap().port();
5874        let config = dir.join("server.conf");
5875        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5876        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5877        configured.program = PathBuf::from("nats-server");
5878        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5879        drop(monitor);
5880        drop(client);
5881        let module =
5882            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5883                .supervise_configured_with_health(
5884                    configured.module_spec(),
5885                    true,
5886                    configured.health,
5887                    configured.drain_timeout_ms,
5888                    configured.restart,
5889                )
5890                .unwrap();
5891        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5892        module.drain().await.unwrap();
5893    }
5894
5895    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5896    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5897        let supervisor =
5898            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5899        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5900        let initial = ModuleSpec {
5901            module_id: "rescan-preserves-spawn-facts".to_string(),
5902            program: PathBuf::from("/spawned/module"),
5903            args: Vec::new(),
5904            env: Vec::new(),
5905            reserved: false,
5906            reserved_prefixes: Vec::new(),
5907            protocol: ModuleProtocol::Subc,
5908            overlap: Default::default(),
5909        };
5910        let module = supervisor.supervised_module(
5911            initial.clone(),
5912            supervisor.runtime_config(),
5913            snapshot,
5914            None,
5915        );
5916        let before = module.status().unwrap();
5917        let mut replacement = initial;
5918        replacement.program = PathBuf::from("/rescanned/replacement-module");
5919
5920        module
5921            .update_configuration(replacement, HealthConfig::default(), None)
5922            .await
5923            .unwrap();
5924
5925        let after = module.status().unwrap();
5926        assert_eq!(after.pid, before.pid);
5927        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5928        assert_eq!(after.spawned_from, before.spawned_from);
5929        drop(module);
5930    }
5931}
5932
5933fn unix_ms_now() -> u64 {
5934    SystemTime::now()
5935        .duration_since(UNIX_EPOCH)
5936        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5937        .unwrap_or(0)
5938}
5939
5940async fn supervise_loop(
5941    mut spec: ModuleSpec,
5942    mut runtime: SupervisorRuntimeConfig,
5943    registry: Arc<Registry>,
5944    process_liveness: Arc<SupervisorProcessLiveness>,
5945    snapshot: SharedSnapshot,
5946    mut child: Option<SupervisedChild>,
5947    mut commands: mpsc::Receiver<SupervisorCommand>,
5948) {
5949    let mut health_probe = HealthProbeRuntime::default();
5950    // All restart backoffs run here, including health and operator requests.
5951    // While one is pending the loop serves commands, so disable or drain can
5952    // cancel the replacement without spawning a process just to stop it.
5953    let mut pending_respawn: Option<PendingRespawn> = None;
5954    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5955    // before anything else so a stop that interrupted a swap runs at once.
5956    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5957    loop {
5958        #[cfg(target_os = "macos")]
5959        if let Some(active) = child.as_mut() {
5960            active.confirm_privacy_exec().await;
5961        }
5962        if let Some(scheduled) = runtime
5963            .scheduled_respawn
5964            .lock()
5965            .unwrap_or_else(|p| p.into_inner())
5966            .take()
5967        {
5968            pending_respawn = Some(scheduled);
5969        }
5970        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5971            pending_respawn = None;
5972            cancel_deferred_reload(
5973                &runtime,
5974                &spec.module_id,
5975                "respawn cancelled by a supervisor command",
5976            );
5977        }
5978        if child.is_none() && pending_respawn.is_none() {
5979            cancel_deferred_reload(
5980                &runtime,
5981                &spec.module_id,
5982                "respawn cancelled before a replacement was spawned",
5983            );
5984            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5985                state.respawn_pending = false;
5986                state.coalesced_restart_pending = false;
5987                if matches!(
5988                    state.state,
5989                    ModuleState::Restarting
5990                        | ModuleState::Starting
5991                        | ModuleState::Draining
5992                        | ModuleState::Unresponsive
5993                ) {
5994                    error!(module_id = %spec.module_id, state = ?state.state,
5995                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5996                    state.state = ModuleState::Failed;
5997                    clear_current_process_facts(state);
5998                }
5999            });
6000        }
6001        if let Some(command) = requeued.pop_front() {
6002            if !handle_supervisor_command(
6003                command,
6004                &mut spec,
6005                &mut runtime,
6006                &registry,
6007                &process_liveness,
6008                &snapshot,
6009                &mut child,
6010                &mut commands,
6011                &mut requeued,
6012            )
6013            .await
6014            {
6015                return;
6016            }
6017            if child.is_some() || !respawn_still_pending(&snapshot) {
6018                pending_respawn = None;
6019                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6020                    state.respawn_pending = false
6021                });
6022            }
6023            continue;
6024        }
6025        if child.is_some() {
6026            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
6027            let probe_sleep = sleep(health_probe.wake_after());
6028            tokio::pin!(probe_sleep);
6029            let active_child = child.as_mut().expect("child checked above");
6030            tokio::select! {
6031                wait_result = active_child.wait() => {
6032                    // Every arm below that gives up on the CHILD must keep the
6033                    // supervision task itself alive (child = None, loop
6034                    // continues into command-serving mode). Returning here
6035                    // closes the command channel, which makes the module
6036                    // permanently unrestartable in-band: a clean child exit
6037                    // of an enabled module once wedged the fleet this way
6038                    // ('supervisor command channel is closed') and required a
6039                    // full daemon restart to recover.
6040                    let exit_report = match wait_result {
6041                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6042                        Err(err) => {
6043                            active_child.drain_stderr(&spec.module_id).await;
6044                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6045                            // Every other exit path (on_child_exit's Clean/Crash arms,
6046                            // the reload-registration-failure path) records a terminal
6047                            // before moving on. Without one here, a module whose wait()
6048                            // itself errored (e.g. already reaped) leaves no terminal
6049                            // record at all -- an empty ring reads as "nothing died".
6050                            record_wait_error_terminal(
6051                                &spec.module_id,
6052                                &runtime.terminal_ring,
6053                                &runtime.spawn_events,
6054                            );
6055                            untrack_if_registration_released(
6056                                &process_liveness,
6057                                &registry,
6058                                &spec.module_id,
6059                                &snapshot,
6060                            );
6061                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6062                            child = None;
6063                            continue;
6064                        }
6065                    };
6066                    active_child.drain_stderr(&spec.module_id).await;
6067
6068                    let next = on_child_exit(
6069                        &spec,
6070                        runtime.restart_policy,
6071                        &registry,
6072                        &snapshot,
6073                        &runtime.terminal_ring,
6074                        &runtime.spawn_events,
6075                        &runtime.child_roster,
6076                        exit_report,
6077                    ).await;
6078                    // The exit is recorded, so a daemon shutdown may stop
6079                    // waiting for this child (see `SupervisedChild::wait`).
6080                    active_child.release_roster();
6081                    match next {
6082                        NextAction::Stop { registration_released } => {
6083                            if registration_released {
6084                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6085                            }
6086                            child = None;
6087                        }
6088                        NextAction::Restart { schedule } => {
6089                            let delay = schedule.map_or(
6090                                runtime.restart_policy.delay_for_restart(0),
6091                                |schedule| schedule.delay,
6092                            );
6093                            if let Some(schedule) = schedule {
6094                                log_crash_respawn(&spec.module_id, schedule);
6095                            }
6096                            // The exited child is fully recorded at this point,
6097                            // so release it and count the backoff down in the
6098                            // command-serving branch below rather than sleeping
6099                            // here: commands cannot be received from inside this
6100                            // select arm, and an operator disable or drain that
6101                            // arrives during the backoff must cancel the pending
6102                            // respawn instead of waiting for it to spawn first.
6103                            child = None;
6104                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6105                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6106                        }
6107                    }
6108                }
6109                command = commands.recv() => {
6110                    let Some(command) = command else {
6111                        return;
6112                    };
6113                    if !handle_supervisor_command(
6114                        command,
6115                        &mut spec,
6116                        &mut runtime,
6117                        &registry,
6118                        &process_liveness,
6119                        &snapshot,
6120                        &mut child,
6121                        &mut commands,
6122                        &mut requeued,
6123                    ).await {
6124                        return;
6125                    }
6126                }
6127                _ = &mut probe_sleep => {
6128                    if health_probe.due() {
6129                        run_health_probe_cycle(
6130                            &spec,
6131                            &runtime,
6132                            &registry,
6133                            &process_liveness,
6134                            &snapshot,
6135                            &mut child,
6136                        ).await;
6137                        if child.is_some() {
6138                            health_probe.schedule_next(&spec, runtime.health.cadence);
6139                        }
6140                    }
6141                }
6142            }
6143        } else if let Some(pending) = pending_respawn {
6144            tokio::select! {
6145                _ = sleep_until(pending.deadline) => {
6146                    pending_respawn = None;
6147                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6148                    // A command handled below while the backoff elapsed may
6149                    // have stopped the module; never respawn past an operator's
6150                    // disable or drain.
6151                    if !respawn_still_pending(&snapshot) {
6152                        continue;
6153                    }
6154                    // The daemon began shutting down during the backoff: the
6155                    // spawn would be refused anyway, and refusing it here
6156                    // leaves the module stopped instead of reporting a
6157                    // failed restart.
6158                    if runtime.child_roster.is_closed() {
6159                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6160                            state.state = ModuleState::Stopped;
6161                        });
6162                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6163                        continue;
6164                    }
6165                    if let Err(err) = release_dead_registration(
6166                        &registry,
6167                        runtime.forwarding.as_deref(),
6168                        &snapshot,
6169                        &spec.module_id,
6170                    ).await {
6171                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6172                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6173                        continue;
6174                    }
6175
6176                    if matches!(pending.kind, RespawnKind::Reload) {
6177                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6178                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6179                        if let Some(reply) = reply { let _ = reply.send(result); }
6180                        continue;
6181                    }
6182                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6183                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6184                        Ok(next_child) => {
6185                            child = Some(next_child);
6186                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6187                        }
6188                        Err(err) => {
6189                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6190                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6191                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6192                        }
6193                    }
6194                }
6195                command = commands.recv() => {
6196                    let Some(command) = command else {
6197                        return;
6198                    };
6199                    if !handle_supervisor_command(
6200                        command,
6201                        &mut spec,
6202                        &mut runtime,
6203                        &registry,
6204                        &process_liveness,
6205                        &snapshot,
6206                        &mut child,
6207                        &mut commands,
6208                        &mut requeued,
6209                    ).await {
6210                        return;
6211                    }
6212                    // Reconcile the pending respawn with what the command did:
6213                    // a start may already have spawned a fresh child,
6214                    // while a disable or drain moved the snapshot out of the
6215                    // state the respawn was counting down from.
6216                    if child.is_some() || !respawn_still_pending(&snapshot) {
6217                        pending_respawn = None;
6218                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6219                    }
6220                }
6221            }
6222        } else {
6223            let Some(command) = commands.recv().await else {
6224                return;
6225            };
6226            if !handle_supervisor_command(
6227                command,
6228                &mut spec,
6229                &mut runtime,
6230                &registry,
6231                &process_liveness,
6232                &snapshot,
6233                &mut child,
6234                &mut commands,
6235                &mut requeued,
6236            )
6237            .await
6238            {
6239                return;
6240            }
6241        }
6242    }
6243}
6244
6245fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6246    info!(
6247        module_id,
6248        restart_in_window = schedule.restart_in_window,
6249        delay_ms = schedule.delay.as_millis() as u64,
6250        "respawning after crash"
6251    );
6252}
6253
6254/// Whether the respawn a backoff was counting down to is still wanted. A
6255/// disable or drain handled while the backoff elapsed moves the snapshot out
6256/// of `Restarting`, and the operator's stop must win over the pending respawn,
6257/// so every sleep-then-spawn path re-validates against the live snapshot
6258/// instead of assuming the state it left behind still holds.
6259fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6260    matches!(
6261        lock_snapshot(snapshot),
6262        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6263    )
6264}
6265
6266enum NextAction {
6267    Stop {
6268        registration_released: bool,
6269    },
6270    Restart {
6271        schedule: Option<CrashRestartSchedule>,
6272    },
6273}
6274
6275#[allow(clippy::too_many_arguments)]
6276async fn handle_supervisor_command(
6277    command: SupervisorCommand,
6278    spec: &mut ModuleSpec,
6279    runtime: &mut SupervisorRuntimeConfig,
6280    registry: &Arc<Registry>,
6281    process_liveness: &SupervisorProcessLiveness,
6282    snapshot: &SharedSnapshot,
6283    child: &mut Option<SupervisedChild>,
6284    commands: &mut mpsc::Receiver<SupervisorCommand>,
6285    requeued: &mut VecDeque<SupervisorCommand>,
6286) -> bool {
6287    match command {
6288        SupervisorCommand::Drain { reply } => {
6289            // A plain stop runs no forwarding drain, so nothing reaches the
6290            // module over its connection before the wait: ask by signal.
6291            let result = drain_optional_child(
6292                &spec.module_id,
6293                spec.protocol,
6294                StopNotice::NotSent,
6295                registry,
6296                runtime.forwarding.as_deref(),
6297                snapshot,
6298                &runtime.terminal_ring,
6299                &runtime.spawn_events,
6300                child,
6301                runtime.drain_timeout,
6302                ModuleState::Stopped,
6303                None,
6304            )
6305            .await;
6306            let registration_released = result.is_ok();
6307            let _ = reply.send(result);
6308            if registration_released {
6309                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6310            }
6311            false
6312        }
6313        SupervisorCommand::Retire { reply } => {
6314            let result = async {
6315                let stop_notice = begin_forwarding_drain_if_configured(
6316                    spec,
6317                    runtime,
6318                    registry,
6319                    snapshot,
6320                    None,
6321                    RouteCloseReason::Disable,
6322                )
6323                .await?;
6324                drain_optional_child(
6325                    &spec.module_id,
6326                    spec.protocol,
6327                    stop_notice,
6328                    registry,
6329                    runtime.forwarding.as_deref(),
6330                    snapshot,
6331                    &runtime.terminal_ring,
6332                    &runtime.spawn_events,
6333                    child,
6334                    runtime.drain_timeout,
6335                    ModuleState::Stopped,
6336                    None,
6337                )
6338                .await
6339            }
6340            .await;
6341            let registration_released = result.is_ok();
6342            let _ = reply.send(result);
6343            if registration_released {
6344                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6345            }
6346            false
6347        }
6348        SupervisorCommand::Restart {
6349            drain_timeout_ms,
6350            received_at_generation,
6351            queued_at,
6352            reply,
6353        } => {
6354            // Without this line a restart that waited in the queue (behind a
6355            // health probe cycle or another command) was invisible: the log
6356            // showed only the drain timing out, minutes after the operator's call.
6357            info!(
6358                module_id = %spec.module_id,
6359                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6360                "restart command dequeued"
6361            );
6362            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6363            // caller whose own request lane rides the module being restarted: the
6364            // caller's in-flight request keeps the drain from quiescing, the drain
6365            // keeps the restart from completing, and the completion keeps the reply
6366            // from releasing the caller — so the drain always timed out and cut the
6367            // initiator with a GOODBYE, even on a healthy module. Replying once the
6368            // restart is validated lets a self-lane caller settle, which is exactly
6369            // what makes the drain succeed. Completion is observable via
6370            // supervisor.list / module status; a post-ack failure lands the module
6371            // in a visible terminal state below rather than in a reply nobody can
6372            // receive.
6373            let validation = match lock_snapshot(snapshot) {
6374                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6375                    module_id: spec.module_id.clone(),
6376                }),
6377                Ok(_) => Ok(()),
6378                Err(err) => Err(err),
6379            };
6380            let initiated = validation.is_ok();
6381            let _ = reply.send(validation);
6382            // A restart asks for a fresh process. Commands run one at a time,
6383            // so a restart queued behind another restart (two operator calls
6384            // in quick succession) is dequeued the moment the first one has
6385            // spawned its replacement -- before that process has sent HELLO.
6386            // Running it would drain and kill the process the first restart
6387            // just produced, which is the opposite of what both callers asked
6388            // for. If a process spawned after this request was received is
6389            // still supervised, the request is already satisfied. Not when the
6390            // configuration changed since that spawn: then the newer process
6391            // predates the spec this restart may exist to apply.
6392            let satisfied_by_generation = if initiated && child.is_some() {
6393                lock_snapshot(snapshot).ok().and_then(|state| {
6394                    (state.spawn_generation > received_at_generation
6395                        && !state.configuration_updated_since_spawn)
6396                        .then_some(state.spawn_generation)
6397                })
6398            } else {
6399                None
6400            };
6401            let satisfied_by_pending = initiated
6402                && child.is_none()
6403                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6404                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6405                    if pending {
6406                        state.coalesced_restart_pending = true;
6407                    }
6408                    pending
6409                });
6410            if satisfied_by_pending {
6411                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6412            } else if let Some(generation) = satisfied_by_generation {
6413                info!(
6414                    module_id = %spec.module_id,
6415                    received_at_generation,
6416                    "restart already satisfied by generation {generation}; not restarting again"
6417                );
6418            } else if initiated {
6419                // Precedence: this restart's operator override, else the module's
6420                // configured budget (already resolved into the runtime).
6421                let drain_timeout = drain_timeout_ms
6422                    .map(Duration::from_millis)
6423                    .unwrap_or(runtime.drain_timeout);
6424                if let Err(err) = restart_child(
6425                    spec,
6426                    runtime,
6427                    registry,
6428                    process_liveness,
6429                    snapshot,
6430                    child,
6431                    drain_timeout,
6432                )
6433                .await
6434                {
6435                    warn!(
6436                        module_id = %spec.module_id,
6437                        error = %err,
6438                        "operator restart failed after initiation ack; module state carries the outcome"
6439                    );
6440                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6441                        state.state = ModuleState::Failed;
6442                        clear_current_process_facts(state);
6443                    });
6444                }
6445            }
6446            true
6447        }
6448        SupervisorCommand::Reload { reply } => {
6449            let result =
6450                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6451            if result.is_ok()
6452                && runtime
6453                    .scheduled_respawn
6454                    .lock()
6455                    .unwrap_or_else(|p| p.into_inner())
6456                    .as_ref()
6457                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6458            {
6459                *runtime
6460                    .deferred_reload_reply
6461                    .lock()
6462                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6463            } else {
6464                let _ = reply.send(result);
6465            }
6466            true
6467        }
6468        SupervisorCommand::SetEnabled { enabled, reply } => {
6469            let result = set_child_enabled(
6470                spec,
6471                runtime,
6472                registry,
6473                process_liveness,
6474                snapshot,
6475                child,
6476                enabled,
6477            )
6478            .await;
6479            let _ = reply.send(result);
6480            true
6481        }
6482        SupervisorCommand::UpdateConfiguration {
6483            spec: next_spec,
6484            health,
6485            drain_timeout_ms,
6486            reply,
6487        } => {
6488            if let Some(handle) = &runtime.supervisor_handle {
6489                handle.apply_identity_configuration(&next_spec);
6490            }
6491            *spec = next_spec;
6492            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6493                state.configuration_updated_since_spawn = true;
6494            });
6495            let health_changed = runtime.health != health;
6496            runtime.health = health;
6497            // Reset the cadence and old endpoint's failure streak on a live
6498            // health-policy change rather than waiting for its old deadline.
6499            if health_changed {
6500                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6501                    state.health = ModuleHealthStatus::default();
6502                });
6503            }
6504            runtime.drain_timeout = drain_timeout_ms
6505                .map(Duration::from_millis)
6506                .unwrap_or(runtime.default_drain_timeout);
6507            *runtime
6508                .effective_drain_timeout
6509                .lock()
6510                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6511            let _ = reply.send(());
6512            true
6513        }
6514        SupervisorCommand::Swap {
6515            ready_timeout,
6516            reply,
6517        } => {
6518            let end = swap::run_swap(
6519                spec,
6520                runtime,
6521                registry,
6522                process_liveness,
6523                snapshot,
6524                child,
6525                commands,
6526                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6527                reply,
6528            )
6529            .await;
6530            requeued.extend(end.requeue);
6531            true
6532        }
6533    }
6534}
6535
6536async fn restart_child(
6537    spec: &ModuleSpec,
6538    runtime: &SupervisorRuntimeConfig,
6539    registry: &Registry,
6540    process_liveness: &SupervisorProcessLiveness,
6541    snapshot: &SharedSnapshot,
6542    child: &mut Option<SupervisedChild>,
6543    drain_timeout: Duration,
6544) -> Result<(), SuperviseError> {
6545    // Restart cycles a running module; it must not silently start a disabled one.
6546    if !lock_snapshot(snapshot)?.enabled {
6547        return Err(SuperviseError::Disabled {
6548            module_id: spec.module_id.clone(),
6549        });
6550    }
6551    let stop_notice = begin_forwarding_drain_with_timeout(
6552        spec,
6553        runtime,
6554        registry,
6555        snapshot,
6556        None,
6557        RouteCloseReason::Restart,
6558        drain_timeout,
6559    )
6560    .await?;
6561
6562    if child.is_some() {
6563        drain_optional_child(
6564            &spec.module_id,
6565            spec.protocol,
6566            stop_notice,
6567            registry,
6568            runtime.forwarding.as_deref(),
6569            snapshot,
6570            &runtime.terminal_ring,
6571            &runtime.spawn_events,
6572            child,
6573            drain_timeout,
6574            ModuleState::Restarting,
6575            Some(true),
6576        )
6577        .await?;
6578    } else {
6579        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6580            state.enabled = true;
6581            state.state = ModuleState::Restarting;
6582            clear_current_process_facts(state);
6583        })?;
6584        release_dead_registration(
6585            registry,
6586            runtime.forwarding.as_deref(),
6587            snapshot,
6588            &spec.module_id,
6589        )
6590        .await?;
6591    }
6592
6593    reset_restart_count(snapshot, &spec.module_id)?;
6594    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6595    schedule_respawn(
6596        runtime,
6597        snapshot,
6598        &spec.module_id,
6599        runtime.restart_policy.backoff,
6600        RespawnKind::Spawn,
6601    )
6602}
6603
6604async fn reload_child(
6605    spec: &ModuleSpec,
6606    runtime: &SupervisorRuntimeConfig,
6607    registry: &Registry,
6608    process_liveness: &SupervisorProcessLiveness,
6609    snapshot: &SharedSnapshot,
6610    child: &mut Option<SupervisedChild>,
6611) -> Result<(), SuperviseError> {
6612    // Reload cycles a running module; it must not silently start a disabled one.
6613    if !lock_snapshot(snapshot)?.enabled {
6614        return Err(SuperviseError::Disabled {
6615            module_id: spec.module_id.clone(),
6616        });
6617    }
6618    let stop_notice = begin_forwarding_drain(
6619        spec,
6620        runtime,
6621        registry,
6622        snapshot,
6623        Some(true),
6624        RouteCloseReason::Reload,
6625    )
6626    .await?;
6627
6628    if child.is_some() {
6629        drain_optional_child(
6630            &spec.module_id,
6631            spec.protocol,
6632            stop_notice,
6633            registry,
6634            runtime.forwarding.as_deref(),
6635            snapshot,
6636            &runtime.terminal_ring,
6637            &runtime.spawn_events,
6638            child,
6639            runtime.drain_timeout,
6640            ModuleState::Restarting,
6641            Some(true),
6642        )
6643        .await?;
6644    } else {
6645        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6646            state.enabled = true;
6647            state.state = ModuleState::Restarting;
6648            clear_current_process_facts(state);
6649        })?;
6650        release_dead_registration(
6651            registry,
6652            runtime.forwarding.as_deref(),
6653            snapshot,
6654            &spec.module_id,
6655        )
6656        .await?;
6657    }
6658
6659    reset_restart_count(snapshot, &spec.module_id)?;
6660    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6661    schedule_respawn(
6662        runtime,
6663        snapshot,
6664        &spec.module_id,
6665        runtime.restart_policy.backoff,
6666        RespawnKind::Reload,
6667    )
6668}
6669
6670async fn finish_reload_child(
6671    spec: &ModuleSpec,
6672    runtime: &SupervisorRuntimeConfig,
6673    registry: &Registry,
6674    process_liveness: &SupervisorProcessLiveness,
6675    snapshot: &SharedSnapshot,
6676    child: &mut Option<SupervisedChild>,
6677) -> Result<(), SuperviseError> {
6678    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6679    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6680        Ok(next_child) => next_child,
6681        Err(err) => {
6682            return handle_reload_spawn_failure(
6683                spec,
6684                runtime,
6685                process_liveness,
6686                snapshot,
6687                child,
6688                format!("new child failed to spawn: {err}"),
6689            )
6690            .await;
6691        }
6692    };
6693    *child = Some(next_child);
6694
6695    let wait_outcome = {
6696        let active_child = child.as_mut().expect("new reload child was just stored");
6697        wait_for_registration_after_reload(
6698            registry,
6699            &spec.module_id,
6700            snapshot,
6701            active_child,
6702            REGISTRY_RELEASE_TIMEOUT,
6703        )
6704        .await?
6705    };
6706
6707    match wait_outcome {
6708        RegistrationWaitOutcome::Registered => {
6709            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6710            Ok(())
6711        }
6712        RegistrationWaitOutcome::Exited(exit_report) => {
6713            if let Some(active_child) = child.as_mut() {
6714                active_child.drain_stderr(&spec.module_id).await;
6715            }
6716            // Keep the reaped child's roster guard until its terminal is written.
6717            // Shutdown waits on that guard, not on the child Option used for respawn.
6718            let mut exited_child = child.take().expect("exited reload child is still stored");
6719            #[cfg(test)]
6720            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6721                gate.reached.notify_one();
6722                gate.resume.notified().await;
6723            }
6724            let result = handle_reload_child_registration_failure(
6725                spec,
6726                runtime,
6727                registry,
6728                process_liveness,
6729                snapshot,
6730                child,
6731                ReloadRegistrationFailure {
6732                    exit_report: registration_failure_exit_report(exit_report),
6733                    reason: exited_child
6734                        .spawn_failure
6735                        .clone()
6736                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6737                },
6738            )
6739            .await;
6740            exited_child.release_roster();
6741            result
6742        }
6743        RegistrationWaitOutcome::TimedOut => {
6744            let mut timed_out_child = child
6745                .take()
6746                .expect("timed-out reload child is still running");
6747            timed_out_child
6748                .start_kill()
6749                .map_err(|source| SuperviseError::Kill {
6750                    module_id: spec.module_id.clone(),
6751                    source,
6752                })?;
6753            let status = timed_out_child
6754                .wait()
6755                .await
6756                .map_err(|source| SuperviseError::Wait {
6757                    module_id: spec.module_id.clone(),
6758                    source,
6759                })?;
6760            timed_out_child.drain_stderr(&spec.module_id).await;
6761            handle_reload_child_registration_failure(
6762                spec,
6763                runtime,
6764                registry,
6765                process_liveness,
6766                snapshot,
6767                child,
6768                ReloadRegistrationFailure {
6769                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6770                        snapshot,
6771                        &timed_out_child,
6772                        &status,
6773                    )),
6774                    reason: format!(
6775                        "new child did not register within {:?}",
6776                        REGISTRY_RELEASE_TIMEOUT
6777                    ),
6778                },
6779            )
6780            .await
6781        }
6782    }
6783}
6784
6785async fn set_child_enabled(
6786    spec: &ModuleSpec,
6787    runtime: &SupervisorRuntimeConfig,
6788    registry: &Registry,
6789    process_liveness: &SupervisorProcessLiveness,
6790    snapshot: &SharedSnapshot,
6791    child: &mut Option<SupervisedChild>,
6792    enabled: bool,
6793) -> Result<bool, SuperviseError> {
6794    let (current_enabled, current_state, respawn_pending) = {
6795        let state = lock_snapshot(snapshot)?;
6796        (state.enabled, state.state, state.respawn_pending)
6797    };
6798    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6799    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6800    // clean (Stopped) has no live process and no other in-band recovery — the
6801    // operator's start is the explicit recovery act and resets the budget. Without
6802    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6803    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6804    // the one providing every agent's shell.
6805    let revive_terminal = enabled
6806        && current_enabled
6807        && child.is_none()
6808        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6809            || (current_state == ModuleState::Restarting && !respawn_pending));
6810    if current_enabled == enabled && !revive_terminal {
6811        return Ok(false);
6812    }
6813
6814    if enabled {
6815        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6816            state.enabled = true;
6817            state.state = ModuleState::Starting;
6818            clear_current_process_facts(state);
6819        })?;
6820        #[cfg(test)]
6821        if runtime.test_seed_stale_facts_before_enable_spawn {
6822            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6823                state.process_alive = true;
6824                state.pid = Some(41);
6825                state.spawned_at_ms = Some(42);
6826                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6827                state.spawned_file_identity = Some(SpawnedFileIdentity {
6828                    device: 43,
6829                    inode: 44,
6830                });
6831            })?;
6832        }
6833        release_dead_registration(
6834            registry,
6835            runtime.forwarding.as_deref(),
6836            snapshot,
6837            &spec.module_id,
6838        )
6839        .await?;
6840        reset_restart_count(snapshot, &spec.module_id)?;
6841        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6842        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6843            Ok(next_child) => next_child,
6844            Err(err) => {
6845                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6846                    state.state = ModuleState::Failed;
6847                    clear_current_process_facts(state);
6848                }) {
6849                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6850                }
6851                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6852                return Err(err);
6853            }
6854        };
6855        *child = Some(next_child);
6856        debug!(module_id = %spec.module_id, "supervised module enabled");
6857        Ok(true)
6858    } else {
6859        let stop_notice = begin_forwarding_drain_if_configured(
6860            spec,
6861            runtime,
6862            registry,
6863            snapshot,
6864            Some(false),
6865            RouteCloseReason::Disable,
6866        )
6867        .await?;
6868        drain_optional_child(
6869            &spec.module_id,
6870            spec.protocol,
6871            stop_notice,
6872            registry,
6873            runtime.forwarding.as_deref(),
6874            snapshot,
6875            &runtime.terminal_ring,
6876            &runtime.spawn_events,
6877            child,
6878            runtime.drain_timeout,
6879            ModuleState::Disabled,
6880            Some(false),
6881        )
6882        .await?;
6883        debug!(module_id = %spec.module_id, "supervised module disabled");
6884        Ok(true)
6885    }
6886}
6887
6888#[allow(clippy::too_many_arguments)]
6889async fn on_child_exit(
6890    spec: &ModuleSpec,
6891    policy: RestartPolicy,
6892    registry: &Registry,
6893    snapshot: &SharedSnapshot,
6894    terminal_ring: &Arc<Mutex<TerminalRing>>,
6895    spawn_events: &SpawnEventFeed,
6896    roster: &ChildRoster,
6897    exit_report: ExitReport,
6898) -> NextAction {
6899    // Once the daemon has begun shutting down, no exit is a crash to recover
6900    // from: the module is exiting because the daemon is going away (EOF on its
6901    // connection, or a service manager signalling the whole cgroup). Record it
6902    // as such and never schedule a respawn, which would only start a process
6903    // for the shutdown to end again.
6904    if roster.is_closed() {
6905        return on_child_exit_during_daemon_shutdown(
6906            spec,
6907            registry,
6908            snapshot,
6909            terminal_ring,
6910            spawn_events,
6911            exit_report,
6912        )
6913        .await;
6914    }
6915    // Every stop the supervisor itself asks for (operator stop, disable,
6916    // restart, reload, swap, a health restart, a drain that runs out of budget)
6917    // takes the child out of the supervise loop and reaps it in
6918    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
6919    // that reaches this point was not requested by the daemon.
6920    //
6921    // For a subc-wire module a clean exit is still a stop: those modules are
6922    // written to re-raise SIGTERM, so a stray outside signal already reads as a
6923    // crash, and exiting 0 is a deliberate choice the module made. A
6924    // `protocol: "none"` module is a stock program we cannot change, and many
6925    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
6926    // stop would leave the module down for good after any stray signal, so it
6927    // goes through the crash path instead: it spends restart budget, respawns
6928    // with the crash backoff, and ends `failed` when the budget runs out.
6929    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6930        && running_protocol(spec, snapshot) == ModuleProtocol::None;
6931    match exit_report.kind {
6932        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6933            info!(
6934                module_id = %spec.module_id,
6935                exit_code = ?exit_report.code,
6936                exit_signal = ?exit_report.signal,
6937                "supervised module exited cleanly"
6938            );
6939            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6940                state.state = ModuleState::Stopped;
6941                clear_current_process_facts(state);
6942                state.last_exit = Some(exit_report.clone());
6943            }) {
6944                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6945            }
6946            record_terminal(
6947                &spec.module_id,
6948                terminal_ring,
6949                spawn_events,
6950                &exit_report,
6951                TerminalDisposition::Stopped,
6952            );
6953            let registration_released = match wait_for_registration_release(
6954                registry,
6955                &spec.module_id,
6956                REGISTRY_RELEASE_TIMEOUT,
6957            )
6958            .await
6959            {
6960                Ok(()) => true,
6961                Err(err) => {
6962                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6963                    false
6964                }
6965            };
6966            NextAction::Stop {
6967                registration_released,
6968            }
6969        }
6970        ExitKind::Clean | ExitKind::Crash => {
6971            if unrequested_clean_exit_of_protocol_none {
6972                warn!(
6973                    module_id = %spec.module_id,
6974                    exit_code = ?exit_report.code,
6975                    exit_signal = ?exit_report.signal,
6976                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
6977                );
6978            } else {
6979                warn!(
6980                    module_id = %spec.module_id,
6981                    exit_code = ?exit_report.code,
6982                    exit_signal = ?exit_report.signal,
6983                    "supervised module exited abnormally (crash)"
6984                );
6985            }
6986            let mut restart_schedule = None;
6987            let mut disposition = TerminalDisposition::Disabled;
6988            // Set only when the budget is what stopped the module, so the
6989            // terminal record says which limit was hit rather than leaving
6990            // `failed` to be read as "crashed once, badly".
6991            let mut disposition_detail = lock_snapshot(snapshot)
6992                .ok()
6993                .and_then(|mut state| state.spawn_failure.take());
6994            let now = Instant::now();
6995            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6996                clear_current_process_facts(state);
6997                state.last_exit = Some(exit_report.clone());
6998                if state.enabled {
6999                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
7000                        state.state = ModuleState::Restarting;
7001                        restart_schedule = Some(schedule);
7002                        disposition = TerminalDisposition::Restarting;
7003                    } else {
7004                        state.state = ModuleState::Failed;
7005                        disposition = TerminalDisposition::Failed;
7006                        let budget = policy.budget_exhausted_detail();
7007                        disposition_detail =
7008                            Some(disposition_detail.take().map_or_else(
7009                                || budget.clone(),
7010                                |cause| format!("{cause}; {budget}"),
7011                            ));
7012                    }
7013                } else {
7014                    state.state = ModuleState::Disabled;
7015                    disposition = TerminalDisposition::Disabled;
7016                }
7017            }) {
7018                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7019                return NextAction::Stop {
7020                    registration_released: false,
7021                };
7022            }
7023            if disposition == TerminalDisposition::Failed {
7024                // The window is in the message, not only in the fields: this line
7025                // is read in a scrollback where a bare `max_restarts=3` reads as a
7026                // lifetime cap and sends the operator looking for three crashes
7027                // that never happened together.
7028                error!(
7029                    module_id = %spec.module_id,
7030                    max_restarts = policy.max_restarts,
7031                    window_secs = policy.window.as_secs(),
7032                    "module stopped: {}",
7033                    policy.budget_exhausted_detail()
7034                );
7035            }
7036            record_terminal_with_detail(
7037                &spec.module_id,
7038                terminal_ring,
7039                spawn_events,
7040                &exit_report,
7041                disposition,
7042                disposition_detail,
7043            );
7044
7045            if let Some(schedule) = restart_schedule {
7046                NextAction::Restart {
7047                    schedule: Some(schedule),
7048                }
7049            } else {
7050                let registration_released = match wait_for_registration_release(
7051                    registry,
7052                    &spec.module_id,
7053                    REGISTRY_RELEASE_TIMEOUT,
7054                )
7055                .await
7056                {
7057                    Ok(()) => true,
7058                    Err(err) => {
7059                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7060                        false
7061                    }
7062                };
7063                NextAction::Stop {
7064                    registration_released,
7065                }
7066            }
7067        }
7068        ExitKind::DeliberateSeverance => {
7069            warn!(
7070                module_id = %spec.module_id,
7071                exit_code = ?exit_report.code,
7072                exit_signal = ?exit_report.signal,
7073                "supervised module exited after deliberate connection severance"
7074            );
7075            let mut should_restart = false;
7076            let mut disposition = TerminalDisposition::Disabled;
7077            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7078                clear_current_process_facts(state);
7079                state.last_exit = Some(exit_report.clone());
7080                state.lifetime_restarts += 1;
7081                if state.enabled {
7082                    state.state = ModuleState::Restarting;
7083                    should_restart = true;
7084                    disposition = TerminalDisposition::Restarting;
7085                } else {
7086                    state.state = ModuleState::Disabled;
7087                }
7088            }) {
7089                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7090                return NextAction::Stop {
7091                    registration_released: false,
7092                };
7093            }
7094            record_terminal(
7095                &spec.module_id,
7096                terminal_ring,
7097                spawn_events,
7098                &exit_report,
7099                disposition,
7100            );
7101
7102            if should_restart {
7103                NextAction::Restart { schedule: None }
7104            } else {
7105                let registration_released = match wait_for_registration_release(
7106                    registry,
7107                    &spec.module_id,
7108                    REGISTRY_RELEASE_TIMEOUT,
7109                )
7110                .await
7111                {
7112                    Ok(()) => true,
7113                    Err(err) => {
7114                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7115                        false
7116                    }
7117                };
7118                NextAction::Stop {
7119                    registration_released,
7120                }
7121            }
7122        }
7123    }
7124}
7125
7126async fn on_child_exit_during_daemon_shutdown(
7127    spec: &ModuleSpec,
7128    registry: &Registry,
7129    snapshot: &SharedSnapshot,
7130    terminal_ring: &Arc<Mutex<TerminalRing>>,
7131    spawn_events: &SpawnEventFeed,
7132    exit_report: ExitReport,
7133) -> NextAction {
7134    info!(
7135        module_id = %spec.module_id,
7136        exit_code = ?exit_report.code,
7137        exit_signal = ?exit_report.signal,
7138        exit_kind = ?exit_report.kind,
7139        "supervised module exited during daemon shutdown; not restarting it"
7140    );
7141    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7142        state.state = ModuleState::Stopped;
7143        clear_current_process_facts(state);
7144        state.last_exit = Some(exit_report.clone());
7145    }) {
7146        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7147    }
7148    record_terminal(
7149        &spec.module_id,
7150        terminal_ring,
7151        spawn_events,
7152        &exit_report,
7153        TerminalDisposition::DaemonShutdown,
7154    );
7155    let registration_released =
7156        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7157            .await
7158            .is_ok();
7159    NextAction::Stop {
7160        registration_released,
7161    }
7162}
7163
7164fn record_wait_error_terminal(
7165    module_id: &str,
7166    terminal_ring: &Arc<Mutex<TerminalRing>>,
7167    spawn_events: &SpawnEventFeed,
7168) {
7169    record_terminal(
7170        module_id,
7171        terminal_ring,
7172        spawn_events,
7173        &wait_error_exit_report(),
7174        TerminalDisposition::Failed,
7175    );
7176}
7177
7178fn record_terminal(
7179    module_id: &str,
7180    terminal_ring: &Arc<Mutex<TerminalRing>>,
7181    spawn_events: &SpawnEventFeed,
7182    exit_report: &ExitReport,
7183    disposition: TerminalDisposition,
7184) {
7185    record_terminal_with_detail(
7186        module_id,
7187        terminal_ring,
7188        spawn_events,
7189        exit_report,
7190        disposition,
7191        None,
7192    );
7193}
7194
7195/// The ring lock is held only to capture the read (see
7196/// `TerminalJournal::capture_read`), so this module's exits keep recording
7197/// while the journal files are read. Blocking: it reads files.
7198fn durable_terminal_history_of(
7199    terminal_ring: &Mutex<TerminalRing>,
7200    module_id: &str,
7201) -> subc_control::TerminalHistory {
7202    let read = terminal_ring
7203        .lock()
7204        .unwrap_or_else(|p| p.into_inner())
7205        .capture_durable_history();
7206    read.read(module_id)
7207}
7208
7209fn record_terminal_with_detail(
7210    module_id: &str,
7211    terminal_ring: &Arc<Mutex<TerminalRing>>,
7212    spawn_events: &SpawnEventFeed,
7213    exit_report: &ExitReport,
7214    disposition: TerminalDisposition,
7215    disposition_detail: Option<String>,
7216) {
7217    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7218    let record = TerminalRecord {
7219        exit_code: exit_report.code,
7220        exit_signal: exit_report.signal,
7221        at_ms: exit_report.at_ms,
7222        disposition,
7223        exit_kind: exit_report.kind.into(),
7224        disposition_detail,
7225    };
7226    terminal_ring
7227        .lock()
7228        .unwrap_or_else(|poisoned| poisoned.into_inner())
7229        .record_exit(module_id, record);
7230}
7231
7232fn untrack_if_registration_released(
7233    process_liveness: &SupervisorProcessLiveness,
7234    registry: &Registry,
7235    module_id: &str,
7236    snapshot: &SharedSnapshot,
7237) {
7238    match registry.get_module(module_id) {
7239        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7240        Ok(Some(_)) => {}
7241        Err(err) => {
7242            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7243        }
7244    }
7245}
7246
7247/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7248/// then apply the module's configured entries minus daemon-private capture keys.
7249///
7250/// Separated from `spawn_child` only so it can be asserted without spawning a
7251/// process — a duplicate of this logic in a test would pass while the real one
7252/// drifted, which is the defect class this function exists to avoid.
7253/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7254/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7255/// either and the argument would stop a stock binary from starting at all.
7256/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7257///
7258/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7259/// through [`apply_wire_spawn_args_for_role`].
7260#[cfg(test)]
7261fn apply_wire_spawn_args(
7262    command: &mut Command,
7263    spec: &ModuleSpec,
7264    connection_file_path: Option<&std::path::Path>,
7265    handle: Option<&SupervisorHandle>,
7266) -> Result<Option<NonceHandoff>, SuperviseError> {
7267    apply_wire_spawn_args_for_role(
7268        command,
7269        spec,
7270        connection_file_path,
7271        handle,
7272        SpawnRole::Plain,
7273    )
7274}
7275
7276/// The read end of a spawn's launch-nonce pipe, prepared by
7277/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7278/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7279/// handoff and keeps only the environment copy.
7280#[cfg(unix)]
7281type NonceHandoff = subc_os::LaunchNonceHandoff;
7282#[cfg(not(unix))]
7283type NonceHandoff = std::convert::Infallible;
7284
7285/// Prepare wire identity for a plain spawn or a swap candidate.
7286///
7287/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7288/// a separate candidate token so the still-serving incumbent and its consumers
7289/// keep their nonce. Both records are installed before the process exists, so
7290/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7291///
7292/// On Unix the nonce is delivered only through a pipe. It is written into
7293/// a pipe whose read end the child gets as descriptor 3, named by
7294/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7295/// process of the same user cannot read it with `ps eww`. That handoff is
7296/// returned rather than installed here, because installing it replaces
7297/// whatever the child has at descriptor 3 and so must be the last pre-exec
7298/// step, after the Linux cgroup placement that the caller registers later.
7299/// Windows retains the environment handoff until restricted handle inheritance
7300/// can be implemented outside std's process primitives.
7301fn apply_wire_spawn_args_for_role(
7302    command: &mut Command,
7303    spec: &ModuleSpec,
7304    connection_file_path: Option<&std::path::Path>,
7305    handle: Option<&SupervisorHandle>,
7306    role: SpawnRole,
7307) -> Result<Option<NonceHandoff>, SuperviseError> {
7308    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7309    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7310    // included: a daemon started from a module's process tree inherits it,
7311    // and passing it on would point the child at a descriptor it does not
7312    // have.
7313    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7314    // Remove inherited or configured copies too: withholding must mean absent.
7315    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7316    if spec.protocol == ModuleProtocol::None {
7317        return Ok(None);
7318    }
7319    if let Some(connection_file_path) = connection_file_path {
7320        command.arg(SUBC_ARG).arg(connection_file_path);
7321    }
7322
7323    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7324    // route.open attestation. Reserved modules additionally use the same nonce
7325    // for HELLO id-squatting protection. A respawn rotates both records.
7326    let nonce = generate_launch_nonce()?;
7327    if let Some(handle) = handle {
7328        match role {
7329            SpawnRole::Plain => {
7330                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7331                if spec.reserved {
7332                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7333                }
7334            }
7335            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7336        }
7337    }
7338    #[cfg(unix)]
7339    let handoff = {
7340        let handoff =
7341            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7342                program: spec.program.clone(),
7343                source,
7344                cgroup_path: None,
7345            })?;
7346        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7347        Some(handoff)
7348    };
7349    #[cfg(not(unix))]
7350    let handoff = None;
7351    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7352    // handle to this child without leaking it to concurrently spawned processes.
7353    #[cfg(not(unix))]
7354    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7355    Ok(handoff)
7356}
7357
7358fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7359    command.env_remove(CK_LOG_ENV);
7360    // The spawn role is the supervisor's to set, and only on a swap candidate
7361    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7362    // it, is what makes it absent on a plain spawn: the daemon's own
7363    // environment could carry it, and so could a spec built outside daemon
7364    // config (config refuses it as an `env` key). A module reading it on a
7365    // plain restart would pick the long swap budget and leave callers waiting.
7366    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7367    for (key, value) in &spec.env {
7368        // cortexkit-log currently exposes retention only as a Rust struct, not
7369        // environment names. These values are daemon-private sink metadata and
7370        // must never become a public child-process contract by being inherited.
7371        if matches!(
7372            key.as_str(),
7373            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7374        ) || key == SUBC_SPAWN_ROLE_ENV
7375        {
7376            continue;
7377        }
7378        command.env(key, value);
7379    }
7380}
7381
7382/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7383/// of a blue/green swap.
7384#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7385enum SpawnRole {
7386    Plain,
7387    SwapCandidate,
7388}
7389
7390/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7391/// `apply_child_env` has already removed the variable for every spawn.
7392fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7393    if role == SpawnRole::SwapCandidate {
7394        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7395    }
7396}
7397
7398fn spawn_child(
7399    spec: &ModuleSpec,
7400    connection_file_path: Option<&std::path::Path>,
7401    handle: Option<&SupervisorHandle>,
7402    ring: &Arc<Mutex<StderrRing>>,
7403    capture_logs_dir: Option<&std::path::Path>,
7404    roster: &ChildRoster,
7405    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7406) -> Result<SupervisedChild, SuperviseError> {
7407    spawn_child_in_slot(
7408        spec,
7409        connection_file_path,
7410        handle,
7411        ring,
7412        capture_logs_dir,
7413        roster,
7414        #[cfg(target_os = "linux")]
7415        cgroup_placement,
7416        SpawnRole::Plain,
7417        false,
7418    )
7419}
7420
7421/// Spawn one process of `spec` into a slot.
7422///
7423/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7424/// A swap candidate needs a different cgroup from the process it is replacing,
7425/// which is still alive: in the same cgroup the two would be one kill domain,
7426/// and killing a failed candidate could take the incumbent with it.
7427///
7428/// The stderr capture file is `<module_id>.stderr.log` for every process of
7429/// the module, whichever slot it is in, because that is the one file
7430/// `ck module logs` reads. During a swap's overlap both processes append to it;
7431/// the daemon writes whole lines, so the two interleave by line, which is also
7432/// the merged view an operator wants while a swap runs.
7433#[allow(clippy::too_many_arguments)]
7434fn spawn_child_in_slot(
7435    spec: &ModuleSpec,
7436    connection_file_path: Option<&std::path::Path>,
7437    handle: Option<&SupervisorHandle>,
7438    ring: &Arc<Mutex<StderrRing>>,
7439    capture_logs_dir: Option<&std::path::Path>,
7440    roster: &ChildRoster,
7441    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7442    role: SpawnRole,
7443    alternate_slot: bool,
7444) -> Result<SupervisedChild, SuperviseError> {
7445    if roster.is_closed() {
7446        return Err(SuperviseError::Spawn {
7447            program: spec.program.clone(),
7448            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7449            cgroup_path: None,
7450        });
7451    }
7452    #[cfg(target_os = "linux")]
7453    let cgroup_name = {
7454        // Slot names alone are not kill domains: a retired incumbent may still
7455        // be draining when a later enable/restart spawns into the same slot.
7456        // Decimal entropy keeps the suffix unambiguous; Placement performs
7457        // the module-id escaping and constructs the filesystem path.
7458        if cgroup_placement.is_none() {
7459            swap::cgroup_name(&spec.module_id, alternate_slot)
7460        } else {
7461            let nonce = generate_launch_nonce()?;
7462            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7463            // Leave room for byte escaping and the suffix under NAME_MAX. The
7464            // label is only for humans; the nonce identifies the kill domain.
7465            let mut end = spec.module_id.len().min(64);
7466            while !spec.module_id.is_char_boundary(end) {
7467                end -= 1;
7468            }
7469            format!(
7470                "{}_{suffix}",
7471                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7472            )
7473        }
7474    };
7475    #[cfg(not(target_os = "linux"))]
7476    let _ = alternate_slot;
7477    #[cfg(target_os = "macos")]
7478    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7479    #[cfg(not(target_os = "macos"))]
7480    let mut command = Command::new(&spec.program);
7481    command.args(&spec.args);
7482    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7483    // that is the whole of the intent, so remove that one key rather than the
7484    // environment.
7485    //
7486    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7487    // and took the POSIX environment with it. Modules spawned that way had no
7488    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7489    // logging:
7490    //
7491    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7492    //     both unset it fell back to the temp dir alone and `ck` could not find
7493    //     a daemon running on the same machine from inside any module's process
7494    //     tree — reporting a path the file has never lived at, which reads as
7495    //     "the daemon did not write its file".
7496    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7497    //     the RELATIVE `.local/share`, so a module deriving its own store path
7498    //     resolved it against its own CWD. That is the store-fragmentation
7499    //     defect the daemon already refuses in config (`parse_doc` rejects a
7500    //     relative `storage.data_home`) arriving by derivation instead.
7501    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7502    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7503    //     quietly rather than erroring.
7504    //
7505    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7506    // offered one candidate under /tmp while the file sat in /run/user/1000.
7507    //
7508    // A configured module is unaffected either way: `module_spec()` puts the
7509    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7510    // wins over anything ambient.
7511    apply_child_env(&mut command, spec);
7512    apply_spawn_role(&mut command, role);
7513    let nonce_handoff =
7514        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7515
7516    #[cfg(target_os = "linux")]
7517    let cgroup_path = cgroup_placement
7518        .map(|placement| placement.module_path(&cgroup_name))
7519        .transpose()
7520        .map_err(|source| SuperviseError::Cgroup {
7521            module_id: spec.module_id.clone(),
7522            source,
7523        })?;
7524    #[cfg(not(target_os = "linux"))]
7525    let cgroup_path: Option<PathBuf> = None;
7526    #[cfg(target_os = "linux")]
7527    if let Some(path) = &cgroup_path {
7528        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7529            if let Some(placement) = cgroup_placement {
7530                remove_module_cgroup(placement, &cgroup_name);
7531            }
7532            return Err(error);
7533        }
7534    }
7535
7536    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7537        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7538        match ChildOutputSink::open(&path, capture_retention(spec)) {
7539            Ok(sink) => sink,
7540            Err(error) => {
7541                warn!(
7542                    module_id = %spec.module_id,
7543                    path = %path.display(),
7544                    error = %error,
7545                    "could not open child output capture file; forwarding to stderr"
7546                );
7547                ChildOutputSink::Stderr
7548            }
7549        }
7550    } else {
7551        ChildOutputSink::Stderr
7552    };
7553
7554    command.stdout(Stdio::piped());
7555    command.stderr(Stdio::piped());
7556    command.kill_on_drop(true);
7557    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7558    // before exec). In the daemon's group, a service manager that kills the
7559    // job's process group when the daemon exits (launchd's default) killed
7560    // every module at the same moment its control connection closed, so no
7561    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7562    // module is reached only by the daemon: the EOF it sees when its
7563    // connection closes, and the bounded stop in `child_roster` for anything
7564    // still running after that. On Linux this composes with the cgroup
7565    // placement above: that is a pre_exec write to cgroup.procs, std performs
7566    // setpgid in the child before running pre_exec callbacks, and the two
7567    // change independent process attributes.
7568    //
7569    // stdin is /dev/null because a process outside the terminal's foreground
7570    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7571    // by hand would otherwise hand down. Under a service manager stdin is
7572    // already /dev/null.
7573    #[cfg(unix)]
7574    command.process_group(0);
7575    command.stdin(Stdio::null());
7576    // The LAST pre-exec step, after the cgroup placement above: installing the
7577    // nonce at descriptor 3 replaces whatever the child had there, which could
7578    // be the descriptor an earlier step writes through.
7579    #[cfg(unix)]
7580    if let Some(handoff) = nonce_handoff {
7581        handoff.install_last(command.as_std_mut());
7582    }
7583    #[cfg(not(unix))]
7584    let _ = nonce_handoff;
7585
7586    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7587    // cannot run a single instruction -- and therefore cannot spawn a
7588    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7589    // other two steps and why the window matters.
7590    #[cfg(windows)]
7591    subc_jobobject::suspend_on_create_async(&mut command);
7592    let mut child = match command.spawn() {
7593        Ok(child) => child,
7594        Err(source) => {
7595            #[cfg(target_os = "linux")]
7596            if let Some(placement) = cgroup_placement {
7597                remove_module_cgroup(placement, &cgroup_name);
7598            }
7599            return Err(SuperviseError::Spawn {
7600                program: spec.program.clone(),
7601                source,
7602                cgroup_path,
7603            });
7604        }
7605    };
7606    // The parent must close its writer now: the acknowledgement pipe reports EOF
7607    // only when every writer is gone, and the child's copy closes when the
7608    // trampoline replaces itself with the module. Command holds only an integer
7609    // in its pre_exec callback, not another writer.
7610    #[cfg(target_os = "macos")]
7611    drop(exec_ack);
7612
7613    // Containment, steps 2 and 3: assign while suspended, then resume.
7614    #[cfg(windows)]
7615    let job = contain_spawned_child(&child, spec)?;
7616    let spawned_at_ms = unix_ms_now();
7617    let spawned_from = spec.program.clone();
7618    let spawned_file_identity = spawned_file_identity(&spawned_from);
7619    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7620        program: spec.program.clone(),
7621        source: io::Error::other("spawned child exposed no live pid"),
7622        cgroup_path: cgroup_path.clone(),
7623    })?;
7624    let process_start_time = crate::provenance::process_start_time(pid);
7625    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7626    #[cfg(all(test, target_os = "macos"))]
7627    privacy_exec_boundary_tests::before_image_sample(spec, pid);
7628    // Unix spawn returns after exec's error pipe closes. The kernel image is
7629    // therefore the executable to compare during a future orphan sweep: PATH
7630    // lookup and shebang interpretation may select a different file from the
7631    // configured program. Keep the literal program's identity for provenance,
7632    // but never use it as proof that a recorded pid may be signalled.
7633    let recorded_image = observe_spawned_image(pid);
7634    // spawn() confirms only the first exec, into the trampoline. Never persist
7635    // the trampoline image; the asynchronous acknowledgement publishes the
7636    // module image once the trampoline has replaced itself with the module.
7637    #[cfg(target_os = "macos")]
7638    let recorded_image = if privacy_exec.is_some() {
7639        None
7640    } else {
7641        recorded_image
7642    };
7643    #[cfg(target_os = "linux")]
7644    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7645    #[cfg(not(target_os = "linux"))]
7646    let recorded_cgroup_name = None;
7647    let roster_guard = roster.admit(
7648        spec.module_id.clone(),
7649        pid,
7650        spec.protocol,
7651        process_start_time,
7652        crate::child_roster::RecordedIdentity {
7653            start_time: recorded_image.map(|image| image.start_time),
7654            executable: recorded_image
7655                .and_then(|image| image.executable)
7656                .map(crate::live_children::ExecutableIdentity::from),
7657            cgroup_name: recorded_cgroup_name,
7658            #[cfg(target_os = "linux")]
7659            cgroup_placement: cgroup_placement.cloned(),
7660        },
7661    );
7662    // The check at the top of this function can pass just before daemon
7663    // shutdown begins, and the process is only in the roster from here on.
7664    // The shutdown stop returns as soon as it finds the roster empty, so a
7665    // process admitted after that look would outlive the daemon. The roster
7666    // is closed before the stop first reads it and admission happens under
7667    // the roster's lock, so either the stop sees this process or this check
7668    // sees the roster closed: end the process now rather than start a module
7669    // the daemon is about to stop.
7670    if roster.is_closed() {
7671        // This child was never admitted, so there is no module protocol shutdown to wait for.
7672        #[cfg(target_os = "linux")]
7673        kill_module_cgroup(cgroup_placement, &cgroup_name);
7674        if let Err(error) = child.start_kill() {
7675            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7676        }
7677        #[cfg(target_os = "linux")]
7678        if let Some(placement) = cgroup_placement {
7679            // This spawn was never admitted, so shutdown has no roster entry
7680            // to await. Do not detach its cleanup: the runtime could exit
7681            // before that task reaps the rejected child and removes its group.
7682            while matches!(child.try_wait(), Ok(None)) {
7683                std::thread::yield_now();
7684            }
7685            if matches!(
7686                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7687                subc_cgroup::KillOutcome::Killed
7688            ) {
7689                if let Ok(path) = placement.module_path(&cgroup_name) {
7690                    while std::fs::read_to_string(path.join("cgroup.events"))
7691                        .ok()
7692                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7693                    {
7694                        std::thread::yield_now();
7695                    }
7696                }
7697            }
7698            remove_module_cgroup(placement, &cgroup_name);
7699        }
7700        drop(roster_guard);
7701        return Err(SuperviseError::Spawn {
7702            program: spec.program.clone(),
7703            source: io::Error::other(
7704                "the daemon began shutting down while this process was starting; ended it",
7705            ),
7706            cgroup_path,
7707        });
7708    }
7709
7710    let stdout_pump = match child.stdout.take() {
7711        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7712        None => {
7713            warn!(
7714                module_id = %spec.module_id,
7715                "spawned child exposed no stdout pipe; file capture will be incomplete"
7716            );
7717            None
7718        }
7719    };
7720    let stderr_pump = match child.stderr.take() {
7721        Some(stderr) => {
7722            let generation = ring
7723                .lock()
7724                .unwrap_or_else(|poisoned| poisoned.into_inner())
7725                .begin_process();
7726            Some(StderrPump {
7727                task: tokio::spawn(pump_stderr_to(
7728                    stderr,
7729                    Arc::clone(ring),
7730                    generation,
7731                    output_sink,
7732                )),
7733                generation,
7734            })
7735        }
7736        None => {
7737            // Spawning succeeded but the pipe did not materialise. Recording it as
7738            // uncaptured keeps the tail honest: the alternative is an empty tail
7739            // that reads as a module which printed nothing.
7740            ring.lock()
7741                .unwrap_or_else(|poisoned| poisoned.into_inner())
7742                .mark_not_captured("stderr pipe was not available on spawn");
7743            warn!(
7744                module_id = %spec.module_id,
7745                "spawned child exposed no stderr pipe; tail will be unavailable"
7746            );
7747            None
7748        }
7749    };
7750
7751    Ok(SupervisedChild {
7752        child,
7753        protocol: spec.protocol,
7754        #[cfg(target_os = "linux")]
7755        module_id: cgroup_name,
7756        #[cfg(target_os = "linux")]
7757        cgroup_placement: cgroup_placement.cloned(),
7758        #[cfg(windows)]
7759        job,
7760        stdout_pump,
7761        stderr_pump,
7762        stderr_ring: Arc::clone(ring),
7763        spawned_at_ms,
7764        spawned_from,
7765        spawned_file_identity,
7766        process_start_time,
7767        process_identity,
7768        pid,
7769        roster_guard: Some(roster_guard),
7770        #[cfg(target_os = "macos")]
7771        privacy_exec,
7772        #[cfg(target_os = "macos")]
7773        report_ready: Arc::new(OnceLock::new()),
7774        spawn_failure: None,
7775    })
7776}
7777
7778#[cfg(target_os = "linux")]
7779pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7780    use subc_cgroup::KillOutcome;
7781    match subc_cgroup::kill_module(placement, module_id) {
7782        KillOutcome::Killed => {}
7783        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7784            debug!(
7785                module_id,
7786                "cgroup tree kill unavailable; using direct-child kill"
7787            );
7788        }
7789        KillOutcome::IoError { path, error } => {
7790            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7791        }
7792    }
7793}
7794
7795/// Contain a freshly spawned Windows child and start it.
7796///
7797/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7798/// child assigned **while it is still suspended** (step 1 is
7799/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7800///
7801/// A child that is never resumed hangs forever holding a pid, so a resume
7802/// failure kills the child and fails the spawn rather than returning a
7803/// `SupervisedChild` that can never run.
7804///
7805/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7806/// it did before this existed, whereas refusing to start one would be a new
7807/// outage. It is logged at warn because it means a helper process could leak.
7808#[cfg(windows)]
7809fn contain_spawned_child(
7810    child: &Child,
7811    spec: &ModuleSpec,
7812) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7813    let module_id = spec.module_id.as_str();
7814    let Some(pid) = child.id() else {
7815        // The child exited between spawn and here. Its tree, if it made one,
7816        // needs no containment: nothing is left to contain.
7817        warn!(
7818            module_id,
7819            "spawned child had already exited before containment; no job object attached"
7820        );
7821        return Ok(None);
7822    };
7823
7824    let job = match subc_jobobject::JobObject::new() {
7825        Ok(job) => job,
7826        Err(source) => {
7827            warn!(
7828                module_id,
7829                error = %source,
7830                "could not create a job object; this module's helper processes will not be \
7831                 reaped on teardown"
7832            );
7833            // Resume regardless: leaving the child suspended would turn a
7834            // containment gap into a hung module.
7835            resume_suspended_child(pid, spec)?;
7836            return Ok(None);
7837        }
7838    };
7839
7840    if let Err(source) = job.assign(child) {
7841        warn!(
7842            module_id,
7843            error = %source,
7844            "could not assign the child to its job object; this module's helper processes \
7845             will not be reaped on teardown"
7846        );
7847        resume_suspended_child(pid, spec)?;
7848        return Ok(None);
7849    }
7850
7851    resume_suspended_child(pid, spec)?;
7852    Ok(Some(job))
7853}
7854
7855/// Resume a suspended child, killing it if it cannot be started.
7856///
7857/// A suspended process holds a pid and does nothing, so there is no useful
7858/// state to return: the caller gets an error and the spawn fails.
7859#[cfg(windows)]
7860fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7861    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7862        // Kill it here rather than leaving a suspended process for the caller
7863        // to notice; `kill_on_drop` would eventually do this, but the module
7864        // would have been reported as running in between.
7865        let _ = std::process::Command::new("taskkill.exe")
7866            .args(["/PID", &pid.to_string(), "/T", "/F"])
7867            .stdin(Stdio::null())
7868            .stdout(Stdio::null())
7869            .stderr(Stdio::null())
7870            .status();
7871        return Err(SuperviseError::Spawn {
7872            program: spec.program.clone(),
7873            source,
7874            cgroup_path: None,
7875        });
7876    }
7877    Ok(())
7878}
7879
7880#[cfg(target_os = "linux")]
7881fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7882    match placement.remove_module(module_id) {
7883        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7884        Err(error) => warn!(
7885            module_id,
7886            error = %error,
7887            "could not remove module cgroup after process exit; continuing teardown"
7888        ),
7889    }
7890}
7891
7892#[cfg(target_os = "linux")]
7893async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7894    // Reaping the direct child is not proof its descendants exited. End the
7895    // residual tree and wait for the kernel's population fact before rmdir;
7896    // otherwise a successful parent wait leaks a directory on each restart.
7897    if matches!(
7898        subc_cgroup::kill_module(Some(placement), module_id),
7899        subc_cgroup::KillOutcome::Killed
7900    ) {
7901        if let Ok(path) = placement.module_path(module_id) {
7902            while std::fs::read_to_string(path.join("cgroup.events"))
7903                .ok()
7904                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7905            {
7906                sleep(Duration::from_millis(1)).await;
7907            }
7908        }
7909    }
7910    remove_module_cgroup(placement, module_id);
7911}
7912
7913#[cfg(target_os = "linux")]
7914fn apply_cgroup_placement(
7915    command: &mut Command,
7916    spec: &ModuleSpec,
7917    path: &std::path::Path,
7918) -> Result<(), SuperviseError> {
7919    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7920        module_id: spec.module_id.clone(),
7921        source,
7922    })
7923}
7924
7925fn capture_retention(spec: &ModuleSpec) -> Retention {
7926    let defaults = Retention::default();
7927    let value = |name: &str| {
7928        spec.env
7929            .iter()
7930            .rev()
7931            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7932    };
7933    Retention {
7934        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7935            .and_then(|value| value.parse().ok())
7936            .unwrap_or(defaults.max_file_mb),
7937        keep: value(CAPTURE_KEEP_ENV)
7938            .and_then(|value| value.parse().ok())
7939            .unwrap_or(defaults.keep),
7940        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7941            .and_then(|value| value.parse().ok())
7942            .unwrap_or(defaults.max_age_days),
7943    }
7944}
7945
7946/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
7947/// module's registration to the exact process the supervisor spawned.
7948fn generate_launch_nonce() -> Result<String, SuperviseError> {
7949    let mut bytes = [0u8; 32];
7950    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7951        reason: source.to_string(),
7952    })?;
7953    let mut hex = String::with_capacity(64);
7954    for b in bytes {
7955        use std::fmt::Write;
7956        let _ = write!(hex, "{b:02x}");
7957    }
7958    Ok(hex)
7959}
7960
7961/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
7962/// signal about how many leading bytes matched.
7963fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7964    if a.len() != b.len() {
7965        return false;
7966    }
7967    let mut diff = 0u8;
7968    for (x, y) in a.iter().zip(b.iter()) {
7969        diff |= x ^ y;
7970    }
7971    diff == 0
7972}
7973
7974/// The kernel's image after an acknowledged exec, shared by ordinary launches
7975/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
7976/// not identities inferred from a configured pathname.
7977fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
7978    subc_os::Process::open(pid)
7979        .ok()
7980        .flatten()
7981        .and_then(|process| process.observe())
7982}
7983
7984fn spawn_and_mark_running(
7985    spec: &ModuleSpec,
7986    runtime: &SupervisorRuntimeConfig,
7987    snapshot: &SharedSnapshot,
7988) -> Result<SupervisedChild, SuperviseError> {
7989    let child = spawn_child(
7990        spec,
7991        runtime.connection_file_path.as_deref(),
7992        runtime.supervisor_handle.as_ref(),
7993        &runtime.stderr_ring,
7994        runtime.capture_logs_dir.as_deref(),
7995        &runtime.child_roster,
7996        #[cfg(target_os = "linux")]
7997        runtime.cgroup_placement.as_ref(),
7998    )?;
7999    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
8000    Ok(child)
8001}
8002
8003enum RegistrationWaitOutcome {
8004    Registered,
8005    Exited(ExitReport),
8006    TimedOut,
8007}
8008
8009struct ReloadRegistrationFailure {
8010    exit_report: ExitReport,
8011    reason: String,
8012}
8013
8014#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8015enum BusyGaugeObservation {
8016    Quiescent,
8017    Busy,
8018    Omitted,
8019}
8020
8021fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8022    let Some(metrics) = metrics.and_then(Value::as_object) else {
8023        return BusyGaugeObservation::Omitted;
8024    };
8025    let mut sum = 0u128;
8026    for gauge in gauges {
8027        let Some(value) = metrics.get(gauge) else {
8028            return BusyGaugeObservation::Omitted;
8029        };
8030        let Some(value) = value.as_u64() else {
8031            return BusyGaugeObservation::Busy;
8032        };
8033        sum = sum.saturating_add(u128::from(value));
8034    }
8035    if sum == 0 {
8036        BusyGaugeObservation::Quiescent
8037    } else {
8038        BusyGaugeObservation::Busy
8039    }
8040}
8041
8042fn declared_busy_gauges(
8043    registry: &Registry,
8044    module_id: &str,
8045) -> Result<Vec<String>, SuperviseError> {
8046    busy_gauges_of(
8047        registry
8048            .get_module(module_id)
8049            .map_err(SuperviseError::Registry)?,
8050    )
8051}
8052
8053/// [`declared_busy_gauges`] for the registration a connection holds, in any
8054/// slot: after cutover the incumbent is no longer the id's active
8055/// registration, and its own manifest is the one that names its gauges.
8056fn declared_busy_gauges_for_connection(
8057    registry: &Registry,
8058    connection_id: ConnectionId,
8059) -> Result<Vec<String>, SuperviseError> {
8060    busy_gauges_of(
8061        registry
8062            .get_module_by_connection(connection_id)
8063            .map_err(SuperviseError::Registry)?,
8064    )
8065}
8066
8067fn busy_gauges_of(
8068    registration: Option<crate::registry::ModuleRegistration>,
8069) -> Result<Vec<String>, SuperviseError> {
8070    let Some(registration) = registration else {
8071        return Ok(Vec::new());
8072    };
8073    let Some(self_signals) = registration.manifest.self_signals else {
8074        return Ok(Vec::new());
8075    };
8076
8077    let mut gauges = Vec::new();
8078    for declaration in self_signals {
8079        if declaration.kind != SelfSignalKind::Busy {
8080            continue;
8081        }
8082        match declaration.anchored_to {
8083            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8084                gauges.extend(declared)
8085            }
8086            _ => {
8087                // An invalid Busy anchor is fail-safe: the empty name cannot be
8088                // present in a conforming health report, so this drain stays busy.
8089                gauges.push(String::new());
8090            }
8091        }
8092    }
8093    Ok(gauges)
8094}
8095
8096/// Wait for `endpoint` to have nothing in flight and, when the module declares
8097/// busy gauges, for a health probe to report them quiet. The probe is addressed
8098/// by `scope`: a swap's superseded incumbent must be asked about its own
8099/// gauges, and by module id the probe would reach the promoted candidate.
8100async fn wait_for_forwarding_quiescence(
8101    forwarding: &ForwardingTable,
8102    module_id: &str,
8103    runtime: &SupervisorRuntimeConfig,
8104    endpoint: crate::ModuleEndpointId,
8105    deadline: Instant,
8106    busy_gauges: &[String],
8107    scope: DrainScope,
8108) -> Result<bool, SuperviseError> {
8109    let mut gauges_quiescent = busy_gauges.is_empty();
8110    let mut next_probe_at = Instant::now();
8111    let mut omission_counted = false;
8112
8113    loop {
8114        let now = Instant::now();
8115        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8116            let report = match scope {
8117                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8118                DrainScope::Endpoint(endpoint) => {
8119                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8120                }
8121            };
8122            gauges_quiescent = match report {
8123                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8124                    BusyGaugeObservation::Quiescent => true,
8125                    BusyGaugeObservation::Busy => false,
8126                    BusyGaugeObservation::Omitted => {
8127                        if !omission_counted {
8128                            forwarding
8129                                .counters()
8130                                .increment_drains_with_undeclared_gauge();
8131                            omission_counted = true;
8132                        }
8133                        false
8134                    }
8135                },
8136                Err(err) => {
8137                    warn!(
8138                        module_id,
8139                        error = %err,
8140                        "drain health.check did not produce declared busy gauges; treating module as busy"
8141                    );
8142                    false
8143                }
8144            };
8145            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8146        }
8147
8148        let in_flight = forwarding
8149            .endpoint_in_flight_count(endpoint)
8150            .map_err(SuperviseError::Forwarding)?;
8151        if in_flight == 0 && gauges_quiescent {
8152            return Ok(true);
8153        }
8154
8155        let now = Instant::now();
8156        if now >= deadline {
8157            return Ok(false);
8158        }
8159        let mut wait = deadline
8160            .saturating_duration_since(now)
8161            .min(REGISTRY_RELEASE_POLL);
8162        if !busy_gauges.is_empty() {
8163            wait = wait.min(next_probe_at.saturating_duration_since(now));
8164        }
8165        sleep(wait).await;
8166    }
8167}
8168
8169/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8170///
8171/// `Ok` is always honest and passed straight through -- the wait actually measured
8172/// in-flight state. `Err` means the wait produced no measurement at all (the
8173/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8174/// constant: the drain did not complete. Never recomputed from route state, never a
8175/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8176fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8177    match wait_result {
8178        Ok(drained) => *drained,
8179        Err(_) => false,
8180    }
8181}
8182
8183fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8184    for released in released_routes {
8185        let frame = match Frame::build_with_version(
8186            released.negotiated_ver,
8187            FrameType::Goodbye,
8188            control_flags(),
8189            released.channel,
8190            released.epoch,
8191            0,
8192            Vec::new(),
8193        ) {
8194            Ok(frame) => frame,
8195            Err(err) => {
8196                warn!(
8197                    route_channel = released.channel,
8198                    error = %err,
8199                    "failed to build supervisor drain route GOODBYE frame"
8200                );
8201                continue;
8202            }
8203        };
8204        if !released.close_on_delivery_failure() {
8205            crate::forwarding::send_module_route_goodbye(
8206                &forwarding.counters(),
8207                &released.sink,
8208                frame,
8209                released.module_id.as_deref(),
8210                "supervisor drain",
8211            );
8212            continue;
8213        }
8214        if let Err(err) = released.sink.try_send(frame) {
8215            warn!(
8216                target_connection_id = released.connection_id.get(),
8217                route_channel = released.channel,
8218                error = %err,
8219                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8220            );
8221            let _ = forwarding.escalate_client_delivery_failure(
8222                released.connection_id,
8223                released.channel,
8224                released.epoch,
8225                CloseReason::new(
8226                    "route_goodbye_delivery_failed",
8227                    format!(
8228                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8229                        released.channel
8230                    ),
8231                ),
8232                crate::forwarding::UndeliveredFrame {
8233                    module_id: released.module_id.as_deref(),
8234                    sink: &released.sink,
8235                },
8236            );
8237        }
8238    }
8239}
8240
8241fn send_module_draining(
8242    module_id: &str,
8243    reason: RouteCloseReason,
8244    deadline_ms: u64,
8245    target: &ModuleDrainTarget,
8246) {
8247    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8248        reason,
8249        deadline_ms,
8250    }) {
8251        Ok(body) => body,
8252        Err(err) => {
8253            warn!(
8254                module_id,
8255                error = %err,
8256                "failed to encode module draining command"
8257            );
8258            return;
8259        }
8260    };
8261    let frame = match Frame::build_with_version(
8262        target.negotiated_ver,
8263        FrameType::Push,
8264        control_flags(),
8265        0,
8266        0,
8267        0,
8268        body,
8269    ) {
8270        Ok(frame) => frame,
8271        Err(err) => {
8272            warn!(
8273                module_id,
8274                error = %err,
8275                "failed to build module draining command frame"
8276            );
8277            return;
8278        }
8279    };
8280    if let Err(err) = target.sink.try_send(frame) {
8281        warn!(
8282            module_id,
8283            target_connection_id = target.endpoint.connection_id.get(),
8284            error = %err,
8285            "module draining command was not delivered to peer"
8286        );
8287    }
8288}
8289
8290/// The channel-0 GOODBYE that tells a module its stop is planned.
8291fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8292    match Frame::build_with_version(
8293        negotiated_ver,
8294        FrameType::Goodbye,
8295        control_flags(),
8296        0,
8297        0,
8298        0,
8299        Vec::new(),
8300    ) {
8301        Ok(frame) => Some(frame),
8302        Err(err) => {
8303            warn!(
8304                module_id,
8305                error = %err,
8306                "failed to build module GOODBYE frame"
8307            );
8308            None
8309        }
8310    }
8311}
8312
8313/// Send every registered module connection its module GOODBYE at daemon
8314/// shutdown, then request that connection's close.
8315///
8316/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8317/// before EOF, so the GOODBYE must reach the socket before the close. A close
8318/// request does not wait for the connection's queued frames: its writer gets a
8319/// bounded grace after the close, is aborted if it overruns it, and the daemon
8320/// process may exit before that grace ends. So with `wait_for_flush`, each
8321/// connection is closed only after its writer has acknowledged writing the
8322/// GOODBYE, or once a short shared budget runs out, so one module that is not
8323/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8324/// are only queued, for a shutdown the operator has told to stop waiting.
8325/// A connection that is already gone is skipped.
8326#[cfg(unix)]
8327async fn send_module_goodbyes_for_daemon_shutdown(
8328    forwarding: &Arc<ForwardingTable>,
8329    reason: &CloseReason,
8330    wait_for_flush: bool,
8331) {
8332    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8333    let targets = match forwarding.module_connections() {
8334        Ok(targets) => targets,
8335        Err(err) => {
8336            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8337            return;
8338        }
8339    };
8340    let deadline = Instant::now() + GOODBYE_BUDGET;
8341    let mut sends = tokio::task::JoinSet::new();
8342    for target in targets {
8343        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8344            continue;
8345        };
8346        if !wait_for_flush {
8347            if let Err(err) = target.sink.try_send(frame) {
8348                debug!(
8349                    module_id = %target.module_id,
8350                    error = %err,
8351                    "shutdown module GOODBYE was not queued"
8352                );
8353            }
8354            continue;
8355        }
8356        let forwarding = Arc::clone(forwarding);
8357        let reason = reason.clone();
8358        sends.spawn(async move {
8359            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8360                Ok(Ok(())) => {}
8361                Ok(Err(err)) => debug!(
8362                    module_id = %target.module_id,
8363                    error = %err,
8364                    "module connection closed before its shutdown GOODBYE was written"
8365                ),
8366                Err(_) => warn!(
8367                    module_id = %target.module_id,
8368                    budget = ?GOODBYE_BUDGET,
8369                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8370                ),
8371            }
8372            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8373        });
8374    }
8375    // Every task ends by the shared deadline, so this wait is bounded too.
8376    while sends.join_next().await.is_some() {}
8377}
8378
8379fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8380    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8381        return;
8382    };
8383    if let Err(err) = target.sink.try_send(frame) {
8384        warn!(
8385            module_id,
8386            target_connection_id = target.endpoint.connection_id.get(),
8387            error = %err,
8388            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8389        );
8390        forwarding.request_connection_close(
8391            target.endpoint.connection_id,
8392            CloseReason::new(
8393                "module_goodbye_delivery_failed",
8394                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8395            ),
8396        );
8397    }
8398}
8399
8400#[derive(Clone, Copy)]
8401struct ForwardingDrainContext<'a> {
8402    spec: &'a ModuleSpec,
8403    runtime: &'a SupervisorRuntimeConfig,
8404    registry: &'a Registry,
8405    scope: DrainScope,
8406}
8407
8408/// Which process a forwarding drain addresses.
8409#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8410enum DrainScope {
8411    /// Whatever endpoint is active for the module id: every plain stop,
8412    /// restart and reload. Also moves the module's state to `Draining`.
8413    Active,
8414    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8415    /// module id would resolve to the promoted candidate and leave neither
8416    /// process routable. The module's state is left alone, since the promoted
8417    /// candidate is what it describes and that process is running.
8418    Endpoint(crate::ModuleEndpointId),
8419}
8420
8421/// Whether a child being drained has already been asked to stop by the time
8422/// its drain wait starts.
8423///
8424/// The drain wait is the same budget whatever this says. What it decides is
8425/// whether the supervisor must ask by signal before that wait begins: a child
8426/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8427/// healthy or not.
8428#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8429enum StopNotice {
8430    /// The module was sent `module.draining` and a module GOODBYE over its own
8431    /// registered connection, and stops itself.
8432    SentOverConnection,
8433    /// The forwarding drain found no registered connection for the module: a
8434    /// subc child spawned moments ago that has not sent HELLO yet, or a
8435    /// `protocol: "none"` child, which never registers.
8436    NoConnection,
8437    /// This path sends nothing over the module's connection: the supervisor has
8438    /// no forwarding table, or the caller stops the child without a forwarding
8439    /// drain.
8440    NotSent,
8441}
8442
8443async fn begin_forwarding_drain(
8444    spec: &ModuleSpec,
8445    runtime: &SupervisorRuntimeConfig,
8446    registry: &Registry,
8447    snapshot: &SharedSnapshot,
8448    enabled: Option<bool>,
8449    reason: RouteCloseReason,
8450) -> Result<StopNotice, SuperviseError> {
8451    let Some(forwarding) = runtime.forwarding.as_ref() else {
8452        return Err(SuperviseError::ReloadUnavailable {
8453            module_id: spec.module_id.clone(),
8454            reason: "supervisor was not configured with a forwarding table".to_string(),
8455        });
8456    };
8457
8458    begin_forwarding_drain_with(
8459        forwarding,
8460        ForwardingDrainContext {
8461            spec,
8462            runtime,
8463            registry,
8464            scope: DrainScope::Active,
8465        },
8466        snapshot,
8467        enabled,
8468        reason,
8469        runtime.drain_timeout,
8470    )
8471    .await
8472}
8473
8474async fn begin_forwarding_drain_if_configured(
8475    spec: &ModuleSpec,
8476    runtime: &SupervisorRuntimeConfig,
8477    registry: &Registry,
8478    snapshot: &SharedSnapshot,
8479    enabled: Option<bool>,
8480    reason: RouteCloseReason,
8481) -> Result<StopNotice, SuperviseError> {
8482    begin_forwarding_drain_with_timeout(
8483        spec,
8484        runtime,
8485        registry,
8486        snapshot,
8487        enabled,
8488        reason,
8489        runtime.drain_timeout,
8490    )
8491    .await
8492}
8493
8494/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8495/// budget, for paths where the operator overrides the module's configured one
8496/// (`supervisor.restart{drain_timeout_ms}`).
8497async fn begin_forwarding_drain_with_timeout(
8498    spec: &ModuleSpec,
8499    runtime: &SupervisorRuntimeConfig,
8500    registry: &Registry,
8501    snapshot: &SharedSnapshot,
8502    enabled: Option<bool>,
8503    reason: RouteCloseReason,
8504    drain_timeout: Duration,
8505) -> Result<StopNotice, SuperviseError> {
8506    let Some(forwarding) = runtime.forwarding.as_ref() else {
8507        return Ok(StopNotice::NotSent);
8508    };
8509
8510    begin_forwarding_drain_with(
8511        forwarding,
8512        ForwardingDrainContext {
8513            spec,
8514            runtime,
8515            registry,
8516            scope: DrainScope::Active,
8517        },
8518        snapshot,
8519        enabled,
8520        reason,
8521        drain_timeout,
8522    )
8523    .await
8524}
8525
8526async fn begin_forwarding_drain_with(
8527    forwarding: &ForwardingTable,
8528    context: ForwardingDrainContext<'_>,
8529    snapshot: &SharedSnapshot,
8530    enabled: Option<bool>,
8531    reason: RouteCloseReason,
8532    drain_timeout: Duration,
8533) -> Result<StopNotice, SuperviseError> {
8534    let ForwardingDrainContext {
8535        spec,
8536        runtime,
8537        registry,
8538        scope,
8539    } = context;
8540    debug_assert_ne!(reason, RouteCloseReason::Crash);
8541    let terminal = matches!(reason, RouteCloseReason::Disable);
8542    let drain_started_at = Instant::now();
8543    let drain_deadline = drain_started_at + drain_timeout;
8544    let deadline_ms =
8545        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8546    let busy_gauges = match scope {
8547        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8548        DrainScope::Endpoint(endpoint) => {
8549            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8550        }
8551    };
8552
8553    // Admission gate first: route.open/commit and route REQUEST admission are closed
8554    // before the first quiescence check, so the outstanding count can only fall.
8555    let gate_started = Instant::now();
8556    let drain_target = match scope {
8557        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8558        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8559    }
8560    .map_err(SuperviseError::Forwarding)?;
8561    // The instant admission closed, and how long taking the forwarding write
8562    // lock to close it took. The timeout line reports only the quiescence
8563    // wait, so without this a drain that started late looked like one that
8564    // started on time.
8565    info!(
8566        module_id = %spec.module_id,
8567        ?reason,
8568        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8569        connected = drain_target.is_some(),
8570        "module drain began; route admission closed"
8571    );
8572    if scope == DrainScope::Active {
8573        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8574            state.state = ModuleState::Draining;
8575            state.draining_to_replace =
8576                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8577            if let Some(enabled) = enabled {
8578                state.enabled = enabled;
8579            }
8580        })?;
8581    }
8582
8583    let Some(target) = drain_target.as_ref() else {
8584        // Nothing was sent: the module has no registered connection to carry
8585        // `module.draining` or a GOODBYE. The caller must not assume the child
8586        // was asked to stop.
8587        return Ok(StopNotice::NoConnection);
8588    };
8589    {
8590        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8591        let routes = forwarding
8592            .endpoint_routes(target.endpoint)
8593            .map_err(SuperviseError::Forwarding)?;
8594        let routes_notified = routes.len();
8595        crate::control::send_route_control_pushes(
8596            forwarding,
8597            routes.clone(),
8598            ClientControlPush::RouteClosing {
8599                module_id: spec.module_id.clone(),
8600                channels: Vec::new(),
8601                reason,
8602            },
8603        );
8604        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8605
8606        // `route.closing` was just sent above: from here on every return path,
8607        // including an early one, MUST send `route.closed` before propagating
8608        // anything else. A client holds `closing` as a promise that a verdict is
8609        // coming; leaving early without `closed` strands it waiting forever, since
8610        // `closing` carries no timeout of its own.
8611        let wait_result = wait_for_forwarding_quiescence(
8612            forwarding,
8613            &spec.module_id,
8614            runtime,
8615            target.endpoint,
8616            drain_deadline,
8617            &busy_gauges,
8618            scope,
8619        )
8620        .await;
8621        let drained = drained_after_quiescence_wait(&wait_result);
8622        if let Err(err) = &wait_result {
8623            error!(
8624                module_id = %spec.module_id,
8625                ?reason,
8626                error = %err,
8627                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8628            );
8629        } else if !drained {
8630            // Name what the drain waited on. Without it the line says only that
8631            // something did not settle, and "one wedged call" and "every
8632            // session's held stream" read the same; the first is a module bug,
8633            // the second is a module that should end its streams on
8634            // module.draining. Read before teardown releases the routes.
8635            let holdouts = forwarding
8636                .endpoint_drain_holdouts(target.endpoint)
8637                .unwrap_or_default();
8638            warn!(
8639                module_id = %spec.module_id,
8640                waited = ?drain_timeout,
8641                ?reason,
8642                held_requests = holdouts.requests,
8643                held_routes = holdouts.routes,
8644                total_routes = holdouts.total_routes,
8645                top_connections = ?holdouts.top_connections,
8646                // `module_channel:corr`, so the module can find each held request
8647                // in its own log; capped, so `held_requests` is the full count.
8648                held = %holdouts
8649                    .held
8650                    .iter()
8651                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8652                    .collect::<Vec<_>>()
8653                    .join(","),
8654                "route drain timed out before request quiescence; forcing teardown"
8655            );
8656        }
8657        crate::control::send_route_control_pushes(
8658            forwarding,
8659            routes,
8660            ClientControlPush::RouteClosed {
8661                module_id: spec.module_id.clone(),
8662                channels: Vec::new(),
8663                reason,
8664                drained,
8665                abandoned: target.abandoned_bindings.len() as u32,
8666                excluded_subscriptions: target.excluded_subscriptions,
8667                terminal: Some(terminal),
8668            },
8669        );
8670        wait_result?;
8671
8672        // `route.closed` has now been sent unconditionally above. From here the
8673        // remaining steps are cleanup (route + module GOODBYE) rather than a
8674        // promise the client is waiting on, but a lock-poisoned
8675        // `release_module_endpoint_routes` would otherwise skip the module
8676        // GOODBYE silently too -- send it before propagating the error.
8677        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8678            Ok(routes) => routes,
8679            Err(err) => {
8680                warn!(
8681                    module_id = %spec.module_id,
8682                    ?reason,
8683                    error = %err,
8684                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8685                );
8686                send_module_goodbye(&spec.module_id, forwarding, target);
8687                return Err(SuperviseError::Forwarding(err));
8688            }
8689        };
8690        let route_goodbye_count = released_routes.len();
8691        send_route_goodbyes(forwarding, released_routes);
8692        send_module_goodbye(&spec.module_id, forwarding, target);
8693
8694        // The drain's happy path was previously silent: every emission above is
8695        // best-effort with only its failure arm logged, so "were consumers told"
8696        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8697        // hang where the open question was exactly whether teardown notice went
8698        // out). One summary line makes that class decidable in one grep.
8699        info!(
8700            module_id = %spec.module_id,
8701            ?reason,
8702            routes_notified,
8703            route_goodbyes = route_goodbye_count,
8704            abandoned_reservations = target.abandoned_bindings.len(),
8705            excluded_subscriptions = target.excluded_subscriptions,
8706            drained,
8707            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8708        );
8709    }
8710
8711    Ok(StopNotice::SentOverConnection)
8712}
8713
8714/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8715/// the only slot a plain (non-swap) spawn can register into.
8716async fn wait_for_registration_after_reload(
8717    registry: &Registry,
8718    module_id: &str,
8719    snapshot: &SharedSnapshot,
8720    child: &mut SupervisedChild,
8721    wait: Duration,
8722) -> Result<RegistrationWaitOutcome, SuperviseError> {
8723    wait_for_slot_registration(
8724        registry,
8725        crate::registry::RegistrationSlot::Active(module_id),
8726        module_id,
8727        snapshot,
8728        child,
8729        wait,
8730    )
8731    .await
8732}
8733
8734/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8735///
8736/// Keyed on the slot rather than the bare module id because during a swap the
8737/// id's active slot is already held by the incumbent: an id-keyed wait would
8738/// report the incumbent's registration as the candidate's and a candidate that
8739/// never registers would look registered. A swap candidate waits on
8740/// `crate::registry::RegistrationSlot::Candidate`.
8741async fn wait_for_slot_registration(
8742    registry: &Registry,
8743    slot: crate::registry::RegistrationSlot<'_>,
8744    module_id: &str,
8745    snapshot: &SharedSnapshot,
8746    child: &mut SupervisedChild,
8747    wait: Duration,
8748) -> Result<RegistrationWaitOutcome, SuperviseError> {
8749    let deadline = Instant::now() + wait;
8750    loop {
8751        if registry
8752            .registration(slot)
8753            .map_err(SuperviseError::Registry)?
8754            .is_some()
8755        {
8756            return Ok(RegistrationWaitOutcome::Registered);
8757        }
8758
8759        let now = Instant::now();
8760        if now >= deadline {
8761            return Ok(RegistrationWaitOutcome::TimedOut);
8762        }
8763        let remaining = deadline.saturating_duration_since(now);
8764        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8765
8766        tokio::select! {
8767            wait_result = child.wait() => {
8768                let status = wait_result.map_err(|source| SuperviseError::Wait {
8769                    module_id: module_id.to_string(),
8770                    source,
8771                })?;
8772                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8773                    snapshot,
8774                    child,
8775                    &status,
8776                )));
8777            }
8778            _ = sleep(poll) => {}
8779        }
8780    }
8781}
8782
8783fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8784    // A replacement process that exits before HELLO did not provide service, even
8785    // if it used status 0. Count it against the restart cap as a new-binary failure.
8786    if exit_report.kind != ExitKind::DeliberateSeverance {
8787        exit_report.kind = ExitKind::Crash;
8788    }
8789    exit_report
8790}
8791
8792async fn handle_reload_child_registration_failure(
8793    spec: &ModuleSpec,
8794    runtime: &SupervisorRuntimeConfig,
8795    registry: &Registry,
8796    process_liveness: &SupervisorProcessLiveness,
8797    snapshot: &SharedSnapshot,
8798    _child: &mut Option<SupervisedChild>,
8799    failure: ReloadRegistrationFailure,
8800) -> Result<(), SuperviseError> {
8801    let ReloadRegistrationFailure {
8802        exit_report,
8803        reason,
8804    } = failure;
8805    match on_child_exit(
8806        spec,
8807        runtime.restart_policy,
8808        registry,
8809        snapshot,
8810        &runtime.terminal_ring,
8811        &runtime.spawn_events,
8812        &runtime.child_roster,
8813        exit_report,
8814    )
8815    .await
8816    {
8817        NextAction::Stop {
8818            registration_released,
8819        } => {
8820            if registration_released {
8821                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8822            }
8823        }
8824        NextAction::Restart { schedule } => {
8825            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8826                schedule.delay
8827            });
8828            if let Some(schedule) = schedule {
8829                log_crash_respawn(&spec.module_id, schedule);
8830            }
8831            schedule_respawn(
8832                runtime,
8833                snapshot,
8834                &spec.module_id,
8835                delay,
8836                RespawnKind::Spawn,
8837            )?;
8838        }
8839    }
8840    Err(SuperviseError::ReloadFailed {
8841        module_id: spec.module_id.clone(),
8842        reason,
8843    })
8844}
8845
8846async fn handle_reload_spawn_failure(
8847    spec: &ModuleSpec,
8848    runtime: &SupervisorRuntimeConfig,
8849    process_liveness: &SupervisorProcessLiveness,
8850    snapshot: &SharedSnapshot,
8851    _child: &mut Option<SupervisedChild>,
8852    reason: String,
8853) -> Result<(), SuperviseError> {
8854    let now = Instant::now();
8855    let mut schedule = None;
8856    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8857        clear_current_process_facts(state);
8858        if state.enabled {
8859            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8860            state.state = if schedule.is_some() {
8861                ModuleState::Restarting
8862            } else {
8863                ModuleState::Failed
8864            };
8865        } else {
8866            state.state = ModuleState::Disabled;
8867        }
8868    })?;
8869    if let Some(schedule) = schedule {
8870        schedule_respawn(
8871            runtime,
8872            snapshot,
8873            &spec.module_id,
8874            schedule.delay,
8875            RespawnKind::Spawn,
8876        )?;
8877    } else {
8878        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8879    }
8880    Err(SuperviseError::ReloadFailed {
8881        module_id: spec.module_id.clone(),
8882        reason,
8883    })
8884}
8885
8886fn control_flags() -> Flags {
8887    Flags::new(false, Priority::Passive, false)
8888}
8889
8890#[allow(clippy::too_many_arguments)]
8891async fn drain_optional_child(
8892    module_id: &str,
8893    protocol: ModuleProtocol,
8894    stop_notice: StopNotice,
8895    registry: &Registry,
8896    forwarding: Option<&ForwardingTable>,
8897    snapshot: &SharedSnapshot,
8898    terminal_ring: &Arc<Mutex<TerminalRing>>,
8899    spawn_events: &SpawnEventFeed,
8900    child: &mut Option<SupervisedChild>,
8901    drain_timeout: Duration,
8902    final_state: ModuleState,
8903    enabled: Option<bool>,
8904) -> Result<(), SuperviseError> {
8905    if let Some(child) = child.take() {
8906        drain_child_to_state(
8907            module_id,
8908            protocol,
8909            stop_notice,
8910            registry,
8911            forwarding,
8912            snapshot,
8913            terminal_ring,
8914            spawn_events,
8915            child,
8916            drain_timeout,
8917            final_state,
8918            enabled,
8919        )
8920        .await
8921    } else {
8922        update_snapshot(snapshot, Some(module_id), |state| {
8923            state.state = final_state;
8924            if let Some(enabled) = enabled {
8925                state.enabled = enabled;
8926            }
8927            clear_current_process_facts(state);
8928        })?;
8929        release_dead_registration(registry, forwarding, snapshot, module_id).await
8930    }
8931}
8932
8933#[allow(clippy::too_many_arguments)]
8934async fn drain_child_to_state(
8935    module_id: &str,
8936    _protocol: ModuleProtocol,
8937    stop_notice: StopNotice,
8938    registry: &Registry,
8939    forwarding: Option<&ForwardingTable>,
8940    snapshot: &SharedSnapshot,
8941    terminal_ring: &Arc<Mutex<TerminalRing>>,
8942    spawn_events: &SpawnEventFeed,
8943    mut child: SupervisedChild,
8944    drain_timeout: Duration,
8945    final_state: ModuleState,
8946    enabled: Option<bool>,
8947) -> Result<(), SuperviseError> {
8948    let protocol = child.protocol;
8949    update_snapshot(snapshot, Some(module_id), |state| {
8950        state.state = ModuleState::Draining;
8951        state.draining_to_replace = final_state == ModuleState::Restarting;
8952        if let Some(enabled) = enabled {
8953            state.enabled = enabled;
8954        }
8955    })?;
8956
8957    // The wait below is the same budget in every case; what differs is
8958    // whether anything has ASKED the child to stop before it starts. Only a
8959    // forwarding drain that reached the module's registered connection has
8960    // (`module.draining`, then a module GOODBYE). Every other child was told
8961    // nothing: a `protocol: "none"` module, which never registers; a subc
8962    // module spawned moments ago that has not sent HELLO yet; or a stop that
8963    // runs no forwarding drain. Without a signal the budget is only a delay
8964    // in front of SIGKILL -- and the not-yet-registered child is the worst
8965    // case, because it registers into a module that is already draining,
8966    // is never told, and is killed while healthy.
8967    if stop_notice != StopNotice::SentOverConnection {
8968        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8969            info!(
8970                module_id,
8971                pid = child.pid,
8972                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8973                "module has no connection yet; requesting stop by signal"
8974            );
8975        }
8976        request_graceful_stop(module_id, &child);
8977    }
8978
8979    let exit_report = match timeout(drain_timeout, child.wait()).await {
8980        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
8981        Ok(Err(source)) => {
8982            fail_snapshot(snapshot, Some(module_id), None);
8983            return Err(SuperviseError::Wait {
8984                module_id: module_id.to_string(),
8985                source,
8986            });
8987        }
8988        Err(_) => {
8989            // Mirror the sibling arm above: state is already `Draining`, and an
8990            // error propagated from here would strand it there -- a state
8991            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
8992            // `Failed | Stopped`), leaving an operator Restart as the only exit.
8993            // `Failed` before `?` keeps the module operator-visible and
8994            // revivable. Trigger is an ESRCH race (process exits between the
8995            // drain timeout firing and the kill) or a post-kill wait failure
8996            // (issue #34).
8997            //
8998            // Logged because the kill is otherwise visible only as signal 9 in
8999            // the terminal ring, and the budget it follows can be long enough
9000            // that consumers see a stretch of refusals with no stated cause.
9001            warn!(
9002                module_id,
9003                pid = child.pid,
9004                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9005                reason = ?final_state,
9006                ?stop_notice,
9007                "drain budget expired before the module exited; killing it"
9008            );
9009            child.start_kill().map_err(|source| {
9010                fail_snapshot(snapshot, Some(module_id), None);
9011                SuperviseError::Kill {
9012                    module_id: module_id.to_string(),
9013                    source,
9014                }
9015            })?;
9016            let status = child.wait().await.map_err(|source| {
9017                fail_snapshot(snapshot, Some(module_id), None);
9018                SuperviseError::Wait {
9019                    module_id: module_id.to_string(),
9020                    source,
9021                }
9022            })?;
9023            classify_reaped_child_exit(snapshot, &child, &status)
9024        }
9025    };
9026
9027    update_snapshot(snapshot, Some(module_id), |state| {
9028        state.state = final_state;
9029        if let Some(enabled) = enabled {
9030            state.enabled = enabled;
9031        }
9032        clear_current_process_facts(state);
9033        state.last_exit = Some(exit_report.clone());
9034        if exit_report.kind == ExitKind::DeliberateSeverance {
9035            state.lifetime_restarts += 1;
9036        }
9037    })?;
9038    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9039    record_terminal_with_detail(
9040        module_id,
9041        terminal_ring,
9042        spawn_events,
9043        &exit_report,
9044        terminal_disposition(final_state),
9045        detail,
9046    );
9047    child.drain_stderr(module_id).await;
9048
9049    release_dead_registration(registry, forwarding, snapshot, module_id).await
9050}
9051
9052/// Ask a child that nothing else has asked to stop, by signal.
9053///
9054/// A registered subc module is asked over its own connection: the drain sends
9055/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
9056/// module GOODBYE, and the module stops itself. A module that speaks no subc
9057/// wire receives none of that, and neither does a subc module that has not
9058/// registered yet, so for them the drain budget would be pure delay in front of
9059/// a SIGKILL -- and for a process with a store to flush (JetStream is the
9060/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
9061/// into a recovery on the next start.
9062///
9063/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
9064/// rule rather than an optimisation: that module's graceful stop is already
9065/// running by the time its child is drained, and a signal would race it.
9066///
9067/// Best-effort by construction. A child that has already exited is the ordinary
9068/// case rather than an error (the kill lands on a reaped or exiting pid), so a
9069/// failure is logged at debug and the wait-then-kill below still decides the
9070/// outcome.
9071#[cfg(unix)]
9072fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9073    let Some(pid) = child
9074        .id()
9075        .and_then(|pid| i32::try_from(pid).ok())
9076        .and_then(rustix::process::Pid::from_raw)
9077    else {
9078        debug!(
9079            module_id,
9080            "no pid to signal for teardown; falling through to the drain wait"
9081        );
9082        return;
9083    };
9084    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9085        Ok(()) => debug!(
9086            module_id,
9087            "sent SIGTERM to a module nothing else asked to stop"
9088        ),
9089        Err(err) => debug!(
9090            module_id,
9091            error = %err,
9092            "SIGTERM to module failed; the drain wait and kill still apply"
9093        ),
9094    }
9095}
9096
9097/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9098/// Windows does offer need cooperation this supervisor cannot assume: a console
9099/// control event requires sharing a console with the child, and `WM_CLOSE`
9100/// requires the child to pump a message loop. A supervised server process does
9101/// neither, so there is nothing to send and teardown is the wait followed by the
9102/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9103/// the thing `protocol: "none"` exists to avoid.
9104#[cfg(not(unix))]
9105fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9106    debug!(
9107        module_id,
9108        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9109    );
9110}
9111
9112fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9113    match final_state {
9114        ModuleState::Stopped => TerminalDisposition::Stopped,
9115        ModuleState::Disabled => TerminalDisposition::Disabled,
9116        ModuleState::Restarting => TerminalDisposition::Restarting,
9117        ModuleState::Failed => TerminalDisposition::Failed,
9118        ModuleState::Starting
9119        | ModuleState::Running
9120        | ModuleState::Unresponsive
9121        | ModuleState::Draining => {
9122            unreachable!("terminal exits only finish in terminal or restarting states")
9123        }
9124    }
9125}
9126
9127/// Release a reaped child's registration before allowing another spawn.
9128///
9129/// EOF is not a process-lifetime signal: an inherited socket can stay open
9130/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9131/// reading EOF. After the normal release grace, request connection close (which
9132/// cancels both reads and dispatch), then allow one more release grace for the
9133/// connection guard's forwarding cleanup. Never evict a different connection.
9134async fn release_dead_registration(
9135    registry: &Registry,
9136    forwarding: Option<&ForwardingTable>,
9137    snapshot: &SharedSnapshot,
9138    module_id: &str,
9139) -> Result<(), SuperviseError> {
9140    let result = async {
9141        let registration = registry
9142            .get_module(module_id)
9143            .map_err(SuperviseError::Registry)?;
9144        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9145            Ok(()) => return Ok(()),
9146            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9147            Err(err) => return Err(err),
9148        }
9149        let pid = lock_snapshot(snapshot)?.reaped_pid;
9150        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9151            warn!(
9152                module_id,
9153                pid,
9154                connection_id = registration.connection_id.get(),
9155                "reaped module registration outlived release grace; closing dead connection"
9156            );
9157            forwarding.request_connection_close(
9158                registration.connection_id,
9159                CloseReason::new(
9160                    "supervised_process_reaped",
9161                    format!("module '{module_id}' pid {pid} exited"),
9162                ),
9163            );
9164            wait_for_slot_registration_release(
9165                registry,
9166                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9167                REGISTRY_RELEASE_TIMEOUT,
9168            )
9169            .await?;
9170        }
9171        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9172    }
9173    .await;
9174    if let Err(err) = &result {
9175        fail_snapshot(snapshot, Some(module_id), None);
9176        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9177    }
9178    result
9179}
9180
9181/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9182/// plain stop or restart waits for before it spawns a replacement.
9183async fn wait_for_registration_release(
9184    registry: &Registry,
9185    module_id: &str,
9186    wait: Duration,
9187) -> Result<(), SuperviseError> {
9188    wait_for_slot_registration_release(
9189        registry,
9190        crate::registry::RegistrationSlot::Active(module_id),
9191        wait,
9192    )
9193    .await
9194}
9195
9196/// Wait for the registration in `slot` to go away.
9197///
9198/// Keyed on the slot rather than the bare module id because a successful swap
9199/// never empties the id's active slot (the promoted candidate is in it), so an
9200/// id-keyed wait for the incumbent's release would always time out. Draining a
9201/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9202/// incumbent's connection instead.
9203async fn wait_for_slot_registration_release(
9204    registry: &Registry,
9205    slot: crate::registry::RegistrationSlot<'_>,
9206    wait: Duration,
9207) -> Result<(), SuperviseError> {
9208    let deadline = Instant::now() + wait;
9209    let mut release_events = registration_release_events().subscribe();
9210    let still_active = |registration: &crate::registry::ModuleRegistration| {
9211        SuperviseError::RegistrationStillActive {
9212            module_id: registration.manifest.module_id.clone(),
9213            waited: wait,
9214        }
9215    };
9216    loop {
9217        let _observed_generation = *release_events.borrow_and_update();
9218        let Some(registration) = registry
9219            .registration(slot)
9220            .map_err(SuperviseError::Registry)?
9221        else {
9222            return Ok(());
9223        };
9224
9225        let now = Instant::now();
9226        if now >= deadline {
9227            return Err(still_active(&registration));
9228        }
9229
9230        let remaining = deadline.saturating_duration_since(now);
9231        match timeout(remaining, release_events.changed()).await {
9232            Ok(Ok(())) | Ok(Err(_)) => {}
9233            Err(_) => return Err(still_active(&registration)),
9234        }
9235    }
9236}
9237
9238#[cfg(test)]
9239mod slot_registration_wait_tests {
9240    use super::*;
9241    use crate::registry::{ConnectionId, RegistrationSlot};
9242    use subc_protocol::manifest::ModuleManifest;
9243
9244    #[tokio::test]
9245    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9246        let registry = Arc::new(Registry::default());
9247        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9248        let runtime = supervisor.runtime_config();
9249        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9250        let spec = ModuleSpec {
9251            module_id: "enable-stale-registration".to_string(),
9252            program: PathBuf::from("/missing/enable-retry-test"),
9253            args: Vec::new(),
9254            env: Vec::new(),
9255            reserved: false,
9256            reserved_prefixes: Vec::new(),
9257            protocol: ModuleProtocol::Subc,
9258            overlap: Default::default(),
9259        };
9260        let connection = ConnectionId::new(90);
9261        registry
9262            .register_with_control_ops(
9263                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9264                1,
9265                connection,
9266                Vec::new(),
9267            )
9268            .unwrap();
9269        let mut child = None;
9270        let err = set_child_enabled(
9271            &spec,
9272            &runtime,
9273            &registry,
9274            &supervisor.process_liveness,
9275            &snapshot,
9276            &mut child,
9277            true,
9278        )
9279        .await
9280        .unwrap_err();
9281        assert!(matches!(
9282            err,
9283            SuperviseError::RegistrationStillActive { .. }
9284        ));
9285        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9286        assert!(child.is_none());
9287        registry.deregister_connection(connection).unwrap();
9288        let err = set_child_enabled(
9289            &spec,
9290            &runtime,
9291            &registry,
9292            &supervisor.process_liveness,
9293            &snapshot,
9294            &mut child,
9295            true,
9296        )
9297        .await
9298        .unwrap_err();
9299        assert!(
9300            matches!(err, SuperviseError::Spawn { .. }),
9301            "second enable must attempt a spawn: {err}"
9302        );
9303        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9304    }
9305
9306    const INCUMBENT: u64 = 1;
9307    const CANDIDATE: u64 = 2;
9308
9309    fn swapped_registry() -> Arc<Registry> {
9310        let registry = Arc::new(Registry::default());
9311        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9312        registry
9313            .register_with_control_ops(
9314                manifest.clone(),
9315                1,
9316                ConnectionId::new(INCUMBENT),
9317                Vec::new(),
9318            )
9319            .unwrap();
9320        registry
9321            .register_candidate_with_control_ops(
9322                manifest,
9323                1,
9324                ConnectionId::new(CANDIDATE),
9325                Vec::new(),
9326            )
9327            .unwrap();
9328        registry
9329    }
9330
9331    /// After a promotion the id's active slot is held by the new process, so an
9332    /// id-keyed wait for the incumbent's release can never succeed; the
9333    /// connection-keyed wait completes as soon as the incumbent deregisters.
9334    #[tokio::test]
9335    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9336        let registry = swapped_registry();
9337        registry.promote_candidate("m").unwrap().unwrap();
9338
9339        assert!(matches!(
9340            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9341            Err(SuperviseError::RegistrationStillActive { .. })
9342        ));
9343
9344        // Still held while the incumbent's connection has not deregistered.
9345        assert!(matches!(
9346            wait_for_slot_registration_release(
9347                &registry,
9348                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9349                Duration::from_millis(50),
9350            )
9351            .await,
9352            Err(SuperviseError::RegistrationStillActive { .. })
9353        ));
9354
9355        let releaser = Arc::clone(&registry);
9356        let release = tokio::spawn(async move {
9357            sleep(Duration::from_millis(20)).await;
9358            releaser
9359                .deregister_connection(ConnectionId::new(INCUMBENT))
9360                .unwrap();
9361            notify_registration_release();
9362        });
9363        wait_for_slot_registration_release(
9364            &registry,
9365            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9366            Duration::from_secs(5),
9367        )
9368        .await
9369        .expect("the incumbent's own registration is released");
9370        release.await.unwrap();
9371        assert!(registry.get_module("m").unwrap().is_some());
9372    }
9373
9374    /// The candidate slot is waited on separately from the active slot: the
9375    /// incumbent's registration neither holds up nor stands in for it.
9376    #[tokio::test]
9377    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9378        let registry = swapped_registry();
9379        assert!(matches!(
9380            wait_for_slot_registration_release(
9381                &registry,
9382                RegistrationSlot::Candidate("m"),
9383                Duration::from_millis(50),
9384            )
9385            .await,
9386            Err(SuperviseError::RegistrationStillActive { .. })
9387        ));
9388        registry
9389            .deregister_connection(ConnectionId::new(CANDIDATE))
9390            .unwrap();
9391        wait_for_slot_registration_release(
9392            &registry,
9393            RegistrationSlot::Candidate("m"),
9394            Duration::from_millis(50),
9395        )
9396        .await
9397        .expect("a candidate slot with no candidate is released");
9398        assert!(registry
9399            .registration(RegistrationSlot::Active("m"))
9400            .unwrap()
9401            .is_some());
9402    }
9403}
9404
9405fn classify_exit(status: &ExitStatus) -> ExitReport {
9406    ExitReport {
9407        kind: if status.success() {
9408            ExitKind::Clean
9409        } else {
9410            ExitKind::Crash
9411        },
9412        code: status.code(),
9413        signal: exit_signal(status),
9414        at_ms: unix_ms_now(),
9415    }
9416}
9417
9418/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9419/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9420/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9421/// disposition still must be `Failed` so the terminal ring is not silently missing
9422/// an entry, matching what `fail_snapshot` records for this same arm.
9423fn wait_error_exit_report() -> ExitReport {
9424    ExitReport {
9425        kind: ExitKind::Crash,
9426        code: None,
9427        signal: None,
9428        at_ms: unix_ms_now(),
9429    }
9430}
9431
9432#[cfg(unix)]
9433fn exit_signal(status: &ExitStatus) -> Option<i32> {
9434    use std::os::unix::process::ExitStatusExt;
9435
9436    status.signal()
9437}
9438
9439#[cfg(not(unix))]
9440fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9441    None
9442}
9443
9444/// Give an operator-touched module its full crash budget back.
9445///
9446/// Named for the counter it used to zero; it now empties the in-window ring,
9447/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9448/// ledger of what happened survives every operator action.
9449fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9450    update_snapshot(snapshot, Some(module_id), |state| {
9451        state.clear_crash_restarts();
9452    })
9453}
9454
9455fn set_running(
9456    snapshot: &SharedSnapshot,
9457    child: &SupervisedChild,
9458    module_id: &str,
9459    spawn_events: &SpawnEventFeed,
9460) -> Result<(), SuperviseError> {
9461    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9462        module_id: Some(module_id.to_string()),
9463    })?;
9464    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9465    if std::mem::take(&mut state.coalesced_restart_pending) {
9466        let generation = state.spawn_generation;
9467        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9468    }
9469    state.drain_disposition_detail = None;
9470    state.spawn_failure = None;
9471    // Every caller of this is a plain spawn, which always uses the primary key;
9472    // a promoted swap candidate sets the flag itself after this returns.
9473    state.in_alternate_slot = false;
9474    state.configuration_updated_since_spawn = false;
9475    state.spawned_protocol = Some(child.protocol);
9476    state.state = ModuleState::Running;
9477    state.enabled = true;
9478    state.process_alive = true;
9479    state.pid = child.id();
9480    #[cfg(target_os = "macos")]
9481    {
9482        state.report_ready = Some(Arc::clone(&child.report_ready));
9483    }
9484    state.spawned_at_ms = Some(child.spawned_at_ms);
9485    state.spawned_from = Some(child.spawned_from.clone());
9486    state.spawned_file_identity = child.spawned_file_identity;
9487    state.process_start_time = child.process_start_time;
9488    Ok(())
9489}
9490
9491fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9492    state.process_alive = false;
9493    state.spawned_protocol = None;
9494    state.pid = None;
9495    #[cfg(target_os = "macos")]
9496    {
9497        state.report_ready = None;
9498    }
9499    state.spawned_at_ms = None;
9500    state.spawned_from = None;
9501    state.spawned_file_identity = None;
9502    state.process_start_time = None;
9503    state.deliberate_severance = None;
9504}
9505
9506#[cfg(test)]
9507fn record_deliberate_severance(
9508    snapshot: &SharedSnapshot,
9509    identity: ProcessIdentity,
9510) -> Result<(), SuperviseError> {
9511    update_snapshot(snapshot, None, |state| {
9512        state.deliberate_severance = Some(identity);
9513    })
9514}
9515
9516fn apply_deliberate_severance_marker(
9517    snapshot: &SharedSnapshot,
9518    exited_identity: Option<ProcessIdentity>,
9519    mut exit_report: ExitReport,
9520) -> ExitReport {
9521    let marker = lock_snapshot(snapshot)
9522        .ok()
9523        .and_then(|mut state| state.deliberate_severance.take());
9524    if marker.is_some() && marker == exited_identity {
9525        exit_report.kind = ExitKind::DeliberateSeverance;
9526    }
9527    exit_report
9528}
9529
9530fn classify_reaped_child_exit(
9531    snapshot: &SharedSnapshot,
9532    child: &SupervisedChild,
9533    status: &ExitStatus,
9534) -> ExitReport {
9535    let _ = update_snapshot(snapshot, None, |state| {
9536        state.reaped_pid = Some(child.pid);
9537        state.spawn_failure = child.spawn_failure.clone();
9538    });
9539    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9540}
9541
9542fn fail_snapshot(
9543    snapshot: &SharedSnapshot,
9544    module_id: Option<&str>,
9545    last_exit: Option<ExitReport>,
9546) {
9547    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9548        state.state = ModuleState::Failed;
9549        clear_current_process_facts(state);
9550        if let Some(last_exit) = last_exit {
9551            state.last_exit = Some(last_exit);
9552        }
9553    }) {
9554        error!(error = %err, "failed to mark supervisor state failed");
9555    }
9556}
9557
9558fn update_snapshot(
9559    snapshot: &SharedSnapshot,
9560    module_id: Option<&str>,
9561    update: impl FnOnce(&mut SupervisorSnapshot),
9562) -> Result<(), SuperviseError> {
9563    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9564        module_id: module_id.map(ToOwned::to_owned),
9565    })?;
9566    update(&mut state);
9567    Ok(())
9568}
9569
9570const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9571
9572fn lock_snapshot_for_control<'a>(
9573    snapshot: &'a SharedSnapshot,
9574    module_id: &str,
9575    caller: &'static str,
9576) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9577    let started_at = Instant::now();
9578    let guard = lock_snapshot(snapshot)?;
9579    let waited = started_at.elapsed();
9580    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9581        warn!(
9582            module_id = %module_id,
9583            waited_ms = waited.as_millis() as u64,
9584            caller = %caller,
9585            "slow snapshot lock"
9586        );
9587    }
9588    Ok(guard)
9589}
9590
9591fn lock_snapshot(
9592    snapshot: &SharedSnapshot,
9593) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9594    snapshot
9595        .lock()
9596        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9597}
9598
9599#[cfg(test)]
9600mod terminal_history_tests {
9601    use std::{
9602        path::PathBuf,
9603        sync::Arc,
9604        time::{Duration, Instant},
9605    };
9606
9607    use tokio::time::sleep;
9608
9609    use super::{
9610        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9611        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9612        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9613        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9614        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9615        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
9616        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
9617    };
9618    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9619    // use for their own wall-clock deadlines: crash-restart instants must be on
9620    // the same clock the production code stamps them with, which is tokio's (and
9621    // is what `start_paused` tests can move).
9622    use super::Instant as ClockInstant;
9623    use crate::{
9624        registry::Registry,
9625        terminal_ring::{TerminalRing, TerminalRingConfig},
9626    };
9627    use std::sync::Mutex;
9628    use subc_control::TerminalDisposition;
9629
9630    /// See the twin in `control.rs` for why this derives the path from
9631    /// `current_exe()` and why the existence check is here: `--lib` alone does
9632    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9633    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9634    pub(super) fn fake_aft_stub_path() -> PathBuf {
9635        let mut path = std::env::current_exe().expect("current_exe available in tests");
9636        path.pop();
9637        path.pop();
9638        path.push(if cfg!(windows) {
9639            "fake-aft-stub.exe"
9640        } else {
9641            "fake-aft-stub"
9642        });
9643        assert!(
9644            path.exists(),
9645            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9646             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9647            path.display()
9648        );
9649        path
9650    }
9651
9652    #[test]
9653    fn reserved_never_spawned_refuses_every_hello() {
9654        // The canary hole: a reserved id whose module has never spawned had NO
9655        // gate entry and admitted anyone -- the reservation protected the nonce
9656        // holder, not the NAME. Now the entry is present with no legitimate
9657        // holder and refuses all comers.
9658        let supervisor = SupervisorHandle::default();
9659        supervisor.apply_identity_configuration(&ModuleSpec {
9660            module_id: "never-spawned".to_string(),
9661            program: PathBuf::from("/usr/bin/false"),
9662            args: Vec::new(),
9663            env: Vec::new(),
9664            reserved: true,
9665            reserved_prefixes: Vec::new(),
9666            protocol: ModuleProtocol::Subc,
9667            overlap: Default::default(),
9668        });
9669        assert!(
9670            supervisor
9671                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9672                .is_some(),
9673            "forged nonce must refuse on a reserved never-spawned id"
9674        );
9675        assert!(
9676            supervisor
9677                .reserved_hello_rejection("never-spawned", None)
9678                .is_some(),
9679            "absent nonce must refuse on a reserved never-spawned id"
9680        );
9681        // And a real spawn nonce minted later admits exactly that nonce.
9682        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9683        supervisor.apply_identity_configuration(&ModuleSpec {
9684            module_id: "never-spawned".to_string(),
9685            program: PathBuf::from("/usr/bin/false"),
9686            args: Vec::new(),
9687            env: Vec::new(),
9688            reserved: true,
9689            reserved_prefixes: Vec::new(),
9690            protocol: ModuleProtocol::Subc,
9691            overlap: Default::default(),
9692        });
9693        assert!(supervisor
9694            .reserved_hello_rejection("never-spawned", Some("minted"))
9695            .is_none());
9696        assert!(supervisor
9697            .reserved_hello_rejection("never-spawned", Some("forged"))
9698            .is_some());
9699    }
9700
9701    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9702    /// happened, which is what "spent budget" looks like to every reader.
9703    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9704        let now = ClockInstant::now();
9705        for _ in 0..count {
9706            state.crash_restarts.push_back(now);
9707        }
9708    }
9709
9710    /// Age the oldest recorded restart out of `window`, standing in for the hours
9711    /// that would otherwise have to pass. Injecting the instant is the point: a
9712    /// test that slept a real window would take ten minutes and still prove less.
9713    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9714        let aged = state
9715            .crash_restarts
9716            .front()
9717            .expect("a crash restart must be recorded before it can be aged")
9718            .checked_sub(window + Duration::from_secs(1))
9719            .expect("the test clock is far enough from its origin to age an instant");
9720        state.crash_restarts[0] = aged;
9721    }
9722
9723    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9724        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9725        seed_crash_restarts(&mut state, count);
9726        state
9727    }
9728
9729    #[test]
9730    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9731        let policy = RestartPolicy::new(3, Duration::ZERO);
9732        let now = ClockInstant::now();
9733        assert!(daemon_will_restart(
9734            &mut snapshot_with_restarts(true, 2),
9735            &policy,
9736            now
9737        ));
9738        assert!(!daemon_will_restart(
9739            &mut snapshot_with_restarts(true, 3),
9740            &policy,
9741            now
9742        ));
9743        assert!(!daemon_will_restart(
9744            &mut snapshot_with_restarts(false, 0),
9745            &policy,
9746            now
9747        ));
9748    }
9749
9750    #[test]
9751    fn crash_restart_backoff_escalates_with_in_window_count() {
9752        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9753            .with_max_backoff(Duration::from_secs(30));
9754        let now = ClockInstant::now();
9755        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9756        let schedules = (0..4)
9757            .map(|_| {
9758                state
9759                    .next_crash_restart(&policy, now)
9760                    .expect("the test policy allows four crash restarts")
9761            })
9762            .collect::<Vec<_>>();
9763
9764        assert_eq!(
9765            schedules
9766                .iter()
9767                .map(|schedule| schedule.restart_in_window)
9768                .collect::<Vec<_>>(),
9769            vec![0, 1, 2, 3]
9770        );
9771        assert_eq!(
9772            schedules
9773                .iter()
9774                .map(|schedule| schedule.delay)
9775                .collect::<Vec<_>>(),
9776            vec![
9777                Duration::from_millis(100),
9778                Duration::from_secs(1),
9779                Duration::from_secs(10),
9780                Duration::from_secs(30),
9781            ]
9782        );
9783    }
9784
9785    #[test]
9786    fn crash_restart_backoff_resets_after_ring_clear() {
9787        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9788        let now = ClockInstant::now();
9789        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9790        assert_eq!(
9791            state.next_crash_restart(&policy, now).unwrap().delay,
9792            Duration::from_millis(100)
9793        );
9794        assert_eq!(
9795            state.next_crash_restart(&policy, now).unwrap().delay,
9796            Duration::from_secs(1)
9797        );
9798
9799        state.clear_crash_restarts();
9800        let schedule = state
9801            .next_crash_restart(&policy, now)
9802            .expect("a cleared ring must allow another restart");
9803        assert_eq!(schedule.restart_in_window, 0);
9804        assert_eq!(schedule.delay, Duration::from_millis(100));
9805    }
9806
9807    #[test]
9808    fn crash_restart_backoff_ignores_aged_restarts() {
9809        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9810        let now = ClockInstant::now();
9811        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9812        state
9813            .next_crash_restart(&policy, now)
9814            .expect("the first restart is allowed");
9815        state
9816            .next_crash_restart(&policy, now)
9817            .expect("the second restart is allowed");
9818        state.crash_restarts[0] = now
9819            .checked_sub(policy.window + Duration::from_secs(1))
9820            .expect("the fake clock can age a restart past the window");
9821
9822        let schedule = state
9823            .next_crash_restart(&policy, now)
9824            .expect("an aged restart must release its slot");
9825        assert_eq!(schedule.restart_in_window, 1);
9826        assert_eq!(schedule.delay, Duration::from_secs(1));
9827        assert_eq!(state.crash_restarts.len(), 2);
9828    }
9829
9830    /// The budget is a rate: the same three spent restarts refuse a respawn
9831    /// while they are recent and allow one once they have aged past the window.
9832    /// Nothing about the module changed in between, which is the whole point.
9833    #[test]
9834    fn a_budget_spent_before_the_window_no_longer_refuses() {
9835        let policy = RestartPolicy::new(3, Duration::ZERO);
9836        let mut state = snapshot_with_restarts(true, 3);
9837        let now = ClockInstant::now();
9838        assert!(!daemon_will_restart(&mut state, &policy, now));
9839
9840        assert!(daemon_will_restart(
9841            &mut state,
9842            &policy,
9843            now + policy.window + Duration::from_secs(1)
9844        ));
9845        assert!(
9846            state.crash_restarts.is_empty(),
9847            "reading the budget must drop the instants that left the window"
9848        );
9849    }
9850
9851    fn module_with_recovery_snapshot(
9852        state: ModuleState,
9853        enabled: bool,
9854        restart_count: u32,
9855    ) -> SupervisedModule {
9856        let registry = Arc::new(Registry::default());
9857        let supervisor =
9858            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9859        let module = supervisor
9860            .spawn(ModuleSpec {
9861                module_id: "recovery-snapshot".to_string(),
9862                program: fake_aft_stub_path(),
9863                args: Vec::new(),
9864                env: Vec::new(),
9865                reserved: false,
9866                reserved_prefixes: Vec::new(),
9867                protocol: ModuleProtocol::Subc,
9868                overlap: Default::default(),
9869            })
9870            .unwrap();
9871        update_snapshot(
9872            &module.inner.snapshot,
9873            Some("recovery-snapshot"),
9874            |snapshot| {
9875                snapshot.state = state;
9876                snapshot.enabled = enabled;
9877                seed_crash_restarts(snapshot, restart_count);
9878            },
9879        )
9880        .unwrap();
9881        module
9882    }
9883
9884    #[cfg(target_os = "linux")]
9885    #[tokio::test]
9886    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9887        let supervisor =
9888            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
9889                .with_cgroup_placement(None);
9890        let result = supervisor.spawn(ModuleSpec {
9891            module_id: "no-cgroup-placement".to_string(),
9892            program: fake_aft_stub_path(),
9893            args: Vec::new(),
9894            env: Vec::new(),
9895            reserved: false,
9896            reserved_prefixes: Vec::new(),
9897            protocol: ModuleProtocol::Subc,
9898            overlap: Default::default(),
9899        });
9900
9901        assert!(
9902            result.is_ok(),
9903            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9904        );
9905    }
9906
9907    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9908    async fn undecided_snapshot_uses_shared_restart_predicate() {
9909        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9910            .will_recover_after_connection_loss()
9911            .unwrap());
9912        assert!(
9913            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9914                .will_recover_after_connection_loss()
9915                .unwrap()
9916        );
9917    }
9918
9919    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9920    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9921        assert!(
9922            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9923                .will_recover_after_connection_loss()
9924                .unwrap()
9925        );
9926    }
9927
9928    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9929    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9930        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9931            .will_recover_after_connection_loss()
9932            .unwrap());
9933        assert!(
9934            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9935                .will_recover_after_connection_loss()
9936                .unwrap()
9937        );
9938    }
9939
9940    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9941    async fn warming_snapshot_is_limited_to_startup_phases() {
9942        for state in [
9943            ModuleState::Starting,
9944            ModuleState::Running,
9945            ModuleState::Restarting,
9946        ] {
9947            assert!(
9948                module_with_recovery_snapshot(state, true, 0)
9949                    .is_warming()
9950                    .unwrap(),
9951                "{state:?} should be warming"
9952            );
9953        }
9954        for state in [
9955            ModuleState::Unresponsive,
9956            ModuleState::Draining,
9957            ModuleState::Stopped,
9958            ModuleState::Failed,
9959            ModuleState::Disabled,
9960        ] {
9961            assert!(
9962                !module_with_recovery_snapshot(state, true, 0)
9963                    .is_warming()
9964                    .unwrap(),
9965                "{state:?} should not be warming"
9966            );
9967        }
9968    }
9969
9970    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9971    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
9972        let registry = Arc::new(Registry::default());
9973        let supervisor =
9974            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
9975        let module = supervisor
9976            .spawn(ModuleSpec {
9977                module_id: "terminal-history".to_string(),
9978                program: fake_aft_stub_path(),
9979                args: Vec::new(),
9980                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9981                reserved: false,
9982                reserved_prefixes: Vec::new(),
9983                protocol: ModuleProtocol::Subc,
9984                overlap: Default::default(),
9985            })
9986            .unwrap();
9987
9988        let deadline = Instant::now() + Duration::from_secs(5);
9989        loop {
9990            let history = module.terminal_history();
9991            if history.entries.len() == 2 {
9992                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
9993                assert_eq!(history.dropped, 0);
9994                assert_eq!(
9995                    history
9996                        .entries
9997                        .iter()
9998                        .map(|entry| entry.exit_code)
9999                        .collect::<Vec<_>>(),
10000                    vec![Some(23), Some(23)]
10001                );
10002                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
10003                return;
10004            }
10005            assert!(
10006                Instant::now() < deadline,
10007                "module did not retain two terminal exits: {history:?}"
10008            );
10009            sleep(Duration::from_millis(10)).await;
10010        }
10011    }
10012
10013    /// A disable issued while a crash respawn is still backing off must preempt
10014    /// that respawn: the operator's stop wins, the disable must not queue behind
10015    /// the backoff, and the module must never come back up afterwards.
10016    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10017    async fn disable_during_crash_backoff_cancels_pending_respawn() {
10018        let backoff = Duration::from_secs(2);
10019        let supervisor = Supervisor::new_for_test(
10020            Arc::new(Registry::default()),
10021            RestartPolicy::new(10, backoff),
10022        );
10023        let module = supervisor
10024            .spawn(ModuleSpec {
10025                module_id: "disable-during-backoff".to_string(),
10026                program: fake_aft_stub_path(),
10027                args: Vec::new(),
10028                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10029                reserved: false,
10030                reserved_prefixes: Vec::new(),
10031                protocol: ModuleProtocol::Subc,
10032                overlap: Default::default(),
10033            })
10034            .unwrap();
10035
10036        // Wait for the first crash to put the module into its backoff window.
10037        let deadline = Instant::now() + Duration::from_secs(5);
10038        loop {
10039            if module.status().unwrap().state == ModuleState::Restarting {
10040                break;
10041            }
10042            assert!(
10043                Instant::now() < deadline,
10044                "module never entered the crash backoff"
10045            );
10046            sleep(Duration::from_millis(10)).await;
10047        }
10048
10049        let started = Instant::now();
10050        module.set_enabled(false).await.unwrap();
10051        let waited = started.elapsed();
10052
10053        assert!(
10054            waited < backoff / 2,
10055            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10056        );
10057        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10058
10059        // Outlast the backoff: the respawn it was counting down to must never run.
10060        sleep(backoff + Duration::from_millis(500)).await;
10061        let status = module.status().unwrap();
10062        assert_eq!(status.state, ModuleState::Disabled);
10063        assert_eq!(
10064            status.spawn_generation, 1,
10065            "module respawned after the operator disabled it"
10066        );
10067    }
10068
10069    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
10070    /// the shape of nats-server, the program this rule exists for.
10071    #[cfg(unix)]
10072    fn protocol_none_sigterm_exits_clean_spec(
10073        module_id: &str,
10074        dir: &std::path::Path,
10075    ) -> (ModuleSpec, PathBuf, PathBuf) {
10076        let ready = dir.join("ready");
10077        let marker = dir.join("sigterm");
10078        let spec = ModuleSpec {
10079            module_id: module_id.to_string(),
10080            program: fake_aft_stub_path(),
10081            args: Vec::new(),
10082            env: vec![
10083                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10084                (
10085                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10086                    marker.display().to_string(),
10087                ),
10088                (
10089                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10090                    ready.display().to_string(),
10091                ),
10092            ],
10093            reserved: false,
10094            reserved_prefixes: Vec::new(),
10095            protocol: ModuleProtocol::None,
10096            overlap: Default::default(),
10097        };
10098        (spec, ready, marker)
10099    }
10100
10101    /// Wait for a file the child writes, so a signal is never sent before the
10102    /// child's SIGTERM handler is installed (the default disposition would
10103    /// kill it by signal and the exit would not be clean).
10104    #[cfg(unix)]
10105    async fn wait_for_file(path: &std::path::Path) {
10106        let deadline = Instant::now() + Duration::from_secs(10);
10107        while !path.exists() {
10108            assert!(
10109                Instant::now() < deadline,
10110                "{} never appeared",
10111                path.display()
10112            );
10113            sleep(Duration::from_millis(10)).await;
10114        }
10115    }
10116
10117    /// A protocol-none module that exits 0 because something OUTSIDE the
10118    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10119    /// the crash-path disposition rather than `stopped`.
10120    #[cfg(unix)]
10121    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10122    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10123        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10124        let (spec, ready, marker) =
10125            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10126        let supervisor = Supervisor::new_for_test(
10127            Arc::new(Registry::default()),
10128            RestartPolicy::new(3, Duration::ZERO),
10129        );
10130        let module = supervisor.spawn(spec).unwrap();
10131        wait_for_file(&ready).await;
10132        let first_pid = module
10133            .status()
10134            .unwrap()
10135            .pid
10136            .expect("a running module reports its pid");
10137
10138        rustix::process::kill_process(
10139            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10140            rustix::process::Signal::TERM,
10141        )
10142        .unwrap();
10143
10144        let deadline = Instant::now() + Duration::from_secs(10);
10145        let respawned = loop {
10146            let status = module.status().unwrap();
10147            if status.state == ModuleState::Running
10148                && status.pid.is_some_and(|pid| pid != first_pid)
10149            {
10150                break status;
10151            }
10152            assert!(
10153                Instant::now() < deadline,
10154                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10155            );
10156            sleep(Duration::from_millis(10)).await;
10157        };
10158        assert_eq!(respawned.spawn_generation, 2);
10159        assert!(
10160            marker.exists(),
10161            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10162        );
10163
10164        let history = module.terminal_history();
10165        assert_eq!(history.entries.len(), 1, "{history:?}");
10166        let entry = &history.entries[0];
10167        assert_eq!(entry.exit_code, Some(0));
10168        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10169        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10170
10171        module.stop().await.unwrap();
10172    }
10173
10174    /// Repeated unrequested clean exits of a protocol-none module spend the
10175    /// restart budget exactly as crashes do, and the module ends `failed` with
10176    /// the budget named.
10177    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10178    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10179        let supervisor = Supervisor::new_for_test(
10180            Arc::new(Registry::default()),
10181            RestartPolicy::new(1, Duration::ZERO),
10182        );
10183        let module = supervisor
10184            .spawn(ModuleSpec {
10185                module_id: "none-clean-exit-budget".to_string(),
10186                program: fake_aft_stub_path(),
10187                args: Vec::new(),
10188                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10189                reserved: false,
10190                reserved_prefixes: Vec::new(),
10191                protocol: ModuleProtocol::None,
10192                overlap: Default::default(),
10193            })
10194            .unwrap();
10195
10196        let deadline = Instant::now() + Duration::from_secs(10);
10197        loop {
10198            let status = module.status().unwrap();
10199            if status.state == ModuleState::Failed {
10200                break;
10201            }
10202            assert!(
10203                Instant::now() < deadline,
10204                "module never exhausted its budget: {status:?} {:?}",
10205                module.terminal_history()
10206            );
10207            sleep(Duration::from_millis(10)).await;
10208        }
10209        let history = module.terminal_history();
10210        assert_eq!(
10211            history
10212                .entries
10213                .iter()
10214                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10215                .collect::<Vec<_>>(),
10216            vec![
10217                (Some(0), TerminalDisposition::Restarting),
10218                (Some(0), TerminalDisposition::Failed),
10219            ]
10220        );
10221        let detail = history.entries[1]
10222            .disposition_detail
10223            .as_deref()
10224            .expect("a budget failure names the budget");
10225        assert!(detail.contains("max_restarts=1"), "{detail}");
10226        assert_eq!(module.status().unwrap().spawn_generation, 2);
10227    }
10228
10229    /// A stop the supervisor itself requests still stops a protocol-none
10230    /// module, even though the child answers the SIGTERM with exit 0.
10231    #[cfg(unix)]
10232    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10233    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10234        for disable in [false, true] {
10235            let label = if disable {
10236                "none-requested-disable"
10237            } else {
10238                "none-requested-stop"
10239            };
10240            let dir = subc_test_support::TestTempDir::new(label);
10241            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10242            let supervisor = Supervisor::new_for_test(
10243                Arc::new(Registry::default()),
10244                RestartPolicy::new(3, Duration::ZERO),
10245            );
10246            let module = supervisor.spawn(spec).unwrap();
10247            wait_for_file(&ready).await;
10248
10249            if disable {
10250                module.set_enabled(false).await.unwrap();
10251            } else {
10252                module.stop().await.unwrap();
10253            }
10254            assert!(
10255                marker.exists(),
10256                "{label}: the child must have left through its SIGTERM handler with exit 0"
10257            );
10258
10259            // Long enough for a zero-backoff respawn to have happened if the
10260            // exit had been treated as a crash.
10261            sleep(Duration::from_millis(500)).await;
10262            let status = module.status().unwrap();
10263            let expected = if disable {
10264                ModuleState::Disabled
10265            } else {
10266                ModuleState::Stopped
10267            };
10268            assert_eq!(status.state, expected, "{label}");
10269            assert_eq!(
10270                status.spawn_generation, 1,
10271                "{label}: respawned after a requested stop"
10272            );
10273            let history = module.terminal_history();
10274            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10275            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10276            assert_ne!(
10277                history.entries[0].disposition,
10278                TerminalDisposition::Restarting,
10279                "{label}"
10280            );
10281        }
10282    }
10283
10284    /// A subc-wire module that exits 0 on its own is still a stop: the
10285    /// protocol-none rule must not reach it.
10286    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10287    async fn subc_wire_clean_exit_is_still_a_stop() {
10288        let supervisor = Supervisor::new_for_test(
10289            Arc::new(Registry::default()),
10290            RestartPolicy::new(3, Duration::ZERO),
10291        );
10292        let module = supervisor
10293            .spawn(ModuleSpec {
10294                module_id: "wire-clean-exit".to_string(),
10295                program: fake_aft_stub_path(),
10296                args: Vec::new(),
10297                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10298                reserved: false,
10299                reserved_prefixes: Vec::new(),
10300                protocol: ModuleProtocol::Subc,
10301                overlap: Default::default(),
10302            })
10303            .unwrap();
10304
10305        let deadline = Instant::now() + Duration::from_secs(10);
10306        while module.terminal_history().entries.is_empty() {
10307            assert!(Instant::now() < deadline, "module never exited");
10308            sleep(Duration::from_millis(10)).await;
10309        }
10310        // Long enough for a zero-backoff respawn to have happened.
10311        sleep(Duration::from_millis(500)).await;
10312        let status = module.status().unwrap();
10313        assert_eq!(status.state, ModuleState::Stopped);
10314        assert_eq!(status.spawn_generation, 1);
10315        let history = module.terminal_history();
10316        assert_eq!(history.entries.len(), 1, "{history:?}");
10317        assert_eq!(history.entries[0].exit_code, Some(0));
10318        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10319    }
10320
10321    #[cfg(unix)]
10322    #[tokio::test]
10323    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10324        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10325        let record = dir.join("live-children.json");
10326        let supervisor = Supervisor::new_for_test(
10327            Arc::new(Registry::default()),
10328            RestartPolicy::new(0, Duration::ZERO),
10329        );
10330        let mut runtime = supervisor.runtime_config();
10331        runtime.child_roster.record_to(record.clone());
10332        let gate = Arc::new(super::ReloadExitRecordGate::default());
10333        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10334        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10335        let spec = ModuleSpec {
10336            module_id: "reload-exit-roster".into(),
10337            program: fake_aft_stub_path(),
10338            args: Vec::new(),
10339            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10340            reserved: false,
10341            reserved_prefixes: Vec::new(),
10342            protocol: ModuleProtocol::Subc,
10343            overlap: Default::default(),
10344        };
10345        let mut child = None;
10346        let reload = super::finish_reload_child(
10347            &spec,
10348            &runtime,
10349            &supervisor.registry,
10350            &supervisor.process_liveness,
10351            &snapshot,
10352            &mut child,
10353        );
10354        tokio::pin!(reload);
10355        tokio::select! {
10356            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10357            _ = gate.reached.notified() => {}
10358        }
10359        assert!(runtime
10360            .terminal_ring
10361            .lock()
10362            .unwrap()
10363            .snapshot()
10364            .entries
10365            .is_empty());
10366        assert_eq!(
10367            crate::live_children::read_record(&record).unwrap().len(),
10368            1,
10369            "shutdown must still wait for the reaped child until its terminal record exists"
10370        );
10371        runtime.child_roster.close();
10372        gate.resume.notify_one();
10373        assert!(reload.await.is_err());
10374        assert!(crate::live_children::read_record(&record)
10375            .unwrap()
10376            .is_empty());
10377        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10378        assert_eq!(history.entries.len(), 1);
10379        assert_eq!(
10380            history.entries[0].disposition,
10381            TerminalDisposition::DaemonShutdown
10382        );
10383    }
10384
10385    /// Each restart-producing arm has its own state transition. Keeping their
10386    /// lifetime count assertions adjacent prevents a later new arm from silently
10387    /// spending budget without recording the historical restart.
10388    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10389    async fn every_restart_increment_path_advances_lifetime_count() {
10390        let supervisor = Supervisor::new_for_test(
10391            Arc::new(Registry::default()),
10392            RestartPolicy::new(1, Duration::ZERO),
10393        );
10394        let runtime = supervisor.runtime_config();
10395        let spec = ModuleSpec {
10396            module_id: "lifetime-increment-path".to_string(),
10397            program: PathBuf::from("/unused/lifetime-increment-path"),
10398            args: Vec::new(),
10399            env: Vec::new(),
10400            reserved: false,
10401            reserved_prefixes: Vec::new(),
10402            protocol: ModuleProtocol::Subc,
10403            overlap: Default::default(),
10404        };
10405
10406        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10407        assert!(matches!(
10408            on_child_exit(
10409                &spec,
10410                runtime.restart_policy,
10411                &supervisor.registry,
10412                &crash_snapshot,
10413                &runtime.terminal_ring,
10414                &runtime.spawn_events,
10415                &runtime.child_roster,
10416                ExitReport {
10417                    kind: ExitKind::Crash,
10418                    code: Some(1),
10419                    signal: None,
10420                    at_ms: 1,
10421                },
10422            )
10423            .await,
10424            NextAction::Restart { schedule: _ }
10425        ));
10426        let (crash_restarts, crash_lifetime) = {
10427            let state = lock_snapshot(&crash_snapshot).unwrap();
10428            (state.crash_restarts.len(), state.lifetime_restarts)
10429        };
10430        assert_eq!(crash_restarts, 1);
10431        assert_eq!(crash_lifetime, 1);
10432
10433        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10434        let mut health_child = None;
10435        assert!(matches!(
10436            health_restart_child(
10437                &spec,
10438                &runtime,
10439                &supervisor.registry,
10440                &supervisor.process_liveness,
10441                &health_snapshot,
10442                &mut health_child,
10443                SupervisorHealthStatus::Failing,
10444                None,
10445                2,
10446            )
10447            .await,
10448            Ok(())
10449        ));
10450        assert!(health_child.is_none());
10451        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10452        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10453        let (health_restarts, health_lifetime) = {
10454            let state = lock_snapshot(&health_snapshot).unwrap();
10455            (state.crash_restarts.len(), state.lifetime_restarts)
10456        };
10457        assert_eq!(health_restarts, 1);
10458        assert_eq!(health_lifetime, 1);
10459
10460        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10461        let mut reload_child = None;
10462        assert!(matches!(
10463            handle_reload_spawn_failure(
10464                &spec,
10465                &runtime,
10466                &supervisor.process_liveness,
10467                &reload_snapshot,
10468                &mut reload_child,
10469                "forced reload spawn failure".to_string(),
10470            )
10471            .await,
10472            Err(SuperviseError::ReloadFailed { .. })
10473        ));
10474        let (reload_restarts, reload_lifetime) = {
10475            let state = lock_snapshot(&reload_snapshot).unwrap();
10476            (state.crash_restarts.len(), state.lifetime_restarts)
10477        };
10478        assert_eq!(reload_restarts, 1);
10479        assert_eq!(reload_lifetime, 1);
10480    }
10481
10482    #[tokio::test]
10483    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10484        let supervisor = Supervisor::new_for_test(
10485            Arc::new(Registry::default()),
10486            RestartPolicy::new(3, Duration::ZERO),
10487        );
10488        let runtime = supervisor.runtime_config();
10489        let spec = ModuleSpec {
10490            module_id: "deliberately-severed".to_string(),
10491            program: PathBuf::from("/unused/deliberately-severed"),
10492            args: Vec::new(),
10493            env: Vec::new(),
10494            reserved: false,
10495            reserved_prefixes: Vec::new(),
10496            protocol: ModuleProtocol::Subc,
10497            overlap: Default::default(),
10498        };
10499        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10500        let process = ProcessIdentity {
10501            pid: 41,
10502            start_time: 101,
10503        };
10504        record_deliberate_severance(&snapshot, process).unwrap();
10505        let exit_report = apply_deliberate_severance_marker(
10506            &snapshot,
10507            Some(process),
10508            ExitReport {
10509                kind: ExitKind::Crash,
10510                code: Some(1),
10511                signal: None,
10512                at_ms: 1,
10513            },
10514        );
10515        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10516
10517        assert!(matches!(
10518            on_child_exit(
10519                &spec,
10520                runtime.restart_policy,
10521                &supervisor.registry,
10522                &snapshot,
10523                &runtime.terminal_ring,
10524                &runtime.spawn_events,
10525                &runtime.child_roster,
10526                exit_report,
10527            )
10528            .await,
10529            NextAction::Restart { schedule: _ }
10530        ));
10531        let state = lock_snapshot(&snapshot).unwrap();
10532        assert_eq!(state.lifetime_restarts, 1);
10533        assert_eq!(state.crash_restarts.len(), 0);
10534    }
10535
10536    #[tokio::test]
10537    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10538        let supervisor = Supervisor::new_for_test(
10539            Arc::new(Registry::default()),
10540            RestartPolicy::new(3, Duration::ZERO),
10541        );
10542        let runtime = supervisor.runtime_config();
10543        let spec = ModuleSpec {
10544            module_id: "genuine-crash".to_string(),
10545            program: PathBuf::from("/unused/genuine-crash"),
10546            args: Vec::new(),
10547            env: Vec::new(),
10548            reserved: false,
10549            reserved_prefixes: Vec::new(),
10550            protocol: ModuleProtocol::Subc,
10551            overlap: Default::default(),
10552        };
10553        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10554
10555        assert!(matches!(
10556            on_child_exit(
10557                &spec,
10558                runtime.restart_policy,
10559                &supervisor.registry,
10560                &snapshot,
10561                &runtime.terminal_ring,
10562                &runtime.spawn_events,
10563                &runtime.child_roster,
10564                ExitReport {
10565                    kind: ExitKind::Crash,
10566                    code: Some(1),
10567                    signal: None,
10568                    at_ms: 1,
10569                },
10570            )
10571            .await,
10572            NextAction::Restart { schedule: _ }
10573        ));
10574        let state = lock_snapshot(&snapshot).unwrap();
10575        assert_eq!(state.lifetime_restarts, 1);
10576        assert_eq!(state.crash_restarts.len(), 1);
10577    }
10578
10579    fn crash_exit_report(at_ms: u64) -> ExitReport {
10580        ExitReport {
10581            kind: ExitKind::Crash,
10582            code: Some(1),
10583            signal: None,
10584            at_ms,
10585        }
10586    }
10587
10588    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10589        ModuleSpec {
10590            module_id: module_id.to_string(),
10591            program: PathBuf::from("/unused").join(module_id),
10592            args: Vec::new(),
10593            env: Vec::new(),
10594            reserved: false,
10595            reserved_prefixes: Vec::new(),
10596            protocol: ModuleProtocol::Subc,
10597            overlap: Default::default(),
10598        }
10599    }
10600
10601    /// A real crash loop still stops. Three crashes with nothing aging out spend
10602    /// a budget of two and the third respawn is refused, and both surfaces an
10603    /// operator has -- the log line and the retained terminal record -- name the
10604    /// window rather than only the cap, because `max_restarts=2` alone is what
10605    /// this budget used to mean.
10606    #[tokio::test]
10607    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10608        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10609        let supervisor = Supervisor::new_for_test(
10610            Arc::new(Registry::default()),
10611            RestartPolicy::new(2, Duration::ZERO),
10612        );
10613        let runtime = supervisor.runtime_config();
10614        let spec = windowed_crash_spec("crash-loop-in-window");
10615        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10616
10617        for attempt in 1..=2 {
10618            assert!(
10619                matches!(
10620                    on_child_exit(
10621                        &spec,
10622                        runtime.restart_policy,
10623                        &supervisor.registry,
10624                        &snapshot,
10625                        &runtime.terminal_ring,
10626                        &runtime.spawn_events,
10627                        &runtime.child_roster,
10628                        crash_exit_report(attempt),
10629                    )
10630                    .await,
10631                    NextAction::Restart { schedule: _ }
10632                ),
10633                "crash {attempt} is inside the budget and must respawn"
10634            );
10635        }
10636
10637        assert!(matches!(
10638            on_child_exit(
10639                &spec,
10640                runtime.restart_policy,
10641                &supervisor.registry,
10642                &snapshot,
10643                &runtime.terminal_ring,
10644                &runtime.spawn_events,
10645                &runtime.child_roster,
10646                crash_exit_report(3),
10647            )
10648            .await,
10649            NextAction::Stop { .. }
10650        ));
10651
10652        {
10653            let state = lock_snapshot(&snapshot).unwrap();
10654            assert_eq!(state.state, ModuleState::Failed);
10655            assert_eq!(state.crash_restarts.len(), 2);
10656            assert_eq!(state.lifetime_restarts, 2);
10657        }
10658
10659        let history = runtime
10660            .terminal_ring
10661            .lock()
10662            .expect("terminal ring is not poisoned")
10663            .snapshot();
10664        let last = history
10665            .entries
10666            .last()
10667            .expect("the refused crash is retained");
10668        assert_eq!(last.disposition, TerminalDisposition::Failed);
10669        assert_eq!(
10670            last.disposition_detail.as_deref(),
10671            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10672        );
10673
10674        let captured = crate::router::test_log::captured_logs(&logs);
10675        assert!(
10676            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10677            "the stop must be logged with its window: {captured}"
10678        );
10679    }
10680
10681    /// The rate, stated as a test: three crashes where the first has aged past
10682    /// the window are two crashes as far as the budget is concerned, so the
10683    /// third respawn is allowed and the ring holds only the two recent ones.
10684    ///
10685    /// This is the case a lifetime counter got wrong -- and the case the daemon
10686    /// now hits routinely, since a module exits non-zero every time its
10687    /// connection to the daemon drops.
10688    #[tokio::test]
10689    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10690        let supervisor = Supervisor::new_for_test(
10691            Arc::new(Registry::default()),
10692            RestartPolicy::new(2, Duration::ZERO),
10693        );
10694        let runtime = supervisor.runtime_config();
10695        let spec = windowed_crash_spec("crash-across-windows");
10696        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10697
10698        for attempt in 1..=2 {
10699            assert!(matches!(
10700                on_child_exit(
10701                    &spec,
10702                    runtime.restart_policy,
10703                    &supervisor.registry,
10704                    &snapshot,
10705                    &runtime.terminal_ring,
10706                    &runtime.spawn_events,
10707                    &runtime.child_roster,
10708                    crash_exit_report(attempt),
10709                )
10710                .await,
10711                NextAction::Restart { schedule: _ }
10712            ));
10713        }
10714
10715        // The oldest crash moves out of the window; nothing else about the
10716        // module changes.
10717        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10718            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10719        })
10720        .unwrap();
10721
10722        assert!(
10723            matches!(
10724                on_child_exit(
10725                    &spec,
10726                    runtime.restart_policy,
10727                    &supervisor.registry,
10728                    &snapshot,
10729                    &runtime.terminal_ring,
10730                    &runtime.spawn_events,
10731                    &runtime.child_roster,
10732                    crash_exit_report(3),
10733                )
10734                .await,
10735                NextAction::Restart { schedule: _ }
10736            ),
10737            "a crash older than the window must not hold a budget slot"
10738        );
10739
10740        let state = lock_snapshot(&snapshot).unwrap();
10741        assert_eq!(state.state, ModuleState::Restarting);
10742        assert_eq!(
10743            state.crash_restarts.len(),
10744            2,
10745            "the aged instant is dropped and the new one takes its place"
10746        );
10747        assert_eq!(
10748            state.lifetime_restarts, 3,
10749            "the ledger counts every restart, including the ones the window forgot"
10750        );
10751    }
10752
10753    /// An operator restart hands the budget back whole, and the ledger keeps
10754    /// counting. Those are different questions -- "how close is this module to
10755    /// being stopped" and "how many times has it been replaced" -- and the
10756    /// operator action answers only the first.
10757    #[tokio::test]
10758    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10759        let supervisor = Supervisor::new_for_test(
10760            Arc::new(Registry::default()),
10761            RestartPolicy::new(2, Duration::ZERO),
10762        );
10763        let runtime = supervisor.runtime_config();
10764        let spec = windowed_crash_spec("operator-cleared-budget");
10765        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10766
10767        for attempt in 1..=2 {
10768            assert!(matches!(
10769                on_child_exit(
10770                    &spec,
10771                    runtime.restart_policy,
10772                    &supervisor.registry,
10773                    &snapshot,
10774                    &runtime.terminal_ring,
10775                    &runtime.spawn_events,
10776                    &runtime.child_roster,
10777                    crash_exit_report(attempt),
10778                )
10779                .await,
10780                NextAction::Restart { schedule: _ }
10781            ));
10782        }
10783
10784        reset_restart_count(&snapshot, &spec.module_id).unwrap();
10785        {
10786            let state = lock_snapshot(&snapshot).unwrap();
10787            assert!(
10788                state.crash_restarts.is_empty(),
10789                "an operator restart returns the full budget"
10790            );
10791            assert_eq!(
10792                state.lifetime_restarts, 2,
10793                "clearing the budget must not unmake the crashes"
10794            );
10795        }
10796
10797        assert!(
10798            matches!(
10799                on_child_exit(
10800                    &spec,
10801                    runtime.restart_policy,
10802                    &supervisor.registry,
10803                    &snapshot,
10804                    &runtime.terminal_ring,
10805                    &runtime.spawn_events,
10806                    &runtime.child_roster,
10807                    crash_exit_report(3),
10808                )
10809                .await,
10810                NextAction::Restart { schedule: _ }
10811            ),
10812            "the cleared budget must be spendable again"
10813        );
10814        let state = lock_snapshot(&snapshot).unwrap();
10815        assert_eq!(state.crash_restarts.len(), 1);
10816        assert_eq!(state.lifetime_restarts, 3);
10817    }
10818
10819    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10820    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
10821        let severed = ProcessIdentity {
10822            pid: 41,
10823            start_time: 101,
10824        };
10825        let successor = ProcessIdentity {
10826            pid: 41,
10827            start_time: 202,
10828        };
10829        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
10830        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
10831            state.pid = Some(successor.pid);
10832            state.process_start_time = Some(successor.start_time);
10833        })
10834        .unwrap();
10835        assert!(!module.record_deliberate_severance(severed).unwrap());
10836
10837        let exit_report = apply_deliberate_severance_marker(
10838            &module.inner.snapshot,
10839            Some(successor),
10840            ExitReport {
10841                kind: ExitKind::Crash,
10842                code: Some(1),
10843                signal: None,
10844                at_ms: 1,
10845            },
10846        );
10847
10848        assert_eq!(exit_report.kind, ExitKind::Crash);
10849    }
10850
10851    #[tokio::test]
10852    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
10853        let registry = Registry::default();
10854        let supervisor = Supervisor::new_for_test(
10855            Arc::new(Registry::default()),
10856            RestartPolicy::new(3, Duration::ZERO),
10857        );
10858        let runtime = supervisor.runtime_config();
10859        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10860        let spec = ModuleSpec {
10861            module_id: "drain-deliberate-severance".to_string(),
10862            program: fake_aft_stub_path(),
10863            args: Vec::new(),
10864            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10865            reserved: false,
10866            reserved_prefixes: Vec::new(),
10867            protocol: ModuleProtocol::Subc,
10868            overlap: Default::default(),
10869        };
10870        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10871        let process = ProcessIdentity {
10872            pid: 41,
10873            start_time: 101,
10874        };
10875        child.process_identity = Some(process);
10876        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10877            state.pid = Some(process.pid);
10878            state.process_start_time = Some(process.start_time);
10879        })
10880        .unwrap();
10881        record_deliberate_severance(&snapshot, process).unwrap();
10882
10883        drain_child_to_state(
10884            &spec.module_id,
10885            spec.protocol,
10886            // The child exits on its own; no signal may change the exit this
10887            // test classifies.
10888            StopNotice::SentOverConnection,
10889            &registry,
10890            None,
10891            &snapshot,
10892            &runtime.terminal_ring,
10893            &runtime.spawn_events,
10894            child,
10895            Duration::from_secs(1),
10896            ModuleState::Stopped,
10897            Some(false),
10898        )
10899        .await
10900        .unwrap();
10901
10902        let state = lock_snapshot(&snapshot).unwrap();
10903        assert_eq!(
10904            state.last_exit.as_ref().map(|exit| exit.kind),
10905            Some(ExitKind::DeliberateSeverance)
10906        );
10907        assert_eq!(state.lifetime_restarts, 1);
10908        assert_eq!(state.crash_restarts.len(), 0);
10909        drop(state);
10910        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10911        assert_eq!(
10912            history.entries[0].exit_kind,
10913            subc_control::TerminalExitKind::DeliberateSeverance
10914        );
10915    }
10916
10917    #[tokio::test]
10918    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
10919        let registry = Registry::default();
10920        let supervisor = Supervisor::new_for_test(
10921            Arc::new(Registry::default()),
10922            RestartPolicy::new(3, Duration::ZERO),
10923        );
10924        let runtime = supervisor.runtime_config();
10925        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10926        let spec = ModuleSpec {
10927            module_id: "ordinary-drain".to_string(),
10928            program: fake_aft_stub_path(),
10929            args: Vec::new(),
10930            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10931            reserved: false,
10932            reserved_prefixes: Vec::new(),
10933            protocol: ModuleProtocol::Subc,
10934            overlap: Default::default(),
10935        };
10936        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10937
10938        drain_child_to_state(
10939            &spec.module_id,
10940            spec.protocol,
10941            // The child exits on its own; no signal may change the exit this
10942            // test classifies.
10943            StopNotice::SentOverConnection,
10944            &registry,
10945            None,
10946            &snapshot,
10947            &runtime.terminal_ring,
10948            &runtime.spawn_events,
10949            child,
10950            Duration::from_secs(1),
10951            ModuleState::Stopped,
10952            Some(false),
10953        )
10954        .await
10955        .unwrap();
10956
10957        let state = lock_snapshot(&snapshot).unwrap();
10958        assert_eq!(
10959            state.last_exit.as_ref().map(|exit| exit.kind),
10960            Some(ExitKind::Crash)
10961        );
10962        assert_eq!(state.lifetime_restarts, 0);
10963        assert_eq!(state.crash_restarts.len(), 0);
10964    }
10965
10966    #[test]
10967    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
10968        // The server's generic fatal-routing branch only knows that the
10969        // connection failed; it does not know that the daemon deliberately
10970        // initiated a process-killing severance. Keep this seam explicit so a
10971        // future connection error path cannot silently reintroduce the stale
10972        // exemption that mislabels a later genuine crash.
10973        assert!(!include_str!("server.rs")
10974            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
10975    }
10976
10977    /// The `route.closed` `drained` value must be the quiescence wait's own
10978    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
10979    /// measurement at all and `false` is the one honest constant. This is the exact
10980    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
10981    /// on every return path, including the one that used to return early via `?`
10982    /// with `route.closing` already sent and no `route.closed` ever following.
10983    #[test]
10984    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
10985        assert!(drained_after_quiescence_wait(&Ok(true)));
10986        assert!(!drained_after_quiescence_wait(&Ok(false)));
10987        assert!(!drained_after_quiescence_wait(&Err(
10988            SuperviseError::StatePoisoned { module_id: None }
10989        )));
10990    }
10991
10992    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
10993    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
10994    /// already reaped out-of-band) still leaves a terminal record rather than none
10995    /// at all. Triggering the real `wait()` I/O error from an integration test would
10996    /// need a genuine already-reaped-child race, which is OS-specific and not
10997    /// something this suite attempts elsewhere; this test instead verifies the
10998    /// record produced for that arm end-to-end through the real `TerminalRing`, and
10999    /// the call site itself is verified by inspection to sit in that exact arm.
11000    #[test]
11001    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
11002        let ring = Arc::new(Mutex::new(TerminalRing::new(
11003            TerminalRingConfig::default(),
11004            0,
11005        )));
11006        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
11007
11008        let snapshot = ring.lock().unwrap().snapshot();
11009        assert_eq!(snapshot.entries.len(), 1);
11010        let entry = &snapshot.entries[0];
11011        assert_eq!(entry.exit_code, None);
11012        assert_eq!(entry.exit_signal, None);
11013        assert_eq!(entry.disposition, TerminalDisposition::Failed);
11014    }
11015
11016    #[test]
11017    fn wait_error_exit_path_preserves_spawn_event_density() {
11018        let feed = super::SpawnEventFeed::default();
11019        feed.configure_incarnation("wait-error-density".to_string());
11020        feed.emit_spawned("wait-error", 41, 1);
11021        let ring = Arc::new(Mutex::new(TerminalRing::new(
11022            TerminalRingConfig::default(),
11023            0,
11024        )));
11025
11026        record_wait_error_terminal("wait-error", &ring, &feed);
11027        feed.emit_spawned("after-wait-error", 42, 2);
11028
11029        let state = feed.0.lock().unwrap();
11030        let sequences = state
11031            .events
11032            .iter()
11033            .map(|event| event.cursor.seq)
11034            .collect::<Vec<_>>();
11035        assert_eq!(sequences, vec![1, 2, 3]);
11036        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11037        assert_eq!(state.events[1].exit_code, None);
11038        assert_eq!(state.events[1].exit_signal, None);
11039    }
11040
11041    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
11042    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
11043    /// not a clean exit it never actually observed.
11044    #[test]
11045    fn wait_error_exit_report_is_classified_as_a_crash() {
11046        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11047    }
11048}
11049
11050#[cfg(test)]
11051mod health_evidence_tests {
11052    use super::{HealthProbeError, HealthProbeEvidence};
11053    use std::collections::HashSet;
11054
11055    /// The evidential asymmetry, asserted rather than described.
11056    ///
11057    /// Exactly ONE observation is proof a module cannot serve, and the one that
11058    /// fires under CPU starvation is not it. Before the split, all fifteen
11059    /// construction sites collapsed into a single String, so a timeout carried the
11060    /// same weight as a dead lane -- which is how a healthy module was restarted
11061    /// three times in one day.
11062    #[test]
11063    fn only_a_dead_lane_is_proof_of_death() {
11064        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11065        // Three non-proof classes, each for a different reason: silence is
11066        // consistent with health, a bad answer proves the module ALIVE, and a
11067        // daemon-side fault never reached the module at all.
11068        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11069        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11070        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11071    }
11072
11073    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
11074    ///
11075    /// A shared label renders two different observations identically in the line an
11076    /// operator reads after an unexplained restart -- the exact confusion this
11077    /// change removes.
11078    #[test]
11079    fn every_evidence_class_has_a_distinct_label() {
11080        let labels = [
11081            HealthProbeError::lane_dead("").label(),
11082            HealthProbeError::no_answer("").label(),
11083            HealthProbeError::bad_answer("").label(),
11084            HealthProbeError::misconfigured("").label(),
11085        ];
11086        let unique: HashSet<_> = labels.iter().collect();
11087        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11088    }
11089
11090    /// The class is additional information, not a replacement.
11091    ///
11092    /// An operator needs both "this was silence" and the specific text saying how
11093    /// long we waited; a classification that swallowed the message would trade one
11094    /// missing distinction for another.
11095    #[test]
11096    fn classification_preserves_the_original_message() {
11097        let err = HealthProbeError::no_answer("module did not answer within 5s");
11098        assert_eq!(err.to_string(), "module did not answer within 5s");
11099        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11100    }
11101}
11102
11103#[cfg(test)]
11104mod health_tombstone_tests {
11105    use std::{path::PathBuf, sync::Arc, time::Duration};
11106
11107    use subc_protocol::{
11108        manifest::Concurrency,
11109        session::{HealthStatus, ModuleControlResponse},
11110    };
11111    use tokio::sync::mpsc;
11112
11113    use super::{
11114        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11115        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11116    };
11117    use crate::{
11118        control::ControlHandler,
11119        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11120        registry::{ConnectionId, Registry},
11121        router::FrameSink,
11122    };
11123
11124    struct ProbeHarness {
11125        spec: ModuleSpec,
11126        runtime: SupervisorRuntimeConfig,
11127        forwarding: Arc<ForwardingTable>,
11128        module_connection: ConnectionId,
11129        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11130        handler: ControlHandler,
11131        module: super::SupervisedModule,
11132    }
11133
11134    fn probe_harness() -> ProbeHarness {
11135        let registry = Arc::new(Registry::default());
11136        let forwarding = Arc::new(ForwardingTable::default());
11137        let supervisor_handle = super::SupervisorHandle::new();
11138        let health = HealthConfig {
11139            http: None,
11140            cadence: Duration::from_secs(30),
11141            deadline: Duration::from_secs(5),
11142            failure_threshold: 3,
11143            on_degraded: HealthAction::Report,
11144            on_failing: HealthAction::Report,
11145            critical: false,
11146        };
11147        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11148            .with_forwarding(Arc::clone(&forwarding))
11149            .with_handle(supervisor_handle.clone())
11150            .with_health_config(health);
11151        let spec = ModuleSpec {
11152            module_id: "late-health-module".to_string(),
11153            program: PathBuf::from("disabled-module"),
11154            args: Vec::new(),
11155            env: Vec::new(),
11156            reserved: false,
11157            reserved_prefixes: Vec::new(),
11158            protocol: ModuleProtocol::Subc,
11159            overlap: Default::default(),
11160        };
11161        let module = supervisor
11162            .supervise_configured(spec.clone(), false)
11163            .unwrap();
11164        let runtime = supervisor.runtime_config();
11165        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11166            .with_supervisor(supervisor_handle);
11167        let module_connection = ConnectionId::new(700);
11168        let (module_tx, module_rx) = mpsc::channel(8);
11169        forwarding
11170            .register_module_connection(
11171                module_connection,
11172                spec.module_id.clone(),
11173                subc_protocol::PROTOCOL_VERSION,
11174                Concurrency::ModuleManaged,
11175                FrameSink::new(module_tx),
11176            )
11177            .unwrap();
11178
11179        ProbeHarness {
11180            spec,
11181            runtime,
11182            forwarding,
11183            module_connection,
11184            module_rx,
11185            handler,
11186            module,
11187        }
11188    }
11189
11190    async fn finish_after(
11191        harness: &mut ProbeHarness,
11192        stall: Duration,
11193    ) -> ModuleControlRpcCompletion {
11194        assert!(stall > harness.runtime.health.deadline);
11195        let deadline = harness.runtime.health.deadline;
11196        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11197        let answer = async {
11198            let frame = harness.module_rx.recv().await.expect("health.check frame");
11199            tokio::time::advance(deadline).await;
11200            tokio::task::yield_now().await;
11201            tokio::time::advance(stall - deadline).await;
11202            harness
11203                .forwarding
11204                .complete_module_control_rpc(
11205                    harness.module_connection,
11206                    frame.header.corr,
11207                    Some("health.check"),
11208                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11209                        status: HealthStatus::Ok,
11210                        detail: None,
11211                        metrics: None,
11212                    }),
11213                )
11214                .unwrap()
11215        };
11216        let (probe_result, completion) = tokio::join!(probe, answer);
11217        let err = probe_result.expect_err("probe must miss its deadline");
11218        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11219        completion
11220    }
11221
11222    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11223        let deadline = harness.runtime.health.deadline;
11224        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11225        let exhaust_deadline = async {
11226            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11227            tokio::time::advance(deadline).await;
11228            tokio::task::yield_now().await;
11229        };
11230        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11231        let err = probe_result.expect_err("probe must miss its deadline");
11232        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11233    }
11234
11235    async fn run_probe_cycle(harness: &mut ProbeHarness, answer: bool) {
11236        let registry = Arc::clone(&harness.module.inner.registry);
11237        let snapshot = Arc::clone(&harness.module.inner.snapshot);
11238        let process_liveness = super::SupervisorProcessLiveness::default();
11239        let mut child = None;
11240        let cycle = super::run_health_probe_cycle(
11241            &harness.spec,
11242            &harness.runtime,
11243            &registry,
11244            &process_liveness,
11245            &snapshot,
11246            &mut child,
11247        );
11248        let peer = async {
11249            let frame = harness.module_rx.recv().await.expect("health.check frame");
11250            if answer {
11251                harness
11252                    .forwarding
11253                    .complete_module_control_rpc(
11254                        harness.module_connection,
11255                        frame.header.corr,
11256                        Some("health.check"),
11257                        ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11258                            status: HealthStatus::Ok,
11259                            detail: None,
11260                            metrics: Some(serde_json::json!({"ready": true})),
11261                        }),
11262                    )
11263                    .unwrap();
11264            } else {
11265                tokio::time::advance(harness.runtime.health.deadline).await;
11266                tokio::task::yield_now().await;
11267            }
11268        };
11269        tokio::join!(cycle, peer);
11270    }
11271
11272    #[tokio::test(start_paused = true)]
11273    async fn unanswered_probe_is_unknown_until_threshold_and_ok_report_recovers() {
11274        let mut harness = probe_harness();
11275        // Drive the probe cycle directly with an in-memory wire peer. Stop the
11276        // disabled module's monitor so only this test owns lifecycle transitions;
11277        // no OS process is launched, and a restart is observed at scheduling.
11278        let monitor = harness.module.inner.monitor.lock().unwrap().take().unwrap();
11279        monitor.abort();
11280        let _ = monitor.await;
11281        super::update_snapshot(&harness.module.inner.snapshot, None, |state| {
11282            state.enabled = true;
11283            state.state = super::ModuleState::Running;
11284            state.process_alive = true;
11285        })
11286        .unwrap();
11287
11288        run_probe_cycle(&mut harness, true).await;
11289        assert_eq!(
11290            harness.module.status().unwrap().health.status,
11291            super::SupervisorHealthStatus::Ok
11292        );
11293
11294        for failures in 1..harness.runtime.health.failure_threshold {
11295            run_probe_cycle(&mut harness, false).await;
11296            let status = harness.module.status().unwrap();
11297            assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11298            assert_eq!(status.health.consecutive_failures, failures);
11299            assert!(status.health.last_probe_ms.is_some());
11300            assert!(status.health.detail.unwrap().starts_with("[no-answer]"));
11301            assert!(status.health.metrics.is_none());
11302            assert_eq!(status.state, super::ModuleState::Running);
11303            assert!(status.process_alive);
11304            assert_eq!(status.restart_count, 0);
11305            assert_eq!(status.lifetime_restarts, 0);
11306            assert!(status.health.last_action.is_none());
11307            assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11308        }
11309
11310        run_probe_cycle(&mut harness, true).await;
11311        let recovered = harness.module.status().unwrap();
11312        assert_eq!(recovered.health.status, super::SupervisorHealthStatus::Ok);
11313        assert_eq!(recovered.health.consecutive_failures, 0);
11314        assert!(recovered.health.detail.is_none());
11315        assert_eq!(
11316            recovered.health.metrics,
11317            Some(serde_json::json!({"ready": true}))
11318        );
11319        assert_eq!(recovered.lifetime_restarts, 0);
11320
11321        for failures in 1..=harness.runtime.health.failure_threshold {
11322            run_probe_cycle(&mut harness, false).await;
11323            let status = harness.module.status().unwrap();
11324            assert_eq!(status.health.consecutive_failures, failures);
11325            if failures < harness.runtime.health.failure_threshold {
11326                assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11327                assert_eq!(status.state, super::ModuleState::Running);
11328                assert_eq!(status.lifetime_restarts, 0);
11329                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11330            } else {
11331                assert_eq!(
11332                    status.health.status,
11333                    super::SupervisorHealthStatus::Unresponsive
11334                );
11335                assert_eq!(status.state, super::ModuleState::Restarting);
11336                assert_eq!(status.restart_count, 1);
11337                assert_eq!(status.lifetime_restarts, 1);
11338                assert_eq!(status.health.last_action.as_deref(), Some("restart"));
11339                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_some());
11340            }
11341        }
11342    }
11343
11344    #[tokio::test(start_paused = true)]
11345    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
11346        let mut harness = probe_harness();
11347
11348        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
11349        let first_latency = match &first {
11350            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11351            other => panic!("late answer was not retained: {other:?}"),
11352        };
11353        assert!(harness.handler.observe_module_control_completion(first));
11354
11355        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
11356        let second_latency = match &second {
11357            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11358            other => panic!("late answer was not retained: {other:?}"),
11359        };
11360        assert!(harness.handler.observe_module_control_completion(second));
11361
11362        assert_eq!(first_latency, Duration::from_secs(8));
11363        assert_eq!(
11364            second_latency - first_latency,
11365            Duration::from_secs(3),
11366            "latency must grow linearly with the additional stall"
11367        );
11368        let health = harness.module.status().unwrap().health;
11369        assert_eq!(health.late_answer_count, 2);
11370        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
11371    }
11372
11373    /// A module that answers every probe late must never march to the kill
11374    /// threshold: the late answer proves it is alive, so it must clear the miss
11375    /// streak the timeout recorded. Without the reset, a CPU-starved module
11376    /// that serves every probe seconds past the deadline accumulates
11377    /// `consecutive_failures` to the threshold and is killed — the exact
11378    /// sequence from the 2026-08-14 aft disable, where the daemon logged
11379    /// "proves the module is alive" five times while counting five misses.
11380    #[tokio::test(start_paused = true)]
11381    async fn late_answer_clears_the_consecutive_failure_streak() {
11382        let mut harness = probe_harness();
11383
11384        // Timeout recorded first: the probe path saw no answer in time.
11385        time_out_without_answer(&mut harness).await;
11386        harness
11387            .module
11388            .record_health_probe_failure_for_test("[no-answer] test miss")
11389            .unwrap();
11390        assert_eq!(
11391            harness.module.status().unwrap().health.consecutive_failures,
11392            1,
11393            "precondition: the miss must be on the streak before the late answer"
11394        );
11395
11396        // The stalled reply then lands: proof of life.
11397        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
11398        assert!(matches!(
11399            late,
11400            ModuleControlRpcCompletion::LateHealthAnswer { .. }
11401        ));
11402        assert!(harness.handler.observe_module_control_completion(late));
11403
11404        let health = harness.module.status().unwrap().health;
11405        assert_eq!(
11406            health.consecutive_failures, 0,
11407            "a late answer is an answer: the streak must reset"
11408        );
11409        assert_eq!(health.late_answer_count, 1);
11410    }
11411
11412    #[tokio::test(start_paused = true)]
11413    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
11414        let mut harness = probe_harness();
11415
11416        for _ in 0..20 {
11417            time_out_without_answer(&mut harness).await;
11418            assert_eq!(
11419                harness.forwarding.health_probe_tombstone_count().unwrap(),
11420                1
11421            );
11422        }
11423    }
11424}
11425
11426#[cfg(test)]
11427mod child_env_tests {
11428    use super::{
11429        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
11430        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
11431        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
11432    };
11433    use std::{ffi::OsStr, path::PathBuf};
11434    use tokio::process::Command;
11435
11436    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
11437        ModuleSpec {
11438            module_id: "env-plan".to_string(),
11439            program: PathBuf::from("/nonexistent"),
11440            args: Vec::new(),
11441            env,
11442            reserved: false,
11443            reserved_prefixes: Vec::new(),
11444            protocol: ModuleProtocol::Subc,
11445            overlap: Default::default(),
11446        }
11447    }
11448
11449    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
11450    /// one still gets its own.
11451    ///
11452    /// This is the narrow goal `env_clear()` was reached for, and the reason the
11453    /// fix is `env_remove` rather than deleting the line: an operator's ambient
11454    /// filter silently becoming an unconfigured module's log level is a real
11455    /// defect, just a much smaller one than clearing the environment.
11456    ///
11457    /// Asserted on the command plan rather than a spawned child because proving
11458    /// the ABSENCE of an inherited variable needs the parent's environment
11459    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
11460    /// removal as `(key, None)`, which is exactly the distinction wanted: not
11461    /// "absent because nobody set it" but "explicitly unset for the child".
11462    #[test]
11463    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
11464        let mut command = Command::new("/nonexistent");
11465        apply_child_env(&mut command, &spec(Vec::new()));
11466        let removed = command
11467            .as_std()
11468            .get_envs()
11469            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
11470        assert!(
11471            removed,
11472            "ambient CK_LOG must be explicitly removed for an unconfigured module"
11473        );
11474
11475        let mut configured = Command::new("/nonexistent");
11476        apply_child_env(
11477            &mut configured,
11478            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
11479        );
11480        let effective = configured
11481            .as_std()
11482            .get_envs()
11483            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
11484            .last()
11485            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11486        assert_eq!(
11487            effective,
11488            Some(Some("debug".to_string())),
11489            "a module's configured CK_LOG must survive the ambient removal"
11490        );
11491    }
11492
11493    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
11494    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
11495    /// the same reason as the CK_LOG test above.
11496    ///
11497    /// The argument is the load-bearing half: a stock binary exits on an
11498    /// unknown flag before it listens, so with `--subc` appended the mode
11499    /// could not supervise the one process it exists for. Found by the first
11500    /// conformance run (nats-server: `flag provided but not defined: -subc`).
11501    #[test]
11502    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
11503        let connection_file = std::path::Path::new("/run/subc-connection.json");
11504        let handle = SupervisorHandle::new();
11505
11506        let mut none_spec = spec(Vec::new());
11507        none_spec.protocol = ModuleProtocol::None;
11508        let mut none = Command::new("/nonexistent");
11509        let none_handoff =
11510            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
11511                .expect("protocol-none spawn args apply");
11512        assert!(
11513            none_handoff.is_none(),
11514            "protocol:none spawn must not receive a nonce descriptor"
11515        );
11516        assert!(
11517            !none.as_std().get_envs().any(|(key, value)| key
11518                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
11519                && value.is_some()),
11520            "protocol:none spawn must not name a nonce descriptor"
11521        );
11522        let none_args: Vec<String> = none
11523            .as_std()
11524            .get_args()
11525            .map(|a| a.to_string_lossy().into_owned())
11526            .collect();
11527        assert!(
11528            !none_args.iter().any(|a| a == SUBC_ARG),
11529            "protocol:none argv must not carry --subc; got {none_args:?}"
11530        );
11531        let none_has_nonce = none
11532            .as_std()
11533            .get_envs()
11534            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
11535        assert!(
11536            !none_has_nonce,
11537            "protocol:none spawn must not receive a launch nonce"
11538        );
11539        let none_has_module_id = none
11540            .as_std()
11541            .get_envs()
11542            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
11543        assert!(
11544            none_has_module_id,
11545            "SUBC_MODULE_ID is inert and stays on every path"
11546        );
11547        assert!(
11548            handle.spawn_nonce(&none_spec.module_id).is_none(),
11549            "no nonce record for a process that will never present one"
11550        );
11551
11552        // Control: the subc-wire path is unchanged by the branch above.
11553        let wire_spec = spec(Vec::new());
11554        let mut wire = Command::new("/nonexistent");
11555        let wire_handoff =
11556            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
11557                .expect("subc-wire spawn args apply");
11558        let wire_fd_env = wire
11559            .as_std()
11560            .get_envs()
11561            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
11562            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11563        #[cfg(unix)]
11564        assert_eq!(
11565            wire_fd_env,
11566            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11567            "a subc-wire spawn names the pipe it will receive at descriptor 3"
11568        );
11569        #[cfg(not(unix))]
11570        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11571        let wire_args: Vec<String> = wire
11572            .as_std()
11573            .get_args()
11574            .map(|a| a.to_string_lossy().into_owned())
11575            .collect();
11576        assert_eq!(
11577            wire_args,
11578            vec![
11579                SUBC_ARG.to_string(),
11580                connection_file.to_string_lossy().into_owned()
11581            ],
11582            "a subc-wire spawn still carries --subc <path>"
11583        );
11584        assert_eq!(
11585            wire.as_std()
11586                .get_envs()
11587                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11588            !cfg!(unix),
11589            "only Windows supplies the environment nonce"
11590        );
11591        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11592    }
11593
11594    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
11595    /// spec tries to set it; only a swap candidate carries it.
11596    ///
11597    /// "Set it only on candidates" is not enough, because spawn applies the
11598    /// spec's env verbatim and the daemon's own environment is inherited: either
11599    /// could hand a plain restart the swap role, and a module reading it would
11600    /// warm on its long swap budget while callers wait. Asserted as an explicit
11601    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
11602    /// test above gives.
11603    #[test]
11604    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11605        let role = |command: &Command| {
11606            command
11607                .as_std()
11608                .get_envs()
11609                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11610                .last()
11611                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11612        };
11613        let forged = spec(vec![(
11614            SUBC_SPAWN_ROLE_ENV.to_string(),
11615            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11616        )]);
11617
11618        let mut plain = Command::new("/nonexistent");
11619        apply_child_env(&mut plain, &forged);
11620        apply_spawn_role(&mut plain, SpawnRole::Plain);
11621        assert_eq!(
11622            role(&plain),
11623            Some(None),
11624            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11625        );
11626
11627        let mut candidate = Command::new("/nonexistent");
11628        apply_child_env(&mut candidate, &spec(Vec::new()));
11629        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11630        assert_eq!(
11631            role(&candidate),
11632            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11633        );
11634    }
11635
11636    /// Daemon-private capture retention keys never reach the child.
11637    ///
11638    /// cortexkit-log exposes retention as a Rust struct with no environment
11639    /// names, so these entries are supervisor metadata. Passing them through
11640    /// would invent a public child-process contract by accident.
11641    #[test]
11642    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11643        let mut command = Command::new("/nonexistent");
11644        apply_child_env(
11645            &mut command,
11646            &spec(vec![
11647                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11648                ("KEPT".to_string(), "yes".to_string()),
11649            ]),
11650        );
11651        let keys: Vec<String> = command
11652            .as_std()
11653            .get_envs()
11654            .filter(|(_, value)| value.is_some())
11655            .map(|(key, _)| key.to_string_lossy().into_owned())
11656            .collect();
11657        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11658        assert!(
11659            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11660            "daemon-private capture key leaked to the child: {keys:?}"
11661        );
11662    }
11663}
11664
11665#[cfg(test)]
11666mod jitter_tests {
11667    use super::jittered_health_delay;
11668    use std::{collections::HashSet, time::Duration};
11669
11670    /// Module ids drawn from a real fleet, so the dispersal claim is about names
11671    /// that actually occur rather than invented ones.
11672    ///
11673    /// This is a SAMPLE, not a registry: the property under test is that distinct
11674    /// ids disperse, which holds for any set of distinct strings. Several entries
11675    /// are already historical (modules get renamed), and that costs nothing here --
11676    /// but it means a reader must not mistake this for the live module set, and a
11677    /// rename sweep will match it without there being anything to change.
11678    const FLEET: [&str; 14] = [
11679        "aft",
11680        "alfonso-core",
11681        "magic-context",
11682        "broca",
11683        "thalamus",
11684        "quota",
11685        "engram",
11686        "plexus",
11687        "cerebellum",
11688        "astrocyte",
11689        "synapse",
11690        "subc-mcp",
11691        "cortexkit-credentials",
11692        "subc-federation",
11693    ];
11694
11695    /// Probes must not converge after a fleet-wide restart.
11696    ///
11697    /// This is the property the jitter exists for: every module reconnects at
11698    /// once, and without dispersal all fourteen would then probe on the same
11699    /// tick forever. Nothing failed visibly when this went untested -- a
11700    /// convergent fleet still probes correctly, just in a burst, so the symptom
11701    /// is a periodic load spike that looks like whatever else is running.
11702    #[test]
11703    fn probe_delays_disperse_across_the_fleet() {
11704        let cadence = Duration::from_secs(30);
11705        let delays: HashSet<Duration> = FLEET
11706            .iter()
11707            .map(|id| jittered_health_delay(id, 0, cadence))
11708            .collect();
11709        assert_eq!(
11710            delays.len(),
11711            FLEET.len(),
11712            "every supervised module must land on its own probe offset"
11713        );
11714    }
11715
11716    /// The offset may only ever DELAY a probe, never bring it forward.
11717    ///
11718    /// A delay below the cadence would probe a module more often than
11719    /// configured, which is the opposite of what an operator asked for and
11720    /// would tighten the failure budget without anyone changing it.
11721    #[test]
11722    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11723        let cadence = Duration::from_secs(30);
11724        let span = cadence / 10;
11725        for id in FLEET {
11726            for probe_index in 0..8 {
11727                let delay = jittered_health_delay(id, probe_index, cadence);
11728                assert!(
11729                    delay >= cadence,
11730                    "{id}#{probe_index}: jitter must not shorten the cadence"
11731                );
11732                assert!(
11733                    delay < cadence + span,
11734                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11735                );
11736            }
11737        }
11738    }
11739
11740    /// A module keeps its offset across daemon restarts.
11741    ///
11742    /// The delay is derived rather than randomised precisely so a restart does
11743    /// not re-roll every module into a fresh chance of collision. A random
11744    /// source would satisfy the dispersal test above and quietly lose this.
11745    #[test]
11746    fn a_module_offset_is_stable_across_restarts() {
11747        let cadence = Duration::from_secs(30);
11748        for id in FLEET {
11749            assert_eq!(
11750                jittered_health_delay(id, 0, cadence),
11751                jittered_health_delay(id, 0, cadence),
11752                "{id}: the same module and probe index must produce the same offset"
11753            );
11754        }
11755    }
11756
11757    /// A zero cadence disables probing rather than producing a busy loop.
11758    #[test]
11759    fn zero_cadence_yields_zero_delay() {
11760        assert_eq!(
11761            jittered_health_delay("aft", 0, Duration::ZERO),
11762            Duration::ZERO
11763        );
11764    }
11765}
11766
11767#[cfg(all(test, target_os = "linux"))]
11768mod cgroup_placement_tests {
11769    use super::{
11770        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11771        SupervisedChild,
11772    };
11773    use crate::stderr_tail::{StderrRing, StderrTailConfig};
11774    use std::{
11775        fs, io,
11776        path::{Path, PathBuf},
11777        sync::{Arc, Mutex},
11778    };
11779    use subc_test_support::TestTempDir;
11780    use tokio::process::Command;
11781
11782    #[tokio::test]
11783    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11784        use super::*;
11785        let dir = TestTempDir::new("unique-spawn-cgroups");
11786        let root = PathBuf::from(format!(
11787            "/sys/fs/cgroup/subc-unique-test-{}-{}",
11788            std::process::id(),
11789            unix_ms_now()
11790        ));
11791        if let Err(error) = fs::create_dir(&root) {
11792            assert!(
11793                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11794                "required cgroup test cannot execute: {error}"
11795            );
11796            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11797            return;
11798        }
11799        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11800        let group_count = || {
11801            fs::read_dir(root.join("subc-modules"))
11802                .unwrap()
11803                .map(|entry| entry.unwrap().file_type().unwrap())
11804                .filter(|kind| kind.is_dir())
11805                .count()
11806        };
11807        let supervisor =
11808            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
11809                .with_cgroup_placement(Some(placement.clone()));
11810        let runtime = supervisor.runtime_config();
11811        let mut spec = ModuleSpec {
11812            module_id: "unique-spawn".into(),
11813            program: PathBuf::from("/bin/sleep"),
11814            args: vec!["60".into()],
11815            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11816                .into_iter()
11817                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
11818                .collect(),
11819            reserved: false,
11820            reserved_prefixes: vec![],
11821            protocol: ModuleProtocol::None,
11822            overlap: Default::default(),
11823        };
11824        let spawn = |spec: &ModuleSpec| {
11825            spawn_child(
11826                spec,
11827                None,
11828                None,
11829                &runtime.stderr_ring,
11830                None,
11831                &runtime.child_roster,
11832                Some(&placement),
11833            )
11834            .unwrap()
11835        };
11836        let mut live = spawn(&spec);
11837        for _ in 0..3 {
11838            // A new process can enter the old slot while retirement is pending.
11839            let next = spawn(&spec);
11840            assert_ne!(live.module_id, next.module_id);
11841            live.start_kill().unwrap();
11842            live.wait().await.unwrap();
11843            live = next;
11844            assert!(
11845                live.child.try_wait().unwrap().is_none(),
11846                "retiring the old slot must not kill the replacement"
11847            );
11848            assert_eq!(
11849                group_count(),
11850                1,
11851                "only the live spawn's cgroup should remain"
11852            );
11853        }
11854        supervisor.begin_daemon_shutdown();
11855        let reap = tokio::spawn(async move {
11856            live.wait().await.unwrap();
11857        });
11858        supervisor
11859            .end_children_for_daemon_shutdown(false, std::future::pending())
11860            .await;
11861        reap.await.unwrap();
11862        assert_eq!(group_count(), 0);
11863        // A normal exit uses the same tree-cleanup path as a killed spawn.
11864        spec.program = PathBuf::from("/bin/true");
11865        spec.args.clear();
11866        let fresh_roster = ChildRoster::default();
11867        let mut short = spawn_child(
11868            &spec,
11869            None,
11870            None,
11871            &runtime.stderr_ring,
11872            None,
11873            &fresh_roster,
11874            Some(&placement),
11875        )
11876        .unwrap();
11877        short.wait().await.unwrap();
11878        assert_eq!(group_count(), 0);
11879        spec.module_id = "_".repeat(255);
11880        let mut long_id = spawn_child(
11881            &spec,
11882            None,
11883            None,
11884            &runtime.stderr_ring,
11885            None,
11886            &fresh_roster,
11887            Some(&placement),
11888        )
11889        .unwrap();
11890        long_id.wait().await.unwrap();
11891        assert_eq!(
11892            group_count(),
11893            0,
11894            "valid long module IDs must not exceed cgroup NAME_MAX"
11895        );
11896        fs::remove_dir(root.join("subc-modules")).unwrap();
11897        fs::remove_dir(root).unwrap();
11898        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
11899    }
11900
11901    #[test]
11902    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
11903        let path = Path::new("/definitely-missing-subc-cgroup");
11904        let mut command = Command::new("true");
11905        let error = apply_cgroup_placement(
11906            &mut command,
11907            &ModuleSpec {
11908                module_id: "broken-cgroup".to_string(),
11909                program: PathBuf::from("true"),
11910                args: Vec::new(),
11911                env: Vec::new(),
11912                reserved: false,
11913                reserved_prefixes: Vec::new(),
11914                protocol: ModuleProtocol::Subc,
11915                overlap: Default::default(),
11916            },
11917            path,
11918        )
11919        .expect_err("a parent cgroup open failure must reject the supervised spawn");
11920        let reason = error.to_string();
11921
11922        assert!(
11923            matches!(error, SuperviseError::Cgroup { .. }),
11924            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
11925        );
11926        assert!(
11927            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
11928            "parent cgroup open failure must name cgroup.procs: {reason}"
11929        );
11930    }
11931
11932    #[tokio::test]
11933    async fn reaping_a_child_removes_its_empty_module_cgroup() {
11934        let root = TestTempDir::new("supervisor-reap-cgroup");
11935        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11936        let placement = subc_cgroup::prepare_at(&root)
11937            .expect("prepare scratch cgroup root")
11938            .expect("scratch root has a cgroup.procs marker");
11939        let module_id = "reaped-module";
11940        let module = placement
11941            .module_path(module_id)
11942            .expect("create scratch module cgroup");
11943        let child = Command::new("true")
11944            .env("XDG_DATA_HOME", root.path())
11945            .env("XDG_RUNTIME_DIR", root.path())
11946            .env("XDG_CONFIG_HOME", root.path())
11947            .spawn()
11948            .expect("spawn short-lived child");
11949        let pid = child.id().expect("spawned child has pid");
11950        let mut child = SupervisedChild {
11951            child,
11952            protocol: ModuleProtocol::Subc,
11953            module_id: module_id.to_string(),
11954            cgroup_placement: Some(placement),
11955            stdout_pump: None,
11956            stderr_pump: None,
11957            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
11958            spawned_at_ms: 0,
11959            spawned_from: PathBuf::from("true"),
11960            spawned_file_identity: None,
11961            process_start_time: None,
11962            process_identity: None,
11963            pid,
11964            roster_guard: None,
11965            #[cfg(target_os = "macos")]
11966            privacy_exec: None,
11967            spawn_failure: None,
11968        };
11969
11970        child.wait().await.expect("reap short-lived child");
11971
11972        assert!(
11973            !module.exists(),
11974            "reaping the supervised child must remove its empty cgroup"
11975        );
11976    }
11977
11978    #[test]
11979    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
11980        let root = TestTempDir::new("supervisor-non-empty-cgroup");
11981        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11982        let placement = subc_cgroup::prepare_at(&root)
11983            .expect("prepare scratch cgroup root")
11984            .expect("scratch root has a cgroup.procs marker");
11985        let module = placement
11986            .module_path("surviving-module")
11987            .expect("create scratch module cgroup");
11988        fs::write(module.join("surviving-process"), b"still present")
11989            .expect("make scratch cgroup non-empty");
11990        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
11991
11992        remove_module_cgroup(&placement, "surviving-module");
11993
11994        let logs = crate::router::test_log::captured_logs(&logs);
11995        assert!(
11996            module.exists(),
11997            "failed removal must leave the cgroup intact"
11998        );
11999        assert!(
12000            logs.contains("could not remove module cgroup after process exit; continuing teardown")
12001                && logs.contains("surviving-module"),
12002            "best-effort removal must report the failure without returning it: {logs}"
12003        );
12004    }
12005
12006    #[test]
12007    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
12008        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
12009        let reason = SuperviseError::Spawn {
12010            program: PathBuf::from("/bin/true"),
12011            source: io::Error::from_raw_os_error(13),
12012            cgroup_path: Some(cgroup_path.clone()),
12013        }
12014        .to_string();
12015
12016        assert!(
12017            reason.contains(&cgroup_path.display().to_string()),
12018            "a pre_exec spawn failure must name the cgroup path: {reason}"
12019        );
12020    }
12021}
12022
12023#[cfg(test)]
12024mod spawn_subscriber_lag_tests {
12025    use super::*;
12026
12027    /// A subscriber whose connection stops draining is dropped once its frame
12028    /// channel fills. The client must learn that from a terminal Error frame
12029    /// after the frames already queued for it, not from a stream that simply
12030    /// goes quiet.
12031    #[tokio::test]
12032    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
12033        let feed = SpawnEventFeed::default();
12034        feed.configure_incarnation("lag-incarnation".to_string());
12035        // A one-slot connection queue that nobody reads until the emits are
12036        // done: the forwarder parks on it and the subscriber channel fills.
12037        let (tx, mut rx) = mpsc::channel(1);
12038        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
12039            .expect("subscribe");
12040        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
12041        for index in 0..emitted {
12042            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
12043            // Let the forwarder take what it can so the fill point is the
12044            // subscriber channel, not a scheduling accident.
12045            tokio::task::yield_now().await;
12046        }
12047        assert_eq!(
12048            feed.subscriber_count(),
12049            0,
12050            "the lagged subscriber must be removed"
12051        );
12052
12053        let mut data = Vec::new();
12054        let mut last = None;
12055        loop {
12056            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
12057                .await
12058                .expect("the forwarder must finish once the subscriber is dropped");
12059            let Some(outbound) = next else { break };
12060            let frame = outbound.frame;
12061            if frame.header.ty == FrameType::StreamData {
12062                assert!(last.is_none(), "no data may follow the terminal frame");
12063                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
12064                data.push(event.cursor.seq);
12065            } else {
12066                assert!(last.is_none(), "exactly one terminal frame");
12067                last = Some(frame);
12068            }
12069        }
12070        assert!(!data.is_empty(), "queued frames drain before the terminal");
12071        for pair in data.windows(2) {
12072            assert_eq!(
12073                pair[1],
12074                pair[0] + 1,
12075                "queued frames arrive dense and in order"
12076            );
12077        }
12078        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
12079        assert_eq!(terminal.header.ty, FrameType::Error);
12080        assert_eq!(terminal.header.corr, 7);
12081        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
12082        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
12083        let detail = body.detail.expect("lagged error carries detail");
12084        assert_eq!(
12085            detail["first_undelivered_cursor"]["seq"],
12086            data.last().unwrap() + 1,
12087            "the named cursor is the first event the subscriber did not receive"
12088        );
12089        assert_eq!(
12090            detail["first_undelivered_cursor"]["daemon_incarnation"],
12091            "lag-incarnation"
12092        );
12093    }
12094}
12095
12096#[cfg(test)]
12097mod terminal_history_read_concurrency_tests {
12098    use super::*;
12099    use crate::terminal_journal::read_pause;
12100    use std::sync::mpsc as std_mpsc;
12101    use subc_test_support::TestTempDir;
12102
12103    fn journaled_ring(
12104        journal: &Arc<crate::terminal_journal::TerminalJournal>,
12105    ) -> Arc<Mutex<TerminalRing>> {
12106        Arc::new(Mutex::new(
12107            TerminalRing::new(TerminalRingConfig::default(), 1)
12108                .with_journal(Some(Arc::clone(journal))),
12109        ))
12110    }
12111
12112    fn crash(at_ms: u64) -> ExitReport {
12113        ExitReport {
12114            kind: ExitKind::Crash,
12115            code: Some(1),
12116            signal: None,
12117            at_ms,
12118        }
12119    }
12120
12121    /// Record an exit on another thread and report whether it finished within
12122    /// `bound`. The recorder thread is left running if it did not.
12123    fn record_within(
12124        module_id: &'static str,
12125        ring: &Arc<Mutex<TerminalRing>>,
12126        at_ms: u64,
12127        bound: Duration,
12128    ) -> bool {
12129        let ring = Arc::clone(ring);
12130        let (done, done_rx) = std_mpsc::channel();
12131        std::thread::spawn(move || {
12132            record_terminal(
12133                module_id,
12134                &ring,
12135                &SpawnEventFeed::default(),
12136                &crash(at_ms),
12137                TerminalDisposition::Restarting,
12138            );
12139            let _ = done.send(());
12140        });
12141        done_rx.recv_timeout(bound).is_ok()
12142    }
12143
12144    /// A history read in progress must not hold the journal writer (which every
12145    /// module's exit recording needs) or the module's own ring. Exits recorded
12146    /// while the read is paused complete promptly; the paused read answers as of
12147    /// the moment it started, and the next read has each exit exactly once.
12148    #[test]
12149    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12150        let dir = TestTempDir::new("terminal-history-concurrent-read");
12151        let path = dir.join("terminals.jsonl");
12152        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12153            path.clone(),
12154            "daemon".into(),
12155        ));
12156        let reader_ring = journaled_ring(&journal);
12157        let other_ring = journaled_ring(&journal);
12158        assert!(record_within(
12159            "reader-module",
12160            &reader_ring,
12161            10,
12162            Duration::from_secs(5)
12163        ));
12164
12165        let (started, release) = read_pause::install(&path);
12166        let reading = {
12167            let ring = Arc::clone(&reader_ring);
12168            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12169        };
12170        started
12171            .recv_timeout(Duration::from_secs(5))
12172            .expect("the history read reached its pause");
12173
12174        let bound = Duration::from_secs(1);
12175        assert!(
12176            record_within("other-module", &other_ring, 20, bound),
12177            "another module's exit waited on a history read (journal writer held)"
12178        );
12179        assert!(
12180            record_within("reader-module", &reader_ring, 30, bound),
12181            "the read module's own exit waited on its history read (ring held)"
12182        );
12183
12184        drop(release);
12185        let paused = reading.join().unwrap();
12186        assert_eq!(
12187            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12188            vec![10],
12189            "an exit recorded after the read began lands in neither half of it"
12190        );
12191        assert_eq!(paused.journal_skipped_lines, 0);
12192        assert_eq!(paused.journal_read_errors, 0);
12193
12194        let after = durable_terminal_history_of(&reader_ring, "reader-module");
12195        assert_eq!(
12196            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12197            vec![10, 30],
12198            "the next read merges ring and journal with no duplicate"
12199        );
12200        assert_eq!(after.journal_skipped_lines, 0);
12201    }
12202}
12203
12204/// What a restart does with the exited process's stderr reader. These drive
12205/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12206/// holds, so a reader that has not been scheduled by the bound is a controlled
12207/// input rather than something only a loaded machine produces.
12208#[cfg(test)]
12209mod stderr_settle_tests {
12210    use std::{
12211        future::Future,
12212        io,
12213        pin::Pin,
12214        sync::{Arc, Mutex},
12215        task::{Context, Poll},
12216        time::Duration,
12217    };
12218
12219    use tokio::{
12220        io::{AsyncRead, ReadBuf},
12221        sync::oneshot,
12222        time::Instant,
12223    };
12224
12225    use super::{settle_stderr_pump, StderrPump};
12226    use crate::stderr_tail::{
12227        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12228    };
12229
12230    const BOUND: Duration = Duration::from_millis(250);
12231
12232    /// Yields `before`, then stays pending until the gate is released, then
12233    /// yields `after` and reaches EOF. The bytes after the gate were written
12234    /// by a process that has already exited; only the reader is behind.
12235    struct HeldReader {
12236        before: Option<Vec<u8>>,
12237        gate: Option<oneshot::Receiver<()>>,
12238        after: io::Cursor<Vec<u8>>,
12239    }
12240
12241    impl AsyncRead for HeldReader {
12242        fn poll_read(
12243            mut self: Pin<&mut Self>,
12244            cx: &mut Context<'_>,
12245            buf: &mut ReadBuf<'_>,
12246        ) -> Poll<io::Result<()>> {
12247            if let Some(bytes) = self.before.take() {
12248                buf.put_slice(&bytes);
12249                return Poll::Ready(Ok(()));
12250            }
12251            if let Some(gate) = self.gate.as_mut() {
12252                match Pin::new(gate).poll(cx) {
12253                    Poll::Pending => return Poll::Pending,
12254                    Poll::Ready(_) => self.gate = None,
12255                }
12256            }
12257            Pin::new(&mut self.after).poll_read(cx, buf)
12258        }
12259    }
12260
12261    struct DiscardSink;
12262
12263    impl OutputSink for DiscardSink {
12264        fn write_line(&mut self, _line: &[u8]) {}
12265    }
12266
12267    fn line(text: &str) -> TailEntry {
12268        TailEntry::Line {
12269            text: text.to_string(),
12270            truncated: false,
12271            at_ms: None,
12272        }
12273    }
12274
12275    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
12276        ring.lock().unwrap()
12277    }
12278
12279    /// Start a reader for a new process generation that delivers `before`
12280    /// immediately and `after` only once the returned sender fires (or is
12281    /// dropped).
12282    fn held_pump(
12283        ring: &Arc<Mutex<StderrRing>>,
12284        before: &str,
12285        after: &str,
12286    ) -> (StderrPump, oneshot::Sender<()>) {
12287        let generation = lock(ring).begin_process();
12288        let (release, gate) = oneshot::channel();
12289        let reader = HeldReader {
12290            before: Some(before.as_bytes().to_vec()),
12291            gate: Some(gate),
12292            after: io::Cursor::new(after.as_bytes().to_vec()),
12293        };
12294        let task = tokio::spawn(pump_stderr_to(
12295            reader,
12296            Arc::clone(ring),
12297            generation,
12298            DiscardSink,
12299        ));
12300        (StderrPump { task, generation }, release)
12301    }
12302
12303    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
12304        for _ in 0..1000 {
12305            if done(&lock(ring)) {
12306                return;
12307            }
12308            tokio::time::sleep(Duration::from_millis(1)).await;
12309        }
12310        panic!(
12311            "ring never reached the expected state: {:?}",
12312            lock(ring).snapshot(None, None)
12313        );
12314    }
12315
12316    #[tokio::test(start_paused = true)]
12317    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
12318        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12319        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
12320
12321        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
12322        let before_release = lock(&ring).snapshot(None, None);
12323        assert!(
12324            matches!(before_release.capture, CaptureState::Incomplete { .. }),
12325            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
12326        );
12327
12328        // The restart: the next process starts and writes before the old
12329        // reader catches up.
12330        let next = lock(&ring).begin_process();
12331        lock(&ring).push_line_from(next, "next process booting");
12332        release.send(()).unwrap();
12333        wait_until(&ring, |ring| {
12334            ring.snapshot(None, None).capture == CaptureState::Captured
12335        })
12336        .await;
12337
12338        assert_eq!(
12339            untimed(lock(&ring).snapshot(None, None).entries),
12340            vec![
12341                line("booting"),
12342                line("config error: missing storage"),
12343                TailEntry::ProcessStart,
12344                line("next process booting"),
12345            ],
12346            "the crash's last line must survive a slow reader and stay in the crashed process's section"
12347        );
12348    }
12349
12350    #[tokio::test(start_paused = true)]
12351    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
12352    ) {
12353        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12354        // `_held` is never fired: a descendant keeps the pipe open for the
12355        // whole test.
12356        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
12357
12358        let started = Instant::now();
12359        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
12360        assert_eq!(
12361            started.elapsed(),
12362            BOUND,
12363            "the restart must wait exactly the bound for a pipe that stays open, no longer"
12364        );
12365
12366        let next = lock(&ring).begin_process();
12367        lock(&ring).push_line_from(next, "next process booting");
12368        tokio::time::sleep(Duration::from_secs(60)).await;
12369
12370        let snapshot = lock(&ring).snapshot(None, None);
12371        match &snapshot.capture {
12372            CaptureState::Incomplete { reason } => assert!(
12373                reason.contains("had not reached EOF") && reason.contains("250ms"),
12374                "the reason must say what is missing and after how long: {reason}"
12375            ),
12376            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
12377        }
12378        assert_eq!(
12379            untimed(snapshot.entries),
12380            vec![
12381                line("parent exiting"),
12382                TailEntry::ProcessStart,
12383                line("next process booting"),
12384            ]
12385        );
12386    }
12387
12388    #[tokio::test(start_paused = true)]
12389    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
12390        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12391        let (pump, release) = held_pump(&ring, "one\n", "two\n");
12392        release.send(()).unwrap();
12393
12394        settle_stderr_pump("clean", &ring, pump, BOUND).await;
12395
12396        let snapshot = lock(&ring).snapshot(None, None);
12397        assert_eq!(snapshot.capture, CaptureState::Captured);
12398        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
12399    }
12400}
12401
12402/// Containment of a module's process tree (issue #109).
12403///
12404/// The behaviour these defend against is a module helper surviving its module:
12405/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
12406/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
12407/// compounds it.
12408///
12409/// They run against the SUPERVISOR rather than the job-object crate because the
12410/// claim is about teardown: a crate-level test proves a job can reap a tree, not
12411/// that the daemon's drain path reaches it.
12412///
12413/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
12414/// lane there is a separate containment path with its own tests.
12415#[cfg(all(test, windows))]
12416mod job_containment_tests {
12417    use super::*;
12418    use std::{
12419        path::{Path, PathBuf},
12420        sync::{Arc, Mutex},
12421        time::{Duration, Instant},
12422    };
12423    use subc_test_support::TestTempDir;
12424
12425    /// The stub, expected beside this test executable.
12426    ///
12427    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
12428    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
12429    /// failure then reads as a broken test rather than an unbuilt dependency.
12430    fn stub_path() -> PathBuf {
12431        let mut path = std::env::current_exe().expect("current_exe available in tests");
12432        path.pop();
12433        path.pop();
12434        path.push("fake-aft-stub.exe");
12435        assert!(
12436            path.exists(),
12437            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
12438             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
12439            path.display()
12440        );
12441        path
12442    }
12443
12444    /// Poll for the grandchild pid the stub records, and parse it.
12445    fn read_grandchild_pid(path: &Path) -> u32 {
12446        let deadline = Instant::now() + Duration::from_secs(10);
12447        loop {
12448            if let Ok(contents) = std::fs::read_to_string(path) {
12449                if let Ok(pid) = contents.trim().parse() {
12450                    return pid;
12451                }
12452            }
12453            assert!(
12454                Instant::now() < deadline,
12455                "the stub never recorded a grandchild pid at {}",
12456                path.display()
12457            );
12458            std::thread::sleep(Duration::from_millis(10));
12459        }
12460    }
12461
12462    /// Everything one fixture run needs, so the two tests below differ in exactly
12463    /// one place: whether the child is contained.
12464    struct Fixture {
12465        _dir: TestTempDir,
12466        module_id: String,
12467        grandchild: u32,
12468        child: Option<SupervisedChild>,
12469        registry: Arc<Registry>,
12470        snapshot: Arc<Mutex<SupervisorSnapshot>>,
12471        terminal_ring: Arc<Mutex<TerminalRing>>,
12472        spawn_events: SpawnEventFeed,
12473    }
12474
12475    fn fixture(label: &str, module_id: &str) -> Fixture {
12476        let dir = TestTempDir::new(label);
12477        let pid_file = dir.join("grandchild.pid");
12478        let supervisor = Supervisor::new_for_test(
12479            Arc::new(Registry::default()),
12480            RestartPolicy::new(3, Duration::ZERO),
12481        );
12482        let runtime = supervisor.runtime_config();
12483        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12484        let spec = ModuleSpec {
12485            module_id: module_id.to_string(),
12486            program: stub_path(),
12487            // Zero args deliberately: a `--subc` argument would make the stub dial
12488            // a daemon that is not there, and the failure would land in the same
12489            // stderr ring this fixture exists to keep quiet.
12490            args: Vec::new(),
12491            env: vec![
12492                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
12493                (
12494                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
12495                    pid_file.display().to_string(),
12496                ),
12497            ],
12498            reserved: false,
12499            reserved_prefixes: Vec::new(),
12500            protocol: ModuleProtocol::Subc,
12501            overlap: Default::default(),
12502        };
12503        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
12504            .expect("spawn the supervised fixture");
12505        let grandchild = read_grandchild_pid(&pid_file);
12506        Fixture {
12507            _dir: dir,
12508            module_id: module_id.to_string(),
12509            grandchild,
12510            child: Some(child),
12511            registry: Arc::new(Registry::default()),
12512            snapshot,
12513            terminal_ring: Arc::clone(&runtime.terminal_ring),
12514            spawn_events: SpawnEventFeed::default(),
12515        }
12516    }
12517
12518    impl Fixture {
12519        /// Drain through the supervisor's own teardown path.
12520        async fn drain(&mut self) {
12521            let child = self
12522                .child
12523                .take()
12524                .expect("the fixture child is still present");
12525            drain_child_to_state(
12526                &self.module_id,
12527                ModuleProtocol::Subc,
12528                // No forwarding table in this fixture, so nothing reaches the
12529                // child over a connection.
12530                StopNotice::NotSent,
12531                &self.registry,
12532                None,
12533                &self.snapshot,
12534                &self.terminal_ring,
12535                &self.spawn_events,
12536                child,
12537                Duration::from_millis(500),
12538                ModuleState::Stopped,
12539                Some(false),
12540            )
12541            .await
12542            .expect("drain the supervised fixture");
12543        }
12544    }
12545
12546    /// Teardown reaps the grandchild, not merely the direct child.
12547    ///
12548    /// This is the assertion the change exists for. Before containment the
12549    /// grandchild survived: it is a separate process, and `start_kill` is
12550    /// `TerminateProcess` scoped to one pid.
12551    #[tokio::test]
12552    async fn teardown_reaps_the_grandchild() {
12553        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
12554        let grandchild = fixture.grandchild;
12555
12556        assert!(
12557            subc_jobobject::process_exists(grandchild),
12558            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
12559        );
12560
12561        fixture.drain().await;
12562
12563        assert!(
12564            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12565            "grandchild {grandchild} outlived module teardown: the tree was not contained"
12566        );
12567    }
12568
12569    /// The mutation control: with containment withheld, the grandchild survives
12570    /// the same kill.
12571    ///
12572    /// This is the defect reproduction from #109 — a direct-child kill reaches
12573    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
12574    /// supervisor because `spawn_and_mark_running` now always contains on
12575    /// Windows, which is the point: there is no longer a path that spawns
12576    /// uncontained, so the control has to construct one.
12577    ///
12578    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
12579    /// grandchild ever dies here, that test is passing for a reason unrelated to
12580    /// the job object and the containment claim is unproven.
12581    #[test]
12582    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12583        let dir = TestTempDir::new("teardown-uncontained");
12584        let pid_file = dir.join("grandchild.pid");
12585        let mut child = std::process::Command::new(stub_path())
12586            .env("FAKE_AFT_NEVER_CONNECT", "1")
12587            .env(
12588                "FAKE_AFT_GRANDCHILD_PID_FILE",
12589                pid_file.display().to_string(),
12590            )
12591            .stdin(std::process::Stdio::null())
12592            .stdout(std::process::Stdio::null())
12593            .stderr(std::process::Stdio::null())
12594            .spawn()
12595            .expect("spawn the uncontained fixture");
12596        let grandchild = read_grandchild_pid(&pid_file);
12597
12598        // Exactly what the pre-fix teardown did: kill the direct child.
12599        child.kill().expect("kill the direct child");
12600        let _ = child.wait();
12601
12602        assert!(
12603            subc_jobobject::process_exists(grandchild),
12604            "grandchild {grandchild} died with the direct child, so this control no longer \
12605             distinguishes contained from uncontained teardown and the regression test is \
12606             passing vacuously"
12607        );
12608
12609        // The orphan this control demonstrates is the leak the fix prevents, so
12610        // the control must not leave one behind.
12611        kill_tree(grandchild);
12612    }
12613
12614    /// Crash durability: closing the containment handle reaps the tree with no
12615    /// teardown code running at all.
12616    ///
12617    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
12618    /// call anything — and it is why containment is a kernel property of the
12619    /// handle rather than a step in the drain. Discovered by getting the
12620    /// mutation control wrong: clearing `job` to "disable" containment instead
12621    /// killed the tree, which is the guarantee, not a mistake.
12622    #[tokio::test]
12623    async fn dropping_containment_reaps_the_grandchild() {
12624        let mut fixture = fixture("drop-containment", "tree-drop");
12625        let grandchild = fixture.grandchild;
12626
12627        assert!(subc_jobobject::process_exists(grandchild));
12628
12629        // No `drain` call, no kill: dropping the handle is the entire mechanism.
12630        fixture.child.as_mut().expect("child present").job = None;
12631
12632        assert!(
12633            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12634            "grandchild {grandchild} survived the containment handle closing, so a daemon \
12635             crash would leave the tree behind"
12636        );
12637    }
12638
12639    /// Kill a pid and its tree, then confirm it is gone.
12640    fn kill_tree(pid: u32) {
12641        let _ = std::process::Command::new("taskkill.exe")
12642            .args(["/PID", &pid.to_string(), "/T", "/F"])
12643            .stdin(std::process::Stdio::null())
12644            .stdout(std::process::Stdio::null())
12645            .stderr(std::process::Stdio::null())
12646            .status();
12647        assert!(
12648            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12649            "could not clean up grandchild {pid}"
12650        );
12651    }
12652}
12653
12654#[cfg(test)]
12655mod privacy_trampoline_configuration_tests {
12656    #[tokio::test]
12657    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12658    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
12659        #[cfg(target_os = "macos")]
12660        {
12661            let supervisor = super::Supervisor::new(
12662                std::sync::Arc::new(crate::Registry::default()),
12663                super::RestartPolicy::default(),
12664            );
12665            let error = supervisor.spawn(spec()).unwrap_err();
12666            assert!(
12667                error
12668                    .to_string()
12669                    .contains("no privacy trampoline configured"),
12670                "{error}"
12671            );
12672        }
12673    }
12674
12675    #[tokio::test]
12676    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12677    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
12678        #[cfg(target_os = "macos")]
12679        {
12680            let supervisor = super::Supervisor::new(
12681                std::sync::Arc::new(crate::Registry::default()),
12682                super::RestartPolicy::default(),
12683            )
12684            .with_privacy_trampoline(std::env::current_exe().unwrap());
12685            let error = supervisor.spawn(spec()).unwrap_err();
12686            assert!(
12687                error
12688                    .to_string()
12689                    .contains("binary does not implement the privacy trampoline protocol"),
12690                "{error}"
12691            );
12692        }
12693    }
12694
12695    #[cfg(target_os = "macos")]
12696    fn spec() -> super::ModuleSpec {
12697        super::ModuleSpec {
12698            module_id: "privacy-configuration".into(),
12699            program: "/bin/sleep".into(),
12700            args: vec!["30".into()],
12701            env: vec![],
12702            reserved: false,
12703            reserved_prefixes: vec![],
12704            protocol: subc_control::ModuleProtocol::None,
12705            overlap: super::ModuleOverlap::Exclusive,
12706        }
12707    }
12708}
12709
12710#[cfg(test)]
12711mod privacy_exec_boundary_tests {
12712    #[cfg(target_os = "macos")]
12713    use super::*;
12714    #[cfg(target_os = "macos")]
12715    use std::{
12716        io::{Read, Write},
12717        net::{TcpListener, TcpStream},
12718    };
12719
12720    /// Unit-test-only pause at the actual early image read, not at a later
12721    /// status read. Production supervisors never inspect this environment key.
12722    #[cfg(target_os = "macos")]
12723    pub(super) fn before_image_sample(spec: &ModuleSpec, pid: u32) {
12724        if let Some((_, path)) = spec
12725            .env
12726            .iter()
12727            .find(|(key, _)| key == "SUBC_TEST_PRIVACY_SAMPLE_BARRIER")
12728        {
12729            let mut barrier = TcpStream::connect(path).unwrap();
12730            barrier
12731                .set_read_timeout(Some(Duration::from_secs(30)))
12732                .unwrap();
12733            barrier.write_all(&pid.to_ne_bytes()).unwrap();
12734            let mut release = [0];
12735            barrier.read_exact(&mut release).unwrap();
12736            assert_eq!(&release, b"X");
12737        }
12738    }
12739
12740    #[cfg(target_os = "macos")]
12741    fn spec(program: &str, args: &[&str]) -> ModuleSpec {
12742        ModuleSpec {
12743            module_id: "privacy-boundary".into(),
12744            program: program.into(),
12745            args: args.iter().map(|arg| (*arg).into()).collect(),
12746            env: vec![],
12747            reserved: false,
12748            reserved_prefixes: vec![],
12749            protocol: ModuleProtocol::None,
12750            overlap: ModuleOverlap::Exclusive,
12751        }
12752    }
12753
12754    #[cfg(target_os = "macos")]
12755    async fn accept(listener: TcpListener) -> TcpStream {
12756        // Socket readiness, not elapsed time, establishes both pause points.
12757        let listener = tokio::net::TcpListener::from_std({
12758            listener.set_nonblocking(true).unwrap();
12759            listener
12760        })
12761        .unwrap();
12762        let (stream, _) = tokio::time::timeout(Duration::from_secs(30), listener.accept())
12763            .await
12764            .unwrap()
12765            .unwrap();
12766        let stream = stream.into_std().unwrap();
12767        stream.set_nonblocking(false).unwrap();
12768        stream
12769            .set_read_timeout(Some(Duration::from_secs(30)))
12770            .unwrap();
12771        stream
12772    }
12773
12774    #[tokio::test]
12775    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12776    async fn macos_roster_withholds_a_nonnull_trampoline_image_until_exec_confirmation() {
12777        #[cfg(target_os = "macos")]
12778        {
12779            let root = subc_test_support::TestTempDir::new("privacy-roster-barrier");
12780            // Loopback sockets also work when the replay adapter's TMPDIR is
12781            // longer than Darwin's Unix-domain socket path limit.
12782            let exec_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12783            let sample_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12784            let record = root.join("live-children.json");
12785            let supervisor =
12786                Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
12787                    .with_live_children_record(&record);
12788            let runtime = supervisor.runtime_config();
12789            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12790            let mut spec = spec("/bin/sleep", &["30"]);
12791            spec.env = vec![
12792                (
12793                    "SUBC_TEST_PRIVACY_EXEC_BARRIER".into(),
12794                    exec_listener.local_addr().unwrap().to_string(),
12795                ),
12796                (
12797                    "SUBC_TEST_PRIVACY_SAMPLE_BARRIER".into(),
12798                    sample_listener.local_addr().unwrap().to_string(),
12799                ),
12800            ];
12801            let spawn = tokio::task::spawn_blocking(move || {
12802                spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap()
12803            });
12804            let mut sample = accept(sample_listener).await;
12805            let mut pid = [0; 4];
12806            sample.read_exact(&mut pid).unwrap();
12807            let pid = u32::from_ne_bytes(pid);
12808            let mut exec = accept(exec_listener).await;
12809            let mut ready = [0];
12810            exec.read_exact(&mut ready).unwrap();
12811            assert_eq!(&ready, b"R");
12812            // The early read is guaranteed to see a real, non-null trampoline
12813            // image: the fixture has reached its barrier and cannot exec yet.
12814            let trampoline = subc_os::file_identity(&test_privacy_trampoline()).unwrap();
12815            assert_eq!(
12816                observe_spawned_image(pid).unwrap().executable,
12817                Some(trampoline)
12818            );
12819            sample.write_all(b"X").unwrap();
12820            let mut child = spawn.await.unwrap();
12821            let early = crate::live_children::read_record(&record).unwrap();
12822            assert_eq!(early.len(), 1);
12823            assert_eq!(early[0].pid, pid);
12824            assert_eq!(
12825                early[0].executable, None,
12826                "unconfirmed trampoline image entered the roster"
12827            );
12828            assert!(child.report_ready.get().is_none());
12829            // The barrier's duration is unrelated to the production five-second
12830            // exec budget. Start the test's confirmation budget upon release.
12831            child.privacy_exec.as_mut().unwrap().deadline =
12832                tokio::time::Instant::now() + Duration::from_secs(30);
12833            exec.write_all(b"X").unwrap();
12834            child.confirm_privacy_exec().await;
12835            assert_eq!(child.spawn_failure, None);
12836            assert!(child.report_ready.get().is_some());
12837            let confirmed = crate::live_children::read_record(&record).unwrap();
12838            let module = subc_os::file_identity(std::path::Path::new("/bin/sleep")).unwrap();
12839            assert_ne!(module, trampoline);
12840            assert_eq!(confirmed[0].executable, Some(module.into()));
12841            child.start_kill().unwrap();
12842            child.wait().await.unwrap();
12843            child.release_roster();
12844        }
12845    }
12846
12847    #[tokio::test]
12848    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12849    async fn macos_already_exited_121_in_try_wait_is_a_module_exit_not_a_trampoline_refusal() {
12850        #[cfg(target_os = "macos")]
12851        {
12852            let registry = Arc::new(Registry::default());
12853            let policy = RestartPolicy::new(0, Duration::ZERO);
12854            let supervisor = Supervisor::new_for_test(Arc::clone(&registry), policy);
12855            let runtime = supervisor.runtime_config();
12856            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12857            let spec = spec("/bin/sh", &["-c", "exit 121"]);
12858            // Drive spawn and confirmation separately instead of starting the
12859            // monitor. WNOWAIT observes a real exit without consuming its status,
12860            // so confirmation's first try_wait must take the already-exited arm.
12861            let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
12862            let pid = child.pid;
12863            tokio::task::spawn_blocking(move || {
12864                subc_os::fork_exec_test::wait_for_child_exit_without_reaping(pid)
12865            })
12866            .await
12867            .unwrap()
12868            .unwrap();
12869            child.privacy_exec.as_mut().unwrap().deadline =
12870                tokio::time::Instant::now() + Duration::from_secs(30);
12871            let status = child.wait().await.unwrap();
12872            assert_eq!(status.code(), Some(121));
12873            assert!(child.privacy_exec.is_none());
12874            assert!(
12875                child.report_ready.get().is_none(),
12876                "an exited module must not publish a live pid"
12877            );
12878            let report = classify_reaped_child_exit(&snapshot, &child, &status);
12879            on_child_exit(
12880                &spec,
12881                policy,
12882                &registry,
12883                &snapshot,
12884                &runtime.terminal_ring,
12885                &runtime.spawn_events,
12886                &runtime.child_roster,
12887                report,
12888            )
12889            .await;
12890            let state = lock_snapshot(&snapshot).unwrap();
12891            assert_eq!(state.state, ModuleState::Failed);
12892            assert_eq!(state.last_exit.as_ref().unwrap().code, Some(121));
12893            assert_eq!(state.reported_pid(), None);
12894            drop(state);
12895            let history = runtime.terminal_ring.lock().unwrap().snapshot();
12896            assert_eq!(history.entries.len(), 1);
12897            let terminal = &history.entries[0];
12898            assert_eq!(terminal.exit_code, Some(121));
12899            assert_eq!(terminal.exit_kind, subc_control::TerminalExitKind::Crash);
12900            assert_eq!(terminal.disposition, TerminalDisposition::Failed);
12901            assert_eq!(
12902                terminal.disposition_detail.as_deref(),
12903                Some(policy.budget_exhausted_detail().as_str()),
12904                "module exit 121 was classified as a trampoline refusal: {terminal:?}"
12905            );
12906            assert_eq!(child.spawn_failure, None);
12907            child.release_roster();
12908        }
12909    }
12910}
12911
12912/// The daemon's real spawn path hands a subc-wire child its launch nonce on
12913/// descriptor 3, without an environment copy. The shell records the nonce
12914/// and its environment after exec so these tests observe the real handover.
12915#[cfg(all(test, unix))]
12916mod launch_nonce_descriptor_tests {
12917    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
12918    use crate::stderr_tail::{StderrRing, StderrTailConfig};
12919    use std::{
12920        path::PathBuf,
12921        sync::{Arc, Mutex},
12922        time::{Duration, Instant},
12923    };
12924    use subc_test_support::TestTempDir;
12925
12926    async fn probe(role: super::SpawnRole) {
12927        let scratch = TestTempDir::new("launch-nonce-descriptor");
12928        let fd_copy = scratch.join("from-descriptor");
12929        let env_copy = scratch.join("environment");
12930        let script = format!(
12931            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
12932            fd = fd_copy.display(), env = env_copy.display(),
12933        );
12934        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
12935        let spec = ModuleSpec {
12936            module_id: "nonce-descriptor-probe".to_string(),
12937            program: PathBuf::from("/bin/sh"),
12938            args: vec!["-c".to_string(), script],
12939            env: vec![
12940                xdg("XDG_DATA_HOME"),
12941                xdg("XDG_RUNTIME_DIR"),
12942                xdg("XDG_CONFIG_HOME"),
12943                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
12944            ],
12945            reserved: true,
12946            reserved_prefixes: Vec::new(),
12947            protocol: ModuleProtocol::Subc,
12948            overlap: Default::default(),
12949        };
12950        let handle = SupervisorHandle::new();
12951        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12952        let roster = ChildRoster::default();
12953        #[cfg(target_os = "macos")]
12954        {
12955            let path = super::test_privacy_trampoline();
12956            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
12957        }
12958        let child = super::spawn_child_in_slot(
12959            &spec,
12960            None,
12961            Some(&handle),
12962            &ring,
12963            None,
12964            &roster,
12965            #[cfg(target_os = "linux")]
12966            None,
12967            role,
12968            matches!(role, super::SpawnRole::SwapCandidate),
12969        )
12970        .expect("spawn probe");
12971        let deadline = Instant::now() + Duration::from_secs(10);
12972        while !(fd_copy.exists() && env_copy.exists()) {
12973            assert!(Instant::now() < deadline, "probe never wrote its copies");
12974            tokio::time::sleep(Duration::from_millis(20)).await;
12975        }
12976        let nonce = std::fs::read_to_string(fd_copy).unwrap();
12977        assert!(!nonce.is_empty());
12978        let environment = std::fs::read_to_string(env_copy).unwrap();
12979        assert!(environment
12980            .lines()
12981            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
12982        let copy = environment
12983            .lines()
12984            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
12985        assert_eq!(
12986            copy, None,
12987            "Unix children must never receive the environment nonce"
12988        );
12989        if matches!(role, super::SpawnRole::Plain) {
12990            assert_eq!(
12991                handle.spawn_nonce(&spec.module_id).as_deref(),
12992                Some(nonce.as_str())
12993            );
12994        }
12995        drop(child);
12996    }
12997
12998    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12999    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
13000        probe(super::SpawnRole::Plain).await;
13001    }
13002
13003    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13004    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
13005        probe(super::SpawnRole::SwapCandidate).await;
13006    }
13007}
13008
13009#[cfg(all(test, target_os = "linux"))]
13010mod cgroup_containment_tests {
13011    use super::*;
13012    use subc_test_support::TestTempDir;
13013
13014    fn running(pid: u32) -> bool {
13015        // An orphan can remain a zombie until the container init reaps it.
13016        std::fs::read_to_string(format!("/proc/{pid}/stat"))
13017            .ok()
13018            .and_then(|stat| {
13019                stat.rsplit_once(") ")
13020                    .map(|(_, rest)| rest.starts_with('Z'))
13021            })
13022            .is_some_and(|zombie| !zombie)
13023    }
13024
13025    #[tokio::test]
13026    async fn linux_teardown_reaps_the_grandchild() {
13027        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
13028    }
13029
13030    #[tokio::test]
13031    async fn linux_shutdown_straggler_reaps_the_grandchild() {
13032        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
13033    }
13034
13035    async fn teardown_tree(test_name: &str, shutdown: bool) {
13036        let dir = TestTempDir::new(test_name);
13037        let root = PathBuf::from(format!(
13038            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
13039            std::process::id(),
13040            unix_ms_now()
13041        ));
13042        if let Err(error) = std::fs::create_dir(&root) {
13043            assert!(
13044                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13045                "required cgroup test cannot execute: {error}"
13046            );
13047            eprintln!(
13048                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
13049                root.display()
13050            );
13051            return;
13052        }
13053        let placement = subc_cgroup::prepare_at(&root)
13054            .expect("prepare isolated kernel cgroup")
13055            .expect("isolated cgroup is delegated");
13056        let module_id = "tree-teardown";
13057        let module = placement
13058            .module_path(module_id)
13059            .expect("create isolated module cgroup");
13060        if !module.join("cgroup.kill").exists() {
13061            std::fs::remove_dir(&module).unwrap();
13062            std::fs::remove_dir(root.join("subc-modules")).unwrap();
13063            std::fs::remove_dir(&root).unwrap();
13064            assert!(
13065                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13066                "required cgroup.kill interface unavailable"
13067            );
13068            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
13069            return;
13070        }
13071        let supervisor = Supervisor::new_for_test(
13072            Arc::new(Registry::default()),
13073            RestartPolicy::new(3, Duration::ZERO),
13074        )
13075        .with_cgroup_placement(Some(placement));
13076        let mut runtime = supervisor.runtime_config();
13077        runtime.child_roster = runtime
13078            .child_roster
13079            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
13080        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13081        let pid_file = dir.join("grandchild.pid");
13082        let spec = ModuleSpec {
13083            module_id: module_id.to_string(),
13084            program: PathBuf::from("/bin/sh"),
13085            args: vec![
13086                "-c".into(),
13087                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
13088                "fixture".into(),
13089                pid_file.display().to_string(),
13090            ],
13091            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
13092                .into_iter()
13093                .map(|key| (key.to_string(), dir.display().to_string()))
13094                .collect(),
13095            reserved: false,
13096            reserved_prefixes: Vec::new(),
13097            protocol: ModuleProtocol::None,
13098            overlap: Default::default(),
13099        };
13100        let child =
13101            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
13102        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
13103        let grandchild: u32 = loop {
13104            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
13105                if let Ok(pid) = contents.trim().parse() {
13106                    break pid;
13107                }
13108            }
13109            assert!(
13110                tokio::time::Instant::now() < deadline,
13111                "grandchild pid was not recorded"
13112            );
13113            tokio::time::sleep(Duration::from_millis(10)).await;
13114        };
13115        assert!(
13116            running(grandchild),
13117            "grandchild must be alive before teardown"
13118        );
13119        if shutdown {
13120            let mut child = child;
13121            crate::child_roster::end_children_for_daemon_shutdown(
13122                &runtime.child_roster,
13123                false,
13124                std::future::pending(),
13125            )
13126            .await;
13127            child.wait().await.expect("reap shutdown straggler");
13128        } else {
13129            drain_child_to_state(
13130                module_id,
13131                ModuleProtocol::None,
13132                StopNotice::NotSent,
13133                &Registry::default(),
13134                None,
13135                &snapshot,
13136                &runtime.terminal_ring,
13137                &SpawnEventFeed::default(),
13138                child,
13139                Duration::from_millis(100),
13140                ModuleState::Stopped,
13141                Some(false),
13142            )
13143            .await
13144            .expect("real supervisor teardown");
13145        }
13146        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
13147        while running(grandchild) && tokio::time::Instant::now() < deadline {
13148            tokio::time::sleep(Duration::from_millis(10)).await;
13149        }
13150        let survived = running(grandchild);
13151        // Kill a surviving grandchild so a failed test does not leave it behind.
13152        if survived {
13153            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
13154            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
13155            tokio::time::sleep(Duration::from_millis(100)).await;
13156        }
13157        if module.exists() {
13158            std::fs::remove_dir(&module).expect("remove empty module cgroup");
13159        }
13160        std::fs::remove_dir(root.join("subc-modules")).unwrap();
13161        std::fs::remove_dir(&root).unwrap();
13162        assert!(
13163            !survived,
13164            "grandchild {grandchild} outlived module teardown"
13165        );
13166        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
13167    }
13168}