Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    let path = std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture");
134    // Without the fixture every macOS spawn is refused, and the tests that
135    // spawn fail later as a module in state Failed, which names the wrong
136    // cause. `cargo test -p subc-daemon --lib` alone does not build it.
137    assert!(
138        path.exists(),
139        "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140         --bins --features test-support` or `cargo test -p subc-daemon` first",
141        path.display()
142    );
143    path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148    use std::io::Read;
149    let mut probe = std::process::Command::new(path)
150        .args(["__disclaim-exec", "--probe"])
151        .stdin(Stdio::null())
152        .stdout(Stdio::piped())
153        .stderr(Stdio::piped())
154        .spawn()
155        .map_err(|error| {
156            format!(
157                "privacy trampoline probe failed for {}: {error}",
158                path.display()
159            )
160        })?;
161    let deadline = std::time::Instant::now() + Duration::from_secs(5);
162    let status = loop {
163        match probe.try_wait() {
164            Ok(Some(status)) => break status,
165            Ok(None) if std::time::Instant::now() < deadline => {
166                std::thread::sleep(Duration::from_millis(5))
167            }
168            result => {
169                let _ = probe.kill();
170                let _ = probe.wait();
171                return Err(format!(
172                    "privacy trampoline probe failed or timed out for {}: {result:?}",
173                    path.display()
174                ));
175            }
176        }
177    };
178    let mut answer = String::new();
179    if let Some(stdout) = probe.stdout.take() {
180        let _ = stdout.take(256).read_to_string(&mut answer);
181    }
182    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183        return Ok(());
184    }
185    let mut diagnostic = String::new();
186    if let Some(stderr) = probe.stderr.take() {
187        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188    }
189    let cause = diagnostic
190        .trim()
191        .strip_prefix("ck-subc: own privacy identity refused: ")
192        .unwrap_or("binary does not implement the privacy trampoline protocol");
193    Err(format!(
194        "{cause}: probe of {} exited {status}",
195        path.display()
196    ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201    spec: &ModuleSpec,
202    roster: &ChildRoster,
203) -> Result<
204    (
205        Command,
206        Option<PrivacyExec>,
207        subc_os::privacy_identity::ExecAcknowledgement,
208    ),
209    SuperviseError,
210> {
211    let failure = |cause: String| {
212        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213        SuperviseError::Spawn {
214            program: spec.program.clone(),
215            source: io::Error::other(cause),
216            cgroup_path: None,
217        }
218    };
219    let trampoline = roster.privacy_trampoline().map_err(failure)?;
220    // Resolve PATH with the same environment the Command will receive. For
221    // scripts retain the existing orphan-identity rule: the kernel chooses
222    // the interpreter, and its observed image is the one recorded. Do not
223    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
224    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225        let path = spec
226            .env
227            .iter()
228            .find(|(key, _)| key == "PATH")
229            .map(|(_, value)| std::ffi::OsString::from(value))
230            .or_else(|| std::env::var_os("PATH"))
231            .unwrap_or_else(|| "/usr/bin:/bin".into());
232        std::env::split_paths(&path)
233            .map(|dir| dir.join(&spec.program))
234            .find(|path| path.is_file())
235            .unwrap_or_else(|| spec.program.clone())
236    } else {
237        spec.program.clone()
238    };
239    let expected = subc_os::file_identity(&program);
240    let trampoline_image = subc_os::file_identity(&trampoline);
241    let script = {
242        use std::io::Read;
243        let mut prefix = [0u8; 2];
244        std::fs::File::open(&program)
245            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246    };
247    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248        return Err(failure(
249            "privacy identity module executable is missing or is the trampoline itself".to_string(),
250        ));
251    }
252    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253        .map_err(|error| failure(error.to_string()))?;
254    let reader =
255        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256    let mut command = Command::new(&trampoline);
257    command
258        .arg("__disclaim-exec")
259        .arg(ack.fd().to_string())
260        .arg(&program);
261    ack.install(command.as_std_mut());
262    Ok((
263        command,
264        Some(PrivacyExec {
265            reader,
266            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267            expected,
268            trampoline: trampoline_image,
269            script,
270            module_id: spec.module_id.clone(),
271        }),
272        ack,
273    ))
274}
275
276struct SupervisedChild {
277    child: Child,
278    #[cfg(target_os = "macos")]
279    privacy_exec: Option<PrivacyExec>,
280    /// Set once this launch's exec acknowledgement confirms the module image.
281    /// On macOS the pid first runs the `ck-subc` launch trampoline (see
282    /// `subc_os::privacy_identity`), which then replaces itself with the
283    /// module. The supervisor owns and can kill that pid from spawn, but
284    /// status readers report it only after this latch is set, so nothing
285    /// reports the trampoline's image as the module's.
286    #[cfg(target_os = "macos")]
287    report_ready: Arc<OnceLock<()>>,
288    /// Refusal before the module image was accepted, retained for terminal records.
289    spawn_failure: Option<String>,
290    /// The protocol this process was launched with. A reload can store a new
291    /// launch spec with a different protocol, but that takes effect only at the
292    /// next spawn, so this process keeps being handled by the protocol it
293    /// actually speaks.
294    protocol: ModuleProtocol,
295    /// This process's cgroup name: a bounded module/slot label followed by a
296    /// spawn suffix unique to this process (when cgroup placement is on). A
297    /// retired process in a slot may still be draining when a later one is
298    /// spawned into that slot, so the suffix keeps the later process out of
299    /// the retired one's cgroup, which is the domain a kill applies to.
300    #[cfg(target_os = "linux")]
301    module_id: String,
302    #[cfg(target_os = "linux")]
303    cgroup_placement: Option<subc_cgroup::Placement>,
304    /// The job that contains this child and every process it spawns (issue #109).
305    ///
306    /// Dropping this handle is what reaps a surviving tree when no supervisor
307    /// code runs — a daemon crash — because the job carries
308    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
309    ///
310    /// That limit is not crash-only, and the difference is worth knowing: a
311    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
312    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
313    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
314    /// module at once. Before this change they survived that, saw EOF on the
315    /// control socket, and ran their own teardown; Unix keeps that path
316    /// deliberately, so a module can seal a WAL or close a capture rather than
317    /// be killed mid-write. So this trades graceful teardown on every Windows
318    /// daemon stop for containment on a crash, which is the right way round
319    /// today: orphaned GPU workers are a reported, recurring problem, and the
320    /// modules that write most heavily do not run on Windows.
321    ///
322    /// The fix is a real Windows stop path — the daemon draining before it
323    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
324    /// reaches only what the drain left behind, which is what it should reach.
325    #[cfg(windows)]
326    job: Option<subc_jobobject::JobObject>,
327    stdout_pump: Option<JoinHandle<()>>,
328    stderr_pump: Option<StderrPump>,
329    stderr_ring: Arc<Mutex<StderrRing>>,
330    spawned_at_ms: u64,
331    spawned_from: PathBuf,
332    spawned_file_identity: Option<SpawnedFileIdentity>,
333    process_start_time: Option<u64>,
334    process_identity: Option<ProcessIdentity>,
335    pid: u32,
336    /// This process's entry in the daemon's child roster, released when the
337    /// process is reaped or this handle is dropped.
338    roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342    fn id(&self) -> Option<u32> {
343        Some(self.pid)
344    }
345
346    fn process_identity(&self) -> Option<ProcessIdentity> {
347        self.process_identity
348    }
349
350    async fn wait(&mut self) -> io::Result<ExitStatus> {
351        #[cfg(target_os = "macos")]
352        self.confirm_privacy_exec().await;
353        // The roster entry is NOT released here. A daemon shutdown waits for the
354        // roster to empty and then exits the process, so releasing at the reap
355        // let it exit before the exit handler wrote this child's terminal record
356        // (the stderr drain and snapshot update sit in between), and the
357        // shutdown's own `daemon_shutdown` record was intermittently lost. The
358        // caller releases it after recording the exit (`release_roster`), and
359        // dropping the handle releases it too.
360        let result = self.child.wait().await;
361        #[cfg(target_os = "linux")]
362        if result.is_ok() {
363            if let Some(placement) = self.cgroup_placement.as_ref() {
364                cleanup_reaped_cgroup(placement, &self.module_id).await;
365                // Keep ownership while awaiting kernel population changes: a
366                // drain timeout may cancel this wait and then escalate/reap.
367                self.cgroup_placement = None;
368            }
369        }
370        result
371    }
372
373    #[cfg(target_os = "macos")]
374    async fn confirm_privacy_exec(&mut self) {
375        let Some(pending) = &mut self.privacy_exec else {
376            return;
377        };
378        let result = tokio::time::timeout_at(pending.deadline, async {
379            let mut record = Vec::new();
380            loop {
381                let mut ready = pending.reader.readable().await?;
382                let read = ready.try_io(|reader| {
383                    use std::io::Read;
384                    let mut reader = reader.get_ref();
385                    let mut buffer = [0u8; 256];
386                    reader.read(&mut buffer).map(|count| (count, buffer))
387                });
388                match read {
389                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390                    Ok(Ok((count, buffer))) => {
391                        if record.len() + count > 1024 {
392                            return Err(io::Error::other(
393                                "privacy exec refusal record is too long",
394                            ));
395                        }
396                        record.extend_from_slice(&buffer[..count]);
397                    }
398                    Ok(Err(error)) => return Err(error),
399                    Err(_) => continue,
400                }
401            }
402        })
403        .await;
404        // Keep the reader in self across await: select cancellation must not
405        // discard the handshake or reset its original five-second deadline.
406        let pending = self.privacy_exec.as_ref().expect("pending exec");
407        let cause = match result {
408            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409            Ok(Err(error)) => Some(format!(
410                "privacy identity exec acknowledgement failed: {error}"
411            )),
412            Ok(Ok(record)) if !record.is_empty() => Some(
413                std::str::from_utf8(&record)
414                    .ok()
415                    .and_then(|record| {
416                        record
417                            .trim()
418                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419                    })
420                    .filter(|cause| !cause.is_empty())
421                    .unwrap_or("invalid privacy exec refusal record")
422                    .to_string(),
423            ),
424            Ok(Ok(_)) => match self.child.try_wait() {
425                // Empty EOF is the exec acknowledgement. A real module may exit
426                // immediately, including with a reserved trampoline status; no
427                // image is admitted, and its ordinary exit contract stays intact.
428                Ok(Some(_status)) => None,
429                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430                Ok(None) => {
431                    let image = observe_spawned_image(self.pid);
432                    if let Some(image) = image.filter(|image| {
433                        image.executable.is_some()
434                            && image.executable != pending.trampoline
435                            && (image.executable == pending.expected || pending.script)
436                    }) {
437                        if let Some(guard) = &self.roster_guard {
438                            guard.confirm_executable(image);
439                        }
440                        let _ = self.report_ready.set(());
441                        info!(module_id = %pending.module_id, pid = self.pid,
442                            "module spawned with own privacy identity (responsibility disclaimed)");
443                        None
444                    } else if image.is_none()
445                        || image.is_some_and(|image| image.executable.is_none())
446                    {
447                        // A process can exit between try_wait and the kernel
448                        // image read. Empty EOF already acknowledged exec, so
449                        // preserve that module's ordinary exit rather than
450                        // mislabel a disappearing image as trampoline refusal.
451                        // Keep pending in self across await so cancellation does
452                        // not discard validation or reset its original deadline.
453                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454                            Ok(Ok(_status)) => None,
455                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457                        }
458                    } else {
459                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460                    }
461                }
462            },
463        };
464        let pending = self.privacy_exec.take().expect("pending exec");
465        if let Some(cause) = cause {
466            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467            self.spawn_failure = Some(cause);
468            // No image is admitted on failure. Reach the entire fresh process
469            // group, including a module which spawned a helper before refusal.
470            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472            }
473            let _ = self.child.start_kill();
474        }
475    }
476
477    /// Releases this child's daemon-shutdown roster entry once its exit has
478    /// been recorded. The pid is already reaped and free for reuse, so the
479    /// entry must not outlive the record any longer than that.
480    fn release_roster(&mut self) {
481        self.roster_guard = None;
482    }
483
484    /// Kill the child and, where containment is available, its process tree.
485    ///
486    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
487    /// helper process leaked the helper — the Synapse embedding module's CUDA
488    /// worker holds the GPU allocation, so the leak cost VRAM until the next
489    /// restart of something else. Terminating the job reaches grandchildren that
490    /// a tree walk cannot, including one whose parent has already exited and
491    /// been reparented away.
492    ///
493    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
494    /// direct-child kill still decides the outcome, so containment can never
495    /// change whether a module is reported as stopped.
496    fn start_kill(&mut self) -> io::Result<()> {
497        #[cfg(windows)]
498        if let Some(job) = &self.job {
499            if let Err(error) = job.terminate() {
500                debug!(
501                    error = %error,
502                    "job termination failed; the direct-child kill still owns the outcome"
503                );
504            }
505        }
506        #[cfg(target_os = "linux")]
507        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508        self.child.start_kill()
509    }
510
511    async fn drain_stderr(&mut self, module_id: &str) {
512        if let Some(mut pump) = self.stdout_pump.take() {
513            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514                Ok(Ok(())) => {}
515                Ok(Err(error)) => {
516                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517                }
518                Err(_) => {
519                    pump.abort();
520                    warn!(
521                        module_id,
522                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523                        "stdout pump did not drain before restart; stopped it before the next process"
524                    );
525                }
526            }
527        }
528
529        let Some(pump) = self.stderr_pump.take() else {
530            return;
531        };
532        settle_stderr_pump(
533            module_id,
534            &self.stderr_ring,
535            pump,
536            STDERR_PUMP_DRAIN_TIMEOUT,
537        )
538        .await;
539    }
540}
541
542/// The reader task for one process's stderr, with the ring generation its
543/// lines are attributed to.
544struct StderrPump {
545    task: JoinHandle<()>,
546    generation: u64,
547}
548
549/// Retire an exited process's stderr reader and wait up to `bound` for it to
550/// reach EOF. A reader still running at the bound is detached, not stopped: it
551/// keeps filling the exited process's section of the ring until its pipe
552/// closes, and the tail reads `Incomplete` until then. See
553/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
554async fn settle_stderr_pump(
555    module_id: &str,
556    ring: &Arc<Mutex<StderrRing>>,
557    pump: StderrPump,
558    bound: Duration,
559) {
560    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561    let StderrPump {
562        mut task,
563        generation,
564    } = pump;
565    lock().retire_pump(generation);
566    match timeout(bound, &mut task).await {
567        Ok(Ok(())) => {}
568        Ok(Err(err)) => {
569            let mut ring = lock();
570            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571            ring.finish_pump(generation);
572            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573        }
574        Err(_) => {
575            // Dropping the handle detaches the task; it ends at EOF on its pipe.
576            drop(task);
577            lock().mark_pump_late(
578                generation,
579                format!(
580                    "stderr of the exited process had not reached EOF {bound:?} after it was \
581                     retired (a descendant may still hold the pipe open); lines it still \
582                     writes are kept in that process's section"
583                ),
584            );
585            warn!(
586                module_id,
587                waited = ?bound,
588                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589            );
590        }
591    }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596    EVENTS.get_or_init(|| {
597        let (sender, _receiver) = watch::channel(0);
598        sender
599    })
600}
601
602pub(crate) fn notify_registration_release() {
603    let events = registration_release_events();
604    let next_generation = (*events.borrow()).wrapping_add(1);
605    events.send_replace(next_generation);
606}
607
608/// How to launch one singleton module process.
609#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611    pub module_id: String,
612    pub program: PathBuf,
613    pub args: Vec<String>,
614    pub env: Vec<(String, String)>,
615    /// When true this is a reserved module: each spawn gets a fresh one-time launch
616    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
617    /// process can register this module_id (a security-boundary module like the
618    /// credential vault must not be impersonable while it is down/restarting).
619    pub reserved: bool,
620    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
621    /// Prefixes come from daemon config and must end in `:` before they reach the
622    /// supervisor; the owner module's current spawn nonce authorizes claims under
623    /// each prefix.
624    pub reserved_prefixes: Vec<String>,
625    /// The wire protocol this module speaks, as DECLARED in daemon config.
626    ///
627    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
628    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
629    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
630    /// and NO launch nonce, and a clean exit the daemon did not request is
631    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
632    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
633    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
634    /// because a process ignores an environment variable it does not read.
635    ///
636    /// The argument is the part that cannot be "harmless to a process that
637    /// ignores it": a stock binary exits on an unknown flag before it listens
638    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
639    /// first conformance run against this mode found it. The nonce is withheld
640    /// because a process that will never present it gains nothing from holding
641    /// it, and a secret in the environment of a process that does not need it is
642    /// a leak surface for no benefit.
643    pub protocol: ModuleProtocol,
644    /// Whether two processes of this module may run at once, which is what a
645    /// blue/green swap does for the length of its overlap. Declared in daemon
646    /// config because the daemon must be able to answer it while the module is
647    /// down, and so a module cannot talk itself into it after registering.
648    pub overlap: ModuleOverlap,
649}
650
651/// Whether a module tolerates a second process of itself running alongside.
652///
653/// Most modules are single-writer on their store (a WAL, a capture log, a
654/// resident index behind a writer barrier), and two processes on one store
655/// corrupt it. So a swap, which overlaps the old and new process by design,
656/// is refused unless the module's config opts in with `overlap: "safe"`.
657#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659    /// Never run two processes of this module at once. The default.
660    #[default]
661    Exclusive,
662    /// The module has said a second process of itself is harmless for the
663    /// length of a swap.
664    ///
665    /// Declare it only if a second instance can run for a few seconds without
666    /// touching ANY single-writer store: every database, WAL, index, projector
667    /// and scheduled job the module owns. A lease on part of that state is not
668    /// enough. broca's session lease guards WAL appends while its run index, its
669    /// store projector and its archive fold timer (which unlinks live WAL files)
670    /// stay single-writer, so broca is exclusive despite holding a lease. The
671    /// refusal only fires after this has been decided, so the decision is the
672    /// check.
673    Safe,
674}
675
676impl ModuleOverlap {
677    pub fn as_str(self) -> &'static str {
678        match self {
679            Self::Exclusive => "exclusive",
680            Self::Safe => "safe",
681        }
682    }
683}
684
685/// Environment variable telling a spawned module which case it was started
686/// for, before it sends HELLO. Only a swap candidate carries it, as
687/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
688///
689/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
690/// longer because nobody waits on it, while a plain restart must flip ready
691/// quickly because callers see `module_warming` until it does. Absence means
692/// plain restart, the safe reading. The daemon trusts nothing about it; the
693/// candidate is proven by its launch nonce at HELLO.
694pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
696pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697/// How long a swap waits for its candidate to register and declare itself
698/// ready when the operator does not say. A module warming as a swap candidate
699/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
700/// daemon allows that plus time to start the process and send HELLO.
701pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703/// Bounded restart policy for crash exits.
704///
705/// `max_restarts` is the number of replacement processes allowed after the
706/// initial spawn WITHIN `window`. After that many crash restarts inside one
707/// window the module enters [`ModuleState::Failed`] and the supervisor stops
708/// the crash loop.
709///
710/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
711/// and that only survived because crashes were rare: a module that crashed
712/// three times across a week was disabled forever by crashes that had nothing
713/// to do with each other. That stopped being survivable once modules began
714/// exiting non-zero whenever the daemon's connection to them drops, because
715/// then every daemon-side connection drop spends a unit of the same budget and
716/// one flappy hour permanently stops a healthy module. Restarts older than
717/// `window` release their slot, so a module that crashed twice yesterday has a
718/// full budget today, while a genuine crash loop -- which is fast by
719/// definition -- still reaches the cap and stops.
720#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722    pub max_restarts: u32,
723    /// Base delay before a crash replacement. The actual delay escalates with
724    /// the number of recent crash replacements and is capped by `max_backoff`.
725    pub backoff: Duration,
726    /// Maximum delay before a crash replacement.
727    pub max_backoff: Duration,
728    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
729    /// budget effectively infinite (nothing is ever in-window), which is why
730    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
731    pub window: Duration,
732}
733
734impl RestartPolicy {
735    /// A policy with the default crash window. Callers that care about the
736    /// window say so with [`Self::with_window`]; the ones that do not are
737    /// asking for the standard rate limit, not for no limit.
738    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739        Self {
740            max_restarts,
741            backoff,
742            max_backoff: DEFAULT_MAX_BACKOFF,
743            window: DEFAULT_RESTART_WINDOW,
744        }
745    }
746
747    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748        self.max_backoff = max_backoff;
749        self
750    }
751
752    pub fn with_window(mut self, window: Duration) -> Self {
753        self.window = window;
754        self
755    }
756
757    /// Calculate the capped exponential delay for the next crash replacement.
758    /// `restart_in_window` is zero for the first replacement after an operator
759    /// action (restart, reload, re-enable) cleared the crash ring, or after all
760    /// older crash replacements have aged out of the window.
761    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762        if self.backoff.is_zero() || self.max_backoff.is_zero() {
763            return Duration::ZERO;
764        }
765
766        let mut delay = self.backoff;
767        for _ in 0..restart_in_window {
768            if delay >= self.max_backoff {
769                return self.max_backoff;
770            }
771            delay = delay
772                .checked_mul(10)
773                .unwrap_or(self.max_backoff)
774                .min(self.max_backoff);
775        }
776        delay.min(self.max_backoff)
777    }
778
779    /// The one sentence that explains a budget-exhausted stop, used for both the
780    /// log line and the terminal record so the two cannot drift. It names the
781    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
782    /// exactly what this budget is not.
783    fn budget_exhausted_detail(&self) -> String {
784        format!(
785            "crash budget exhausted: max_restarts={} within window_secs={}",
786            self.max_restarts,
787            self.window.as_secs()
788        )
789    }
790}
791
792impl Default for RestartPolicy {
793    fn default() -> Self {
794        Self {
795            max_restarts: DEFAULT_MAX_RESTARTS,
796            backoff: DEFAULT_BACKOFF,
797            max_backoff: DEFAULT_MAX_BACKOFF,
798            window: DEFAULT_RESTART_WINDOW,
799        }
800    }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805    restart_in_window: u32,
806    delay: Duration,
807}
808
809/// Whether the daemon itself will bring this module back after the exit being
810/// handled: it is enabled AND its in-window crash restarts are below the cap.
811///
812/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
813/// the window are dropped here rather than by a timer, so the count is right
814/// the moment somebody asks and no bookkeeping runs for idle modules.
815fn daemon_will_restart(
816    state: &mut SupervisorSnapshot,
817    policy: &RestartPolicy,
818    now: Instant,
819) -> bool {
820    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830    Report,
831    Restart,
832    Alert,
833}
834
835impl fmt::Display for HealthAction {
836    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837        f.write_str(match self {
838            Self::Report => "report",
839            Self::Restart => "restart",
840            Self::Alert => "alert",
841        })
842    }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847    /// Optional loopback HTTP endpoint for a managed non-wire process.
848    /// Changing it applies live on rescan; the process protocol changes only
849    /// at its next spawn.
850    pub http: Option<String>,
851    pub cadence: Duration,
852    pub deadline: Duration,
853    pub failure_threshold: u32,
854    pub on_degraded: HealthAction,
855    pub on_failing: HealthAction,
856    pub critical: bool,
857}
858
859impl Default for HealthConfig {
860    fn default() -> Self {
861        Self {
862            http: None,
863            cadence: DEFAULT_HEALTH_CADENCE,
864            deadline: DEFAULT_HEALTH_DEADLINE,
865            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866            on_degraded: HealthAction::Report,
867            on_failing: HealthAction::Report,
868            critical: false,
869        }
870    }
871}
872
873/// The supervisor's view of one module's health, relayed to clients over
874/// channel-0 and rendered by `ck health`.
875///
876/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
877/// stated here rather than only at the wire type a consumer reads. A reader can
878/// look up what `None` means; only a writer can silently change it, and the
879/// writer has no reason to go looking at a downstream contract before editing.
880///
881/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
882/// back to `None` on re-registration precisely so a respawned module does not
883/// carry its predecessor's timestamp — so an old value and an absent one call for
884/// opposite readings, and anything that defaulted this to a number would make a
885/// never-probed module indistinguishable from one probed at the epoch.
886///
887/// `detail` and `metrics` are `None` when the module published none on this
888/// probe, which does not mean it reported nothing wrong — it is also the shape
889/// when the probe never reached it. `last_probe_ms` is what separates those.
890#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892    pub status: SupervisorHealthStatus,
893    pub last_probe_ms: Option<u64>,
894    pub detail: Option<String>,
895    pub metrics: Option<Value>,
896    pub consecutive_failures: u32,
897    /// Number of replies received after a recurring health probe's deadline.
898    /// Unlike a timeout, every increment proves the module was alive.
899    pub late_answer_count: u64,
900    /// End-to-end latency of the newest late reply, measured from probe start.
901    pub last_late_answer_latency_ms: Option<u64>,
902    pub last_action: Option<String>,
903    /// Set together with `last_action`; the pair moves as one, and both being
904    /// absent means no escalation has ever been taken rather than that the last
905    /// one succeeded.
906    pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910    fn default() -> Self {
911        Self {
912            status: SupervisorHealthStatus::Unknown,
913            last_probe_ms: None,
914            detail: None,
915            metrics: None,
916            consecutive_failures: 0,
917            late_answer_count: 0,
918            last_late_answer_latency_ms: None,
919            last_action: None,
920            last_action_ms: None,
921        }
922    }
923}
924
925/// Typed lifecycle state for a supervised module.
926#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928    Starting,
929    Running,
930    Unresponsive,
931    Restarting,
932    Draining,
933    Stopped,
934    Failed,
935    Disabled,
936}
937
938impl fmt::Display for ModuleState {
939    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940        f.write_str(match self {
941            Self::Starting => "starting",
942            Self::Running => "running",
943            Self::Unresponsive => "unresponsive",
944            Self::Restarting => "restarting",
945            Self::Draining => "draining",
946            Self::Stopped => "stopped",
947            Self::Failed => "failed",
948            Self::Disabled => "disabled",
949        })
950    }
951}
952
953/// Supervisor classification of a child-process exit.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956    Clean,
957    Crash,
958    DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962    fn from(kind: ExitKind) -> Self {
963        match kind {
964            ExitKind::Clean => Self::Clean,
965            ExitKind::Crash => Self::Crash,
966            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967        }
968    }
969}
970
971/// Exact process identity retained when a supervised module registers its
972/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
973#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975    pub(crate) pid: u32,
976    pub(crate) start_time: u64,
977}
978
979/// Last observed child exit, if any.
980#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982    pub kind: ExitKind,
983    pub code: Option<i32>,
984    pub signal: Option<i32>,
985    pub at_ms: u64,
986}
987
988/// Point-in-time module status answerable by subc without forwarding to the
989/// module process.
990#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992    pub module_id: String,
993    pub state: ModuleState,
994    pub enabled: bool,
995    pub process_alive: bool,
996    pub registration_active: bool,
997    /// The module's declared wire protocol, carried beside `live` because it is
998    /// what makes `live` readable: the two fields answer one question together.
999    /// While a process is alive this is its launch declaration, not a later
1000    /// pending-reload edit. When down it is the configured next launch protocol.
1001    pub protocol: ModuleProtocol,
1002    /// Whether the module is serving, under the strongest definition the daemon
1003    /// can assert for its protocol.
1004    ///
1005    /// A subc module must also be REGISTERED: its process being alive says
1006    /// nothing about whether it can take a request. A `protocol: "none"` module
1007    /// never registers, so that term is dropped and this falls back to "enabled,
1008    /// running, and the process the daemon launched is alive" -- which is all
1009    /// the daemon observes about a process that speaks no subc wire. It stays a
1010    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
1011    /// rather than printing it bare.
1012    pub live: bool,
1013    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
1014    /// restarts have already released their slot, so this count can go down
1015    /// without anybody touching the module.
1016    pub restart_count: u32,
1017    /// Replacement processes spawned over this module's entire supervisor lifetime;
1018    /// unlike `restart_count`, this value is never reset by an operator action
1019    /// and never falls out of a window.
1020    pub lifetime_restarts: u32,
1021    pub spawn_generation: u64,
1022    /// The budget `restart_count` is spent against. Carried alongside the count
1023    /// because the count alone does not say how close the module is to being
1024    /// disabled, and reporting one without the other is what makes an
1025    /// about-to-be-retired module look ordinary.
1026    pub max_restarts: u32,
1027    /// The span `restart_count` is counted over. Carried with the pair above for
1028    /// the same reason they are carried together: "2 of 3" means one thing for a
1029    /// ten-minute window and something else entirely for a lifetime.
1030    pub restart_window: Duration,
1031    /// Effective drain and restart timing policy used by this running module.
1032    /// These values are carried together with the restart budget so status
1033    /// readers can compare configured intent with what the supervisor applied.
1034    pub drain_timeout: Duration,
1035    pub restart_backoff: Duration,
1036    pub restart_max_backoff: Duration,
1037    /// The module's process. On macOS this stays absent while the `ck-subc`
1038    /// launch trampoline is still running in that pid, and appears once the
1039    /// exec acknowledgement confirms the module image has replaced it. Launch
1040    /// time and the supervisor's own hold on the process are unaffected.
1041    pub pid: Option<u32>,
1042    pub spawned_at_ms: Option<u64>,
1043    pub spawned_from: Option<PathBuf>,
1044    pub process_start_time: Option<u64>,
1045    pub last_exit: Option<ExitReport>,
1046    pub health: ModuleHealthStatus,
1047}
1048
1049#[derive(Debug, Clone, PartialEq)]
1050struct SupervisorSnapshot {
1051    state: ModuleState,
1052    enabled: bool,
1053    process_alive: bool,
1054    spawned_protocol: Option<ModuleProtocol>,
1055    spawn_failure: Option<String>,
1056    /// When each crash restart was spent, oldest first. This IS the crash
1057    /// budget: its in-window length is the count an operator sees and the count
1058    /// the restart decision is made against, so there is no second counter that
1059    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1060    /// operator actions that used to zero the old lifetime counter.
1061    crash_restarts: VecDeque<Instant>,
1062    lifetime_restarts: u32,
1063    /// Successful child spawns in this daemon incarnation.
1064    ///
1065    /// `lifetime_restarts` was considered and rejected: it starts at zero
1066    /// (line 640), successful initial/operator spawns in `set_running` do not
1067    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1068    /// increments before a successful replacement exists (lines 604, 3846,
1069    /// and 3921), so a failed spawn can consume it. This counter moves only
1070    /// when a live PID is accepted below.
1071    spawn_generation: u64,
1072    pid: Option<u32>,
1073    #[cfg(target_os = "macos")]
1074    report_ready: Option<Arc<OnceLock<()>>>,
1075    /// Last reaped child, retained after current process facts are cleared.
1076    reaped_pid: Option<u32>,
1077    /// Whether the command-serving supervision loop has a scheduled respawn.
1078    respawn_pending: bool,
1079    /// A second restart is waiting for the replacement already scheduled.
1080    coalesced_restart_pending: bool,
1081    spawned_at_ms: Option<u64>,
1082    spawned_from: Option<PathBuf>,
1083    spawned_file_identity: Option<SpawnedFileIdentity>,
1084    process_start_time: Option<u64>,
1085    deliberate_severance: Option<ProcessIdentity>,
1086    last_exit: Option<ExitReport>,
1087    /// Diagnostic attached to the next drain's terminal record, if any.
1088    drain_disposition_detail: Option<String>,
1089    health: ModuleHealthStatus,
1090    /// Whether the current process was started as a swap candidate and so
1091    /// lives in the module's alternate cgroup. The next swap's candidate takes
1092    /// the other one, so the two processes of a swap never share a cgroup. A
1093    /// plain spawn always uses the primary cgroup.
1094    in_alternate_slot: bool,
1095    /// Whether the current `Draining` state ends in a replacement process
1096    /// (restart, reload, health restart) rather than a stop. Only meaningful
1097    /// while `state` is `Draining`; every entry into that state rewrites it.
1098    /// It is what lets route.open answer the retryable `module_reloading` to a
1099    /// consumer that reaches a still-registered process mid-restart, instead of
1100    /// the `supervisor_not_live` a stop or disable deserves.
1101    draining_to_replace: bool,
1102    /// Whether a configuration update has been applied since the current
1103    /// process was spawned, so that process runs an older spec than the one
1104    /// the supervisor now holds. A queued restart is only coalesced into a
1105    /// fresher process when this is false: a restart requested to pick up a
1106    /// new configuration must not be satisfied by a process that predates it.
1107    configuration_updated_since_spawn: bool,
1108}
1109
1110impl SupervisorSnapshot {
1111    /// The pid that status, provenance and resource readings may report. While
1112    /// the launch trampoline still runs in the pid, reading its executable or
1113    /// resource use would describe `ck-subc`, not the module, so none is
1114    /// reported until the exec acknowledgement confirms the module image.
1115    fn reported_pid(&self) -> Option<u32> {
1116        #[cfg(target_os = "macos")]
1117        if self
1118            .report_ready
1119            .as_ref()
1120            .is_some_and(|ready| ready.get().is_none())
1121        {
1122            return None;
1123        }
1124        self.pid
1125    }
1126
1127    fn starting() -> Self {
1128        Self::new(ModuleState::Starting, true)
1129    }
1130
1131    fn disabled() -> Self {
1132        Self::new(ModuleState::Disabled, false)
1133    }
1134
1135    fn failed() -> Self {
1136        Self::new(ModuleState::Failed, true)
1137    }
1138
1139    /// Crash restarts still inside `window`, having dropped the ones that are
1140    /// not. Pruning on read is what makes the budget a rate: an instant older
1141    /// than the window stops holding a slot the moment anybody counts.
1142    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1143        while let Some(oldest) = self.crash_restarts.front() {
1144            if now.duration_since(*oldest) > window {
1145                self.crash_restarts.pop_front();
1146            } else {
1147                break;
1148            }
1149        }
1150        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1151    }
1152
1153    /// Spend one unit of the crash budget and record the restart in the ledger.
1154    ///
1155    /// The ring is bounded by the cap because more than `max_restarts` in-window
1156    /// instants can never be reached (the caller refuses the restart first), so
1157    /// anything beyond that is an unbounded queue waiting to happen.
1158    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1159        self.crash_restarts.push_back(now);
1160        while self.crash_restarts.len() > policy.max_restarts as usize {
1161            self.crash_restarts.pop_front();
1162        }
1163        self.lifetime_restarts += 1;
1164    }
1165
1166    /// Reserve one crash-restart slot and calculate the delay before respawning.
1167    /// The count is captured before recording this restart, so the first retry
1168    /// uses the base delay and each later in-window retry escalates once.
1169    fn next_crash_restart(
1170        &mut self,
1171        policy: &RestartPolicy,
1172        now: Instant,
1173    ) -> Option<CrashRestartSchedule> {
1174        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1175        if restart_in_window >= policy.max_restarts {
1176            return None;
1177        }
1178        self.record_crash_restart(policy, now);
1179        Some(CrashRestartSchedule {
1180            restart_in_window,
1181            delay: policy.delay_for_restart(restart_in_window),
1182        })
1183    }
1184
1185    /// Give the module its full budget back, as an operator restart, reload, or
1186    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1187    /// ledger of what actually happened, and an operator action does not unmake
1188    /// the crashes.
1189    fn clear_crash_restarts(&mut self) {
1190        self.crash_restarts.clear();
1191    }
1192
1193    fn new(state: ModuleState, enabled: bool) -> Self {
1194        Self {
1195            state,
1196            enabled,
1197            process_alive: false,
1198            spawned_protocol: None,
1199            spawn_failure: None,
1200            crash_restarts: VecDeque::new(),
1201            lifetime_restarts: 0,
1202            spawn_generation: 0,
1203            pid: None,
1204            #[cfg(target_os = "macos")]
1205            report_ready: None,
1206            reaped_pid: None,
1207            respawn_pending: false,
1208            coalesced_restart_pending: false,
1209            spawned_at_ms: None,
1210            spawned_from: None,
1211            spawned_file_identity: None,
1212            process_start_time: None,
1213            deliberate_severance: None,
1214            last_exit: None,
1215            drain_disposition_detail: None,
1216            health: ModuleHealthStatus::default(),
1217            in_alternate_slot: false,
1218            draining_to_replace: false,
1219            configuration_updated_since_spawn: false,
1220        }
1221    }
1222}
1223
1224type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1225
1226type SpawnSubscriberKey = (ConnectionId, u64);
1227
1228#[derive(Debug)]
1229struct SpawnSubscriber {
1230    version: u8,
1231    frames: mpsc::Sender<Frame>,
1232    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1233    /// from which event. The full frame channel cannot carry that news, so it
1234    /// travels beside it; see `SpawnEventFeed::subscribe`.
1235    lagged: Option<oneshot::Sender<SpawnCursor>>,
1236}
1237
1238#[derive(Debug)]
1239struct SpawnEventState {
1240    daemon_incarnation: String,
1241    seq: u64,
1242    capacity: usize,
1243    live: HashMap<String, LiveSpawn>,
1244    generations: HashMap<String, u64>,
1245    events: VecDeque<SpawnEvent>,
1246    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1247}
1248
1249impl Default for SpawnEventState {
1250    fn default() -> Self {
1251        Self {
1252            daemon_incarnation: "unconfigured".to_string(),
1253            seq: 0,
1254            capacity: SPAWN_EVENT_RING_CAPACITY,
1255            live: HashMap::new(),
1256            generations: HashMap::new(),
1257            events: VecDeque::new(),
1258            subscribers: HashMap::new(),
1259        }
1260    }
1261}
1262
1263#[derive(Debug, Clone, Default)]
1264struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1265
1266#[derive(Debug, Clone, PartialEq, Eq)]
1267pub(crate) enum SpawnSubscribeRefusal {
1268    ForeignIncarnation { current: String },
1269    TooOld { oldest: SpawnCursor },
1270    Frame(String),
1271}
1272
1273impl SpawnEventFeed {
1274    fn configure_incarnation(&self, daemon_incarnation: String) {
1275        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1276        state.daemon_incarnation = daemon_incarnation;
1277        state.seq = 0;
1278        state.live.clear();
1279        state.generations.clear();
1280        state.events.clear();
1281        state.subscribers.clear();
1282    }
1283
1284    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1285        SpawnCursor {
1286            daemon_incarnation: state.daemon_incarnation.clone(),
1287            seq: state.seq,
1288        }
1289    }
1290
1291    fn snapshot(&self) -> SpawnSnapshot {
1292        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1293        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1294        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1295        SpawnSnapshot {
1296            cursor: Self::cursor(&state),
1297            ring_bound: state.capacity as u64,
1298            live,
1299        }
1300    }
1301
1302    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1303        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1304        let generation = state
1305            .generations
1306            .get(module_id)
1307            .copied()
1308            .unwrap_or(0)
1309            .checked_add(1)
1310            .expect("spawn generation exhausted");
1311        state.generations.insert(module_id.to_string(), generation);
1312        let live = LiveSpawn {
1313            module_id: module_id.to_string(),
1314            spawn_generation: generation,
1315            pid,
1316            spawned_at_ms,
1317        };
1318        state.live.insert(module_id.to_string(), live);
1319        Self::emit_locked(
1320            &mut state,
1321            SpawnEventKind::Spawned,
1322            module_id.to_string(),
1323            generation,
1324            pid,
1325            None,
1326            None,
1327        );
1328        generation
1329    }
1330
1331    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1332        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1333        let Some(live) = state.live.remove(module_id) else {
1334            warn!(
1335                module_id,
1336                "terminal record had no live spawn event identity"
1337            );
1338            return;
1339        };
1340        Self::emit_locked(
1341            &mut state,
1342            SpawnEventKind::Exited,
1343            module_id.to_string(),
1344            live.spawn_generation,
1345            live.pid,
1346            exit_code,
1347            exit_signal,
1348        );
1349    }
1350
1351    /// Report the exit of a process that a swap has already replaced.
1352    ///
1353    /// `emit_exited` removes the module's live entry, which after a swap's
1354    /// cutover describes the promoted candidate, not the old process now
1355    /// exiting. This emits the old generation's exit and leaves the live entry
1356    /// alone unless it still names that generation.
1357    fn emit_superseded_exited(
1358        &self,
1359        module_id: &str,
1360        spawn_generation: u64,
1361        pid: u32,
1362        exit_code: Option<i32>,
1363        exit_signal: Option<i32>,
1364    ) {
1365        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1366        if state
1367            .live
1368            .get(module_id)
1369            .is_some_and(|live| live.spawn_generation == spawn_generation)
1370        {
1371            state.live.remove(module_id);
1372        }
1373        Self::emit_locked(
1374            &mut state,
1375            SpawnEventKind::Exited,
1376            module_id.to_string(),
1377            spawn_generation,
1378            pid,
1379            exit_code,
1380            exit_signal,
1381        );
1382    }
1383
1384    #[allow(clippy::too_many_arguments)]
1385    fn emit_locked(
1386        state: &mut SpawnEventState,
1387        kind: SpawnEventKind,
1388        module_id: String,
1389        spawn_generation: u64,
1390        pid: u32,
1391        exit_code: Option<i32>,
1392        exit_signal: Option<i32>,
1393    ) {
1394        state.seq = state
1395            .seq
1396            .checked_add(1)
1397            .expect("spawn event sequence exhausted");
1398        let event = SpawnEvent {
1399            cursor: Self::cursor(state),
1400            kind,
1401            module_id,
1402            spawn_generation,
1403            pid,
1404            exit_code,
1405            exit_signal,
1406        };
1407        state.events.push_back(event.clone());
1408        while state.events.len() > state.capacity {
1409            state.events.pop_front();
1410        }
1411        let body = match serde_json::to_vec(&event) {
1412            Ok(body) => body,
1413            Err(error) => {
1414                error!(%error, "failed to serialize supervisor spawn event");
1415                return;
1416            }
1417        };
1418        state.subscribers.retain(|(connection_id, corr), subscriber| {
1419            let frame = Frame::build_with_version(
1420                subscriber.version,
1421                FrameType::StreamData,
1422                control_flags(),
1423                0,
1424                0,
1425                *corr,
1426                body.clone(),
1427            );
1428            match frame {
1429                Ok(frame) => {
1430                    if subscriber.frames.try_send(frame).is_ok() {
1431                        true
1432                    } else {
1433                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1434                        if let Some(lagged) = subscriber.lagged.take() {
1435                            let _ = lagged.send(event.cursor.clone());
1436                        }
1437                        false
1438                    }
1439                }
1440                Err(error) => {
1441                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1442                    false
1443                }
1444            }
1445        });
1446    }
1447
1448    fn subscribe(
1449        &self,
1450        connection_id: ConnectionId,
1451        corr: u64,
1452        version: u8,
1453        since: Option<SpawnCursor>,
1454        sink: FrameSink,
1455    ) -> Result<(), SpawnSubscribeRefusal> {
1456        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1457        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1458        {
1459            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1460            let replay = if let Some(since) = since {
1461                if since.daemon_incarnation != state.daemon_incarnation {
1462                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1463                        current: state.daemon_incarnation.clone(),
1464                    });
1465                }
1466                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1467                    if since.seq < oldest.seq.saturating_sub(1) {
1468                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1469                    }
1470                }
1471                state
1472                    .events
1473                    .iter()
1474                    .filter(|event| event.cursor.seq > since.seq)
1475                    .cloned()
1476                    .collect::<Vec<_>>()
1477            } else {
1478                Vec::new()
1479            };
1480            for event in replay {
1481                let body = serde_json::to_vec(&event)
1482                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1483                let frame = Frame::build_with_version(
1484                    version,
1485                    FrameType::StreamData,
1486                    control_flags(),
1487                    0,
1488                    0,
1489                    corr,
1490                    body,
1491                )
1492                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1493                frames
1494                    .try_send(frame)
1495                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1496            }
1497            state.subscribers.insert(
1498                (connection_id, corr),
1499                SpawnSubscriber {
1500                    version,
1501                    frames,
1502                    lagged: Some(lagged),
1503                },
1504            );
1505        }
1506        // The lagged terminal is sent here, by the forwarder, rather than by
1507        // the emitter: at the moment of the drop the subscriber's own channel
1508        // is full, and writing to the connection sink directly from the emitter
1509        // would put the Error AHEAD of the events still queued in that channel
1510        // (and the emitter holds the feed lock, so it cannot await the sink).
1511        // Dropping the subscriber drops the only sender, so `recv` drains every
1512        // queued event and then returns `None`; only then is the Error sent, so
1513        // the client sees each event it can keep, then the reason it was cut.
1514        // Cancel and connection removal drop the oneshot unsent, so they end
1515        // the stream with no Error.
1516        tokio::spawn(async move {
1517            while let Some(frame) = receiver.recv().await {
1518                if sink.send(frame).await.is_err() {
1519                    return;
1520                }
1521            }
1522            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1523                return;
1524            };
1525            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1526                Ok(frame) => {
1527                    let _ = sink.send(frame).await;
1528                }
1529                Err(error) => {
1530                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1531                }
1532            }
1533        });
1534        Ok(())
1535    }
1536
1537    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1538        let Some(subscriber) = self
1539            .0
1540            .lock()
1541            .unwrap_or_else(|p| p.into_inner())
1542            .subscribers
1543            .remove(&(connection_id, corr))
1544        else {
1545            return false;
1546        };
1547        if let Ok(frame) = Frame::build_with_version(
1548            subscriber.version,
1549            FrameType::StreamEnd,
1550            control_flags(),
1551            0,
1552            0,
1553            corr,
1554            Vec::new(),
1555        ) {
1556            tokio::spawn(async move {
1557                let _ = subscriber.frames.send(frame).await;
1558            });
1559        }
1560        true
1561    }
1562
1563    fn remove_connection(&self, connection_id: ConnectionId) {
1564        self.0
1565            .lock()
1566            .unwrap_or_else(|p| p.into_inner())
1567            .subscribers
1568            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1569    }
1570
1571    #[cfg(any(test, feature = "test-support"))]
1572    fn set_capacity(&self, capacity: usize) {
1573        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1574    }
1575
1576    #[cfg(any(test, feature = "test-support"))]
1577    fn subscriber_count(&self) -> usize {
1578        self.0
1579            .lock()
1580            .unwrap_or_else(|p| p.into_inner())
1581            .subscribers
1582            .len()
1583    }
1584}
1585
1586/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1587/// The terminal Error a lagged spawn subscriber receives after its queued events.
1588fn spawn_subscriber_lagged_frame(
1589    version: u8,
1590    corr: u64,
1591    first_undelivered: SpawnCursor,
1592) -> Result<Frame, String> {
1593    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1594        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1595        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1596            .to_string(),
1597        detail: Some(serde_json::json!({
1598            "first_undelivered_cursor": first_undelivered
1599        })),
1600    })
1601    .map_err(|error| error.to_string())?;
1602    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1603        .map_err(|error| error.to_string())
1604}
1605
1606pub trait ModuleProcessLiveness: Send + Sync {
1607    fn process_live(&self, module_id: &str) -> Option<bool>;
1608
1609    /// Whether the supervisor is replacing this module's process right now: an
1610    /// operator restart or reload, a health restart, or a crash respawn whose
1611    /// backoff is running. A module in that state is not live, but a consumer
1612    /// refused now should retry shortly rather than treat the target as gone.
1613    /// Stopped, failed, and disabled modules are not replacing.
1614    fn process_replacing(&self, _module_id: &str) -> bool {
1615        false
1616    }
1617}
1618
1619/// Shared process-liveness registry keyed by supervised `module_id`.
1620#[derive(Debug, Clone, Default)]
1621pub struct SupervisorProcessLiveness {
1622    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1623}
1624
1625impl SupervisorProcessLiveness {
1626    pub fn new() -> Self {
1627        Self::default()
1628    }
1629
1630    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1631        let mut snapshots = self
1632            .snapshots
1633            .lock()
1634            .unwrap_or_else(|poisoned| poisoned.into_inner());
1635        snapshots.insert(module_id, snapshot);
1636    }
1637
1638    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1639        let mut snapshots = self
1640            .snapshots
1641            .lock()
1642            .unwrap_or_else(|poisoned| poisoned.into_inner());
1643        let is_current = snapshots
1644            .get(module_id)
1645            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1646            .unwrap_or(false);
1647        if is_current {
1648            snapshots.remove(module_id);
1649        }
1650    }
1651}
1652
1653impl ModuleProcessLiveness for SupervisorProcessLiveness {
1654    fn process_live(&self, module_id: &str) -> Option<bool> {
1655        let snapshot = {
1656            let snapshots = self
1657                .snapshots
1658                .lock()
1659                .unwrap_or_else(|poisoned| poisoned.into_inner());
1660            snapshots.get(module_id).cloned()
1661        }?;
1662        let snapshot = snapshot
1663            .lock()
1664            .unwrap_or_else(|poisoned| poisoned.into_inner());
1665        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1666    }
1667
1668    fn process_replacing(&self, module_id: &str) -> bool {
1669        let Some(snapshot) = self
1670            .snapshots
1671            .lock()
1672            .unwrap_or_else(|poisoned| poisoned.into_inner())
1673            .get(module_id)
1674            .cloned()
1675        else {
1676            return false;
1677        };
1678        let snapshot = snapshot
1679            .lock()
1680            .unwrap_or_else(|poisoned| poisoned.into_inner());
1681        snapshot.enabled
1682            && match snapshot.state {
1683                ModuleState::Restarting => true,
1684                ModuleState::Draining => snapshot.draining_to_replace,
1685                ModuleState::Starting
1686                | ModuleState::Running
1687                | ModuleState::Unresponsive
1688                | ModuleState::Stopped
1689                | ModuleState::Failed
1690                | ModuleState::Disabled => false,
1691            }
1692    }
1693}
1694
1695#[cfg(test)]
1696#[derive(Debug, Default)]
1697struct ReloadExitRecordGate {
1698    reached: tokio::sync::Notify,
1699    resume: tokio::sync::Notify,
1700}
1701
1702#[derive(Debug, Clone, Copy)]
1703enum RespawnKind {
1704    Spawn,
1705    Reload,
1706}
1707
1708#[derive(Debug, Clone, Copy)]
1709struct PendingRespawn {
1710    deadline: Instant,
1711    kind: RespawnKind,
1712}
1713
1714type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1715
1716#[derive(Debug, Clone)]
1717struct SupervisorRuntimeConfig {
1718    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1719    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1720    /// A reload acknowledges completion only after its replacement registers.
1721    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1722    restart_policy: RestartPolicy,
1723    /// This module's RESOLVED drain budget: per-module config when present,
1724    /// else `default_drain_timeout`.
1725    drain_timeout: Duration,
1726    /// Shared with the status handle so the attested value changes atomically
1727    /// when a rescan updates the running drain policy.
1728    effective_drain_timeout: Arc<Mutex<Duration>>,
1729    /// The supervisor-wide fallback, kept so a configuration update that
1730    /// REMOVES the per-module override can re-resolve to it.
1731    default_drain_timeout: Duration,
1732    health: HealthConfig,
1733    connection_file_path: Option<PathBuf>,
1734    capture_logs_dir: Option<PathBuf>,
1735    forwarding: Option<Arc<ForwardingTable>>,
1736    /// The shared handle, so every spawn path (initial, restart, reload) records the
1737    /// reserved-module launch nonce the HELLO verifier checks against.
1738    supervisor_handle: Option<SupervisorHandle>,
1739    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1740    /// status queries.
1741    ///
1742    /// One ring per module, held across every respawn. The lines explaining an exit
1743    /// are written BEFORE that exit, so a ring recreated per process would be empty
1744    /// exactly when it is asked for.
1745    stderr_ring: Arc<Mutex<StderrRing>>,
1746    terminal_ring: Arc<Mutex<TerminalRing>>,
1747    spawn_events: SpawnEventFeed,
1748    child_roster: ChildRoster,
1749    #[cfg(target_os = "linux")]
1750    cgroup_placement: Option<subc_cgroup::Placement>,
1751    #[cfg(test)]
1752    test_seed_stale_facts_before_enable_spawn: bool,
1753    #[cfg(test)]
1754    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1755}
1756
1757#[derive(Debug, Clone, PartialEq, Eq)]
1758struct SupervisedConfiguration {
1759    spec: ModuleSpec,
1760    health: HealthConfig,
1761}
1762
1763/// Shared daemon lookup table for supervised module handles.
1764///
1765/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1766/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1767/// launch nonces recorded at spawn are checked by the same daemon instance.
1768#[derive(Debug, Clone, Default)]
1769pub struct SupervisorHandle {
1770    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1771    /// Module ids the supervisor has taken on. An id is added BEFORE the
1772    /// module's first process is spawned and removed only when the module
1773    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1774    /// the keys of `modules`.
1775    ///
1776    /// `modules` cannot answer "is this module configured?" on its own: a
1777    /// [`SupervisedModule`] only exists once its process has been spawned, and
1778    /// a fast child can connect, register, sync its scopes and ask about them
1779    /// before the supervisor has inserted it. Answering "not configured" in that
1780    /// gap makes scope admission refuse with the terminal "will never sync"
1781    /// instead of the retryable "has not synced yet".
1782    configured_ids: Arc<Mutex<HashSet<String>>>,
1783    spawn_events: SpawnEventFeed,
1784    /// The current expected launch nonce for each reserved module_id. Set when the
1785    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1786    /// non-reserved module never has an entry here and is never nonce-checked.
1787    /// Reserved module ids and the nonce that authorizes their next HELLO.
1788    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1789    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1790    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1791    /// had NO entry and admitted anyone: the reservation protected the nonce
1792    /// holder, not the NAME (found live by CKCRED's canary probe registering
1793    /// against a reserved scratch id).
1794    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1795    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1796    ///
1797    /// This is deliberately in-memory only: subc is state-free across daemon
1798    /// restarts, and the tombstone only explains the hours-after-removal window
1799    /// while this executing daemon is still alive. Do not persist it in a store.
1800    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1801    /// The current launch nonce for every supervised spawn. This is separate from
1802    /// reserved_nonces because consumer route.open attestation applies to all spawned
1803    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1804    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1805    /// Reserved namespace prefixes mapped to the supervised owner module whose
1806    /// current spawn nonce authorizes HELLO claims below the prefix.
1807    ///
1808    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1809    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1810    /// accidental collisions and lower-trust processes from squatting protected
1811    /// namespaces.
1812    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1813    /// Blue/green swaps in progress, by module id. An entry exists from just
1814    /// before the candidate process is spawned until the swap has failed, or
1815    /// has cut over and the old process is gone. While it exists, HELLO for the
1816    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1817    /// consumer attestation accepts both processes' nonces.
1818    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1819    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1820    promotion_observer: PromotionObserverSlot,
1821    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1822    /// this daemon-wide ordering, a rescan could retire or update a module while a
1823    /// concurrent reload still held its old handle and launch specification.
1824    operation_lock: Arc<AsyncMutex<()>>,
1825}
1826
1827/// Told when a swap has promoted its candidate to be the module's active
1828/// registration.
1829///
1830/// An ordinary HELLO runs the control plane's registration side effects (the
1831/// capability cache, the deny census, the requirement recompute) as it
1832/// registers. A swap candidate's HELLO does not, because it is not routable;
1833/// promotion is when those must run instead, and promotion happens in the
1834/// supervisor, which has no other way into the control handler.
1835pub(crate) trait SwapPromotionObserver: Send + Sync {
1836    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1837}
1838
1839/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1840/// control handler) owns this handle, so a strong reference back would be a
1841/// cycle that keeps both alive.
1842#[derive(Clone, Default)]
1843struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1844
1845impl fmt::Debug for PromotionObserverSlot {
1846    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1847        f.write_str("PromotionObserverSlot")
1848    }
1849}
1850
1851/// The nonces of one open swap.
1852#[derive(Debug, Clone)]
1853struct OpenSwap {
1854    /// The launch nonce minted for the candidate process. It is the swap
1855    /// token: the only thing that admits a HELLO into the candidate slot.
1856    candidate_nonce: String,
1857    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1858    /// here because cutover moves the module's recorded spawn nonce to the
1859    /// candidate while the incumbent is still draining and its consumers are
1860    /// still attesting with this one.
1861    incumbent_nonce: Option<String>,
1862    /// Set once a HELLO has been admitted with the swap token, so the token
1863    /// admits one registration and cannot be replayed after cutover empties
1864    /// the candidate slot.
1865    candidate_admitted: bool,
1866}
1867
1868/// What the swap gate says about a HELLO. See
1869/// [`SupervisorHandle::swap_hello_admission`].
1870#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1871pub(crate) enum SwapHelloAdmission {
1872    /// No swap is open for the id (or the HELLO carries the incumbent's own
1873    /// nonce); the ordinary gates decide.
1874    NotSwapping,
1875    /// The HELLO carries the swap token: register it into the candidate slot.
1876    Candidate,
1877    /// A swap is open and the HELLO carries a nonce the supervisor did not
1878    /// mint for this id, no nonce, or a token already used.
1879    Refused,
1880}
1881
1882#[derive(Debug, Clone, PartialEq, Eq)]
1883pub(crate) enum ReservedHelloRejection {
1884    Exact {
1885        module_id: String,
1886    },
1887    Prefix {
1888        prefix: String,
1889        owner_module_id: String,
1890    },
1891}
1892
1893impl SupervisorHandle {
1894    pub fn new() -> Self {
1895        Self::default()
1896    }
1897
1898    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1899        self.spawn_events.snapshot()
1900    }
1901
1902    pub(crate) fn subscribe_spawns(
1903        &self,
1904        connection_id: ConnectionId,
1905        corr: u64,
1906        version: u8,
1907        since: Option<SpawnCursor>,
1908        sink: FrameSink,
1909    ) -> Result<(), SpawnSubscribeRefusal> {
1910        self.spawn_events
1911            .subscribe(connection_id, corr, version, since, sink)
1912    }
1913
1914    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1915        self.spawn_events.cancel(connection_id, corr)
1916    }
1917
1918    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1919        self.spawn_events.remove_connection(connection_id);
1920    }
1921
1922    #[cfg(any(test, feature = "test-support"))]
1923    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1924        assert!(capacity > 0, "spawn event capacity must be non-zero");
1925        self.spawn_events.set_capacity(capacity);
1926    }
1927
1928    #[cfg(any(test, feature = "test-support"))]
1929    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1930        self.spawn_events.subscriber_count()
1931    }
1932
1933    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1934    /// a respawn invalidates stale consumer identities.
1935    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1936        self.spawn_nonces
1937            .lock()
1938            .unwrap_or_else(|poisoned| poisoned.into_inner())
1939            .insert(module_id.to_string(), nonce);
1940    }
1941
1942    /// Record the launch nonce expected from the next HELLO for a reserved module,
1943    /// replacing any prior nonce (a respawn invalidates the previous one).
1944    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1945        self.reserved_nonces
1946            .lock()
1947            .unwrap_or_else(|poisoned| poisoned.into_inner())
1948            .insert(module_id.to_string(), Some(nonce));
1949    }
1950
1951    /// Record namespace prefixes owned by a supervised module.
1952    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1953        let mut owners = self
1954            .reserved_prefix_owners
1955            .lock()
1956            .unwrap_or_else(|poisoned| poisoned.into_inner());
1957        owners.retain(|_, owner| owner != owner_module_id);
1958        for prefix in prefixes {
1959            owners.insert(prefix.clone(), owner_module_id.to_string());
1960        }
1961    }
1962
1963    /// The launch nonce most recently minted for a module's spawn, if any.
1964    #[cfg(test)]
1965    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1966        self.spawn_nonces
1967            .lock()
1968            .unwrap_or_else(|poisoned| poisoned.into_inner())
1969            .get(module_id)
1970            .cloned()
1971    }
1972
1973    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1974        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1975        let spawn_nonce = self
1976            .spawn_nonces
1977            .lock()
1978            .unwrap_or_else(|poisoned| poisoned.into_inner())
1979            .get(&spec.module_id)
1980            .cloned();
1981        let mut reserved_nonces = self
1982            .reserved_nonces
1983            .lock()
1984            .unwrap_or_else(|poisoned| poisoned.into_inner());
1985        if spec.reserved {
1986            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1987            // reserved name whose module has never spawned has no legitimate
1988            // holder, and the entry's absence is what used to leave the name
1989            // open to the first claimant.
1990            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1991        }
1992        drop(reserved_nonces);
1993        // A later unreserved declaration must not silently unreserve an id that
1994        // was retained after its reserved configuration was removed. The explicit
1995        // release ceremony is the only operation that retires that gate.
1996        self.removal_tombstones
1997            .lock()
1998            .unwrap_or_else(|poisoned| poisoned.into_inner())
1999            .remove(&spec.module_id);
2000    }
2001
2002    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
2003    /// authorized only by its expected nonce; otherwise a matching reserved prefix
2004    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
2005    /// with no matching prefix are always authorized.
2006    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2007        self.reserved_hello_rejection(module_id, presented)
2008            .is_none()
2009    }
2010
2011    pub(crate) fn reserved_hello_rejection(
2012        &self,
2013        module_id: &str,
2014        presented: Option<&str>,
2015    ) -> Option<ReservedHelloRejection> {
2016        let nonces = self
2017            .reserved_nonces
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner());
2020        if let Some(expected) = nonces.get(module_id) {
2021            // `None` = reserved with no legitimate holder: refuse every
2022            // presentation, because no process can hold a nonce that was never
2023            // minted. Only a real minted nonce admits, in constant time.
2024            let authorized = match expected {
2025                Some(expected) => {
2026                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2027                }
2028                None => false,
2029            };
2030            if authorized {
2031                return None;
2032            }
2033            return Some(ReservedHelloRejection::Exact {
2034                module_id: module_id.to_string(),
2035            });
2036        }
2037        drop(nonces);
2038
2039        let matched_prefix = self
2040            .reserved_prefix_owners
2041            .lock()
2042            .unwrap_or_else(|poisoned| poisoned.into_inner())
2043            .iter()
2044            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2045            .max_by_key(|(prefix, _)| prefix.len())
2046            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2047        let (prefix, owner_module_id) = matched_prefix?;
2048
2049        let authorized = presented.is_some_and(|presented| {
2050            self.spawn_nonces
2051                .lock()
2052                .unwrap_or_else(|poisoned| poisoned.into_inner())
2053                .get(&owner_module_id)
2054                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2055                // While the owner is being swapped, children started by
2056                // either of its two processes hold that process's nonce.
2057                || self.swap_nonce_matches(&owner_module_id, presented)
2058        });
2059        if authorized {
2060            None
2061        } else {
2062            Some(ReservedHelloRejection::Prefix {
2063                prefix,
2064                owner_module_id,
2065            })
2066        }
2067    }
2068
2069    /// Whether a consumer connection proved it came from a daemon-spawned module.
2070    ///
2071    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2072    /// accepted only for module ids the supervisor has spawned.
2073    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2074        if presented.is_empty() {
2075            return false;
2076        }
2077        let nonces = self
2078            .spawn_nonces
2079            .lock()
2080            .unwrap_or_else(|poisoned| poisoned.into_inner());
2081        let current = nonces
2082            .get(module_id)
2083            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2084        drop(nonces);
2085        // During a swap two processes of the module are alive, and a consumer
2086        // started by either one presents that process's nonce. Accepting only
2087        // the recorded one would fail the incumbent's consumers for the whole
2088        // overlap once cutover moves the record to the candidate.
2089        current || self.swap_nonce_matches(module_id, presented)
2090    }
2091
2092    /// Whether `presented` is either nonce of an open swap for `module_id`.
2093    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2094        let swaps = self
2095            .swaps
2096            .lock()
2097            .unwrap_or_else(|poisoned| poisoned.into_inner());
2098        swaps.get(module_id).is_some_and(|swap| {
2099            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2100                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2101                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2102                })
2103        })
2104    }
2105
2106    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2107    /// Called before the candidate process exists.
2108    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2109        let incumbent_nonce = self
2110            .spawn_nonces
2111            .lock()
2112            .unwrap_or_else(|poisoned| poisoned.into_inner())
2113            .get(module_id)
2114            .cloned();
2115        self.swaps
2116            .lock()
2117            .unwrap_or_else(|poisoned| poisoned.into_inner())
2118            .insert(
2119                module_id.to_string(),
2120                OpenSwap {
2121                    candidate_nonce,
2122                    incumbent_nonce,
2123                    candidate_admitted: false,
2124                },
2125            );
2126    }
2127
2128    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2129    /// the module's recorded one.
2130    pub(crate) fn close_swap(&self, module_id: &str) {
2131        self.swaps
2132            .lock()
2133            .unwrap_or_else(|poisoned| poisoned.into_inner())
2134            .remove(module_id);
2135    }
2136
2137    /// Install the observer told about swap promotions, replacing any earlier
2138    /// one.
2139    pub(crate) fn set_swap_promotion_observer(
2140        &self,
2141        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2142    ) {
2143        *self
2144            .promotion_observer
2145            .0
2146            .lock()
2147            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2148    }
2149
2150    /// Tell the installed observer, if it is still alive, that a swap promoted
2151    /// `registration`.
2152    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2153        let observer = self
2154            .promotion_observer
2155            .0
2156            .lock()
2157            .unwrap_or_else(|poisoned| poisoned.into_inner())
2158            .as_ref()
2159            .and_then(std::sync::Weak::upgrade);
2160        if let Some(observer) = observer {
2161            observer.swap_promoted(registration);
2162        }
2163    }
2164
2165    /// Whether a swap is open for `module_id`.
2166    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2167        self.swaps
2168            .lock()
2169            .unwrap_or_else(|poisoned| poisoned.into_inner())
2170            .contains_key(module_id)
2171    }
2172
2173    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2174    /// respawn would, once cutover has made the candidate the module's process.
2175    /// The swap stays open so the incumbent's nonce keeps attesting until the
2176    /// incumbent has drained and exited.
2177    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2178        let candidate_nonce = self
2179            .swaps
2180            .lock()
2181            .unwrap_or_else(|poisoned| poisoned.into_inner())
2182            .get(module_id)
2183            .map(|swap| swap.candidate_nonce.clone());
2184        let Some(nonce) = candidate_nonce else {
2185            return;
2186        };
2187        self.set_spawn_nonce(module_id, nonce.clone());
2188        if reserved {
2189            self.set_reserved_nonce(module_id, nonce);
2190        }
2191    }
2192
2193    /// The swap gate for a HELLO claiming `module_id`.
2194    ///
2195    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2196    /// presents the candidate nonce, which the reserved gate (holding the
2197    /// incumbent's nonce) would refuse as `reserved_module` before swap
2198    /// admission was ever reached. And it applies to unreserved ids too: for an
2199    /// unreserved id the only thing that ever stopped a second process claiming
2200    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2201    /// refusal a swap lifts for its candidate.
2202    ///
2203    /// The incumbent's own nonce falls through to the ordinary gates, which
2204    /// treat it as they always have (a live incumbent is refused as a
2205    /// duplicate). Anything else while a swap is open is refused, including an
2206    /// absent nonce.
2207    pub(crate) fn swap_hello_admission(
2208        &self,
2209        module_id: &str,
2210        presented: Option<&str>,
2211    ) -> SwapHelloAdmission {
2212        let swaps = self
2213            .swaps
2214            .lock()
2215            .unwrap_or_else(|poisoned| poisoned.into_inner());
2216        let Some(swap) = swaps.get(module_id) else {
2217            return SwapHelloAdmission::NotSwapping;
2218        };
2219        let Some(presented) = presented else {
2220            return SwapHelloAdmission::Refused;
2221        };
2222        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2223            return if swap.candidate_admitted {
2224                SwapHelloAdmission::Refused
2225            } else {
2226                SwapHelloAdmission::Candidate
2227            };
2228        }
2229        if swap
2230            .incumbent_nonce
2231            .as_deref()
2232            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2233        {
2234            return SwapHelloAdmission::NotSwapping;
2235        }
2236        SwapHelloAdmission::Refused
2237    }
2238
2239    /// Record that the swap token has registered a candidate, so it admits no
2240    /// second HELLO.
2241    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2242        if let Some(swap) = self
2243            .swaps
2244            .lock()
2245            .unwrap_or_else(|poisoned| poisoned.into_inner())
2246            .get_mut(module_id)
2247        {
2248            swap.candidate_admitted = true;
2249        }
2250    }
2251
2252    /// Test/support lookup for the current launch nonce of a supervised spawn.
2253    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2254        self.spawn_nonces
2255            .lock()
2256            .unwrap_or_else(|poisoned| poisoned.into_inner())
2257            .get(module_id)
2258            .cloned()
2259    }
2260
2261    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2262    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2263        self.reserved_nonces
2264            .lock()
2265            .unwrap_or_else(|poisoned| poisoned.into_inner())
2266            .get(module_id)
2267            .cloned()
2268            .flatten()
2269    }
2270
2271    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2272        // Normally already marked before the process was spawned; marking here
2273        // too keeps `configured_ids` a superset of the roster for any caller
2274        // that inserts a module directly.
2275        self.mark_configured(module.module_id());
2276        let mut modules = self
2277            .modules
2278            .lock()
2279            .unwrap_or_else(|poisoned| poisoned.into_inner());
2280        modules.insert(module.module_id().to_string(), module)
2281    }
2282
2283    /// Record that the supervisor has taken on `module_id`. Called before the
2284    /// module's first process is spawned, so that by the time that process can
2285    /// register, [`Self::is_configured`] already answers true.
2286    fn mark_configured(&self, module_id: &str) {
2287        self.configured_ids
2288            .lock()
2289            .unwrap_or_else(|poisoned| poisoned.into_inner())
2290            .insert(module_id.to_string());
2291    }
2292
2293    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2294    /// before it was ever put on the roster. A module already on the roster
2295    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2296    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2297        let modules = self
2298            .modules
2299            .lock()
2300            .unwrap_or_else(|poisoned| poisoned.into_inner());
2301        if !modules.contains_key(module_id) {
2302            self.configured_ids
2303                .lock()
2304                .unwrap_or_else(|poisoned| poisoned.into_inner())
2305                .remove(module_id);
2306        }
2307    }
2308
2309    /// Whether `module_id` is a module this daemon supervises: on the roster,
2310    /// or about to be (its process is being spawned right now).
2311    ///
2312    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2313    /// for scopes: a supervised module's process can register and sync before
2314    /// [`Self::get`] can return it, and in that window it is still a module
2315    /// that will sync, not one that never will.
2316    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2317        self.configured_ids
2318            .lock()
2319            .unwrap_or_else(|poisoned| poisoned.into_inner())
2320            .contains(module_id)
2321    }
2322
2323    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2324        let modules = self
2325            .modules
2326            .lock()
2327            .unwrap_or_else(|poisoned| poisoned.into_inner());
2328        modules.get(module_id).cloned()
2329    }
2330
2331    pub(crate) fn record_late_health_answer(
2332        &self,
2333        module_id: &str,
2334        latency_ms: u64,
2335    ) -> Result<bool, SuperviseError> {
2336        let Some(module) = self.get(module_id) else {
2337            return Ok(false);
2338        };
2339        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2340            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2341            state.health.last_late_answer_latency_ms = Some(latency_ms);
2342            // A late answer is an answer: the module served the probe, just past
2343            // the deadline. Leaving the miss streak in place while logging
2344            // "proves the module is alive" is how a CPU-starved module that
2345            // answers every probe a few seconds late still marches to the
2346            // threshold and gets killed — the exact kill class `NoAnswer` is
2347            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2348            // is degradation, and degradation reports; it does not restart.
2349            state.health.consecutive_failures = 0;
2350        })?;
2351        Ok(true)
2352    }
2353
2354    /// Arm the one-shot marker for the module process that this caller
2355    /// deliberately initiated severance against. Generic connection teardown
2356    /// must not call this:
2357    /// a surviving process would otherwise retain an exemption for a later
2358    /// genuine crash.
2359    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2360        let Some(module) = self.get(module_id) else {
2361            return Ok(false);
2362        };
2363        let snapshot = lock_snapshot(&module.inner.snapshot)?;
2364        let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2365            return Ok(false);
2366        };
2367        drop(snapshot);
2368        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2369    }
2370
2371    pub fn list(&self) -> Vec<SupervisedModule> {
2372        let modules = self
2373            .modules
2374            .lock()
2375            .unwrap_or_else(|poisoned| poisoned.into_inner());
2376        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2377        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2378        modules
2379    }
2380
2381    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2382        self.spawn_nonces
2383            .lock()
2384            .unwrap_or_else(|poisoned| poisoned.into_inner())
2385            .remove(module_id);
2386        self.close_swap(module_id);
2387        let mut reserved_nonces = self
2388            .reserved_nonces
2389            .lock()
2390            .unwrap_or_else(|poisoned| poisoned.into_inner());
2391        if reserved_nonces.contains_key(module_id) {
2392            // The old nonce must die with the removed process, but the exact-id
2393            // gate remains until an operator explicitly releases it.
2394            reserved_nonces.insert(module_id.to_string(), None);
2395        }
2396        drop(reserved_nonces);
2397        self.reserved_prefix_owners
2398            .lock()
2399            .unwrap_or_else(|poisoned| poisoned.into_inner())
2400            .retain(|_, owner| owner != module_id);
2401        let removed = self
2402            .modules
2403            .lock()
2404            .unwrap_or_else(|poisoned| poisoned.into_inner())
2405            .remove(module_id);
2406        self.configured_ids
2407            .lock()
2408            .unwrap_or_else(|poisoned| poisoned.into_inner())
2409            .remove(module_id);
2410        removed
2411    }
2412
2413    /// Remember a module removed by a non-preview rescan so route.open can
2414    /// distinguish that intentional removal from an unknown id.
2415    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2416        self.removal_tombstones
2417            .lock()
2418            .unwrap_or_else(|poisoned| poisoned.into_inner())
2419            .insert(module_id.to_string(), unix_ms_now());
2420    }
2421
2422    /// Return how long ago a rescan removed this module in milliseconds.
2423    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2424        self.removal_tombstones
2425            .lock()
2426            .unwrap_or_else(|poisoned| poisoned.into_inner())
2427            .get(module_id)
2428            .copied()
2429            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2430    }
2431
2432    /// Retire a reserved-id gate only after its module has left supervision.
2433    ///
2434    /// A retained gate has no live nonce (`None`), so releasing any other entry
2435    /// would weaken a currently configured or otherwise active reservation.
2436    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2437        if self.get(module_id).is_some() {
2438            return false;
2439        }
2440        let mut reserved_nonces = self
2441            .reserved_nonces
2442            .lock()
2443            .unwrap_or_else(|poisoned| poisoned.into_inner());
2444        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2445            return false;
2446        }
2447        reserved_nonces.remove(module_id);
2448        true
2449    }
2450
2451    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2452        Arc::clone(&self.operation_lock)
2453    }
2454}
2455
2456/// Process supervisor for subc-owned singleton modules.
2457#[derive(Debug, Clone)]
2458pub struct Supervisor {
2459    registry: Arc<Registry>,
2460    restart_policy: RestartPolicy,
2461    drain_timeout: Duration,
2462    connection_file_path: Option<PathBuf>,
2463    capture_logs_dir: Option<PathBuf>,
2464    forwarding: Option<Arc<ForwardingTable>>,
2465    process_liveness: Arc<SupervisorProcessLiveness>,
2466    supervisor_handle: Option<SupervisorHandle>,
2467    health: HealthConfig,
2468    daemon_start_clock: crate::clock::StartClock,
2469    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2470    spawn_events: SpawnEventFeed,
2471    provenance_probe: ExecutableIdentityProbe,
2472    /// Every process spawned through this supervisor (and its clones) and not
2473    /// yet reaped, so daemon shutdown can end them.
2474    child_roster: ChildRoster,
2475    #[cfg(target_os = "linux")]
2476    cgroup_placement: Option<subc_cgroup::Placement>,
2477    #[cfg(test)]
2478    test_after_first_spawn: AfterFirstSpawnHook,
2479}
2480
2481/// Test-only hook run on the path that takes on a new module, right after its
2482/// first `spawn_child` returns (the process exists and could already be
2483/// registering) and before that process is handed to the module's supervise
2484/// loop and put on the roster. Lets a test observe what a fast child would see
2485/// in that window without racing a real one.
2486#[cfg(test)]
2487#[derive(Clone, Default)]
2488struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2489
2490#[cfg(test)]
2491type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2492
2493#[cfg(test)]
2494impl fmt::Debug for AfterFirstSpawnHook {
2495    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2496        f.write_str("AfterFirstSpawnHook")
2497    }
2498}
2499
2500#[cfg(test)]
2501impl AfterFirstSpawnHook {
2502    fn run(&self, module_id: &str) {
2503        if let Some(hook) = &self.0 {
2504            hook(module_id);
2505        }
2506    }
2507}
2508
2509impl Supervisor {
2510    #[cfg(test)]
2511    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2512        let supervisor = Self::new(registry, policy);
2513        #[cfg(target_os = "macos")]
2514        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2515        supervisor
2516    }
2517    /// Verify the trampoline once when configured. Missing private OS support
2518    /// refuses every macOS launch by name but does not stop the daemon's control
2519    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2520    /// entry point; the library must not exec an arbitrary hosting program.
2521    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2522        let path = path.into();
2523        #[cfg(target_os = "macos")]
2524        {
2525            let result = probe_privacy_trampoline(&path).map(|()| path);
2526            if let Err(cause) = &result {
2527                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2528            }
2529            self.child_roster.set_privacy_trampoline(result);
2530        }
2531        #[cfg(not(target_os = "macos"))]
2532        let _ = path;
2533        self
2534    }
2535    /// The first step of an announced daemon shutdown, before the notice and
2536    /// before any connection is closed.
2537    ///
2538    /// Sets the daemon-shutdown flag first: from here on no module is
2539    /// respawned (crash restart, operator restart, or swap), and every child
2540    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2541    /// the module exits on the EOF this shutdown gives it or is signalled by a
2542    /// service manager that kills the whole cgroup. Then writes the journal's
2543    /// shutdown marker, which records the instant and closes this daemon
2544    /// incarnation's stretch of the journal.
2545    #[cfg(unix)]
2546    pub(crate) fn begin_daemon_shutdown(&self) {
2547        self.child_roster.close();
2548        if let Some(journal) = &self.terminal_journal {
2549            journal.stamp_shutdown();
2550        }
2551    }
2552
2553    /// Announce a cut while established connections can still carry replies.
2554    /// These budgets promise notice and a bounded wait, not child completion;
2555    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2556    #[cfg(unix)]
2557    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2558        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2559        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2560        let Some(forwarding) = &self.forwarding else {
2561            return Ok(());
2562        };
2563        let module_ids = forwarding
2564            .begin_daemon_drain()
2565            .map_err(SuperviseError::Forwarding)?;
2566        let deadline_ms =
2567            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2568        let mut notices = tokio::task::JoinSet::new();
2569        let mut drains = Vec::new();
2570        for module_id in module_ids {
2571            let Some(target) = forwarding
2572                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2573                .map_err(SuperviseError::Forwarding)?
2574            else {
2575                continue;
2576            };
2577            let routes = forwarding
2578                .endpoint_routes(target.endpoint)
2579                .map_err(SuperviseError::Forwarding)?;
2580            // Restart allows deployed consumers to reopen after the new daemon
2581            // appears. The wire reason stays `restart`; what tells a daemon cut
2582            // apart from a module restart afterwards is the terminal record
2583            // itself, whose disposition is `daemon_shutdown` for every exit
2584            // observed once `begin_daemon_shutdown` has run.
2585            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2586                reason: RouteCloseReason::Restart,
2587                deadline_ms,
2588            })
2589            .expect("module draining serializes");
2590            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2591            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2592            for route in routes {
2593                let client = route.goodbye_target;
2594                if let Some((_, channels)) = clients
2595                    .iter_mut()
2596                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2597                {
2598                    channels.push(client.channel);
2599                } else {
2600                    let channel = client.channel;
2601                    clients.push((client, vec![channel]));
2602                }
2603            }
2604            for (client, mut channels) in clients {
2605                channels.sort_unstable();
2606                channels.dedup();
2607                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2608                    module_id: module_id.clone(),
2609                    channels,
2610                    reason: RouteCloseReason::Restart,
2611                })
2612                .expect("route closing serializes");
2613                recipients.push((client.sink, client.negotiated_ver, closing));
2614            }
2615            for (sink, version, body) in recipients {
2616                notices.spawn(async move {
2617                    let frame = Frame::build_with_version(
2618                        version,
2619                        FrameType::Push,
2620                        control_flags(),
2621                        0,
2622                        0,
2623                        0,
2624                        body,
2625                    )
2626                    .expect("bounded lifecycle notice frame builds");
2627                    sink.send_flushed(frame).await
2628                });
2629            }
2630            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2631            drains.push((module_id, target.endpoint, gauges));
2632        }
2633        // A quiet forwarding table is not proof that queued notices reached the
2634        // socket. Wait for writer flush acknowledgements before testing quiescence.
2635        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2636        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2637            if !matches!(result, Ok(Ok(()))) {
2638                warn!(?result, "daemon shutdown notice delivery failed");
2639            }
2640        }
2641        notices.abort_all();
2642        let deadline = Instant::now() + DRAIN_BUDGET;
2643        let mut waits = tokio::task::JoinSet::new();
2644        for (module_id, endpoint, gauges) in drains {
2645            let forwarding = Arc::clone(forwarding);
2646            let mut runtime = self.runtime_config();
2647            runtime.health.cadence = Duration::from_millis(100);
2648            waits.spawn(async move {
2649                wait_for_forwarding_quiescence(
2650                    &forwarding,
2651                    &module_id,
2652                    &runtime,
2653                    endpoint,
2654                    deadline,
2655                    &gauges,
2656                    DrainScope::Active,
2657                )
2658                .await
2659            });
2660        }
2661        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2662            if !matches!(result, Ok(Ok(true))) {
2663                warn!(?result, "daemon shutdown drain did not reach quiescence");
2664            }
2665        }
2666        Ok(())
2667    }
2668
2669    /// The last step of an announced daemon shutdown, after the notice and the
2670    /// drain: send every registered module a module GOODBYE, the same planned
2671    /// stop signal `ck module stop` gives, then close every connection so each
2672    /// subc module sees EOF and starts its own teardown, then end every
2673    /// supervised child that has not exited
2674    /// by its own deadline (its drain budget, capped). Modules lead their own
2675    /// process groups, so a
2676    /// service manager's group kill no longer reaches them; without this a
2677    /// child that does not stop on EOF (every `protocol: "none"` child, which
2678    /// has no connection) would outlive the daemon. Every wait is bounded (see
2679    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2680    #[cfg(unix)]
2681    pub(crate) async fn end_children_for_daemon_shutdown(
2682        &self,
2683        already_escalated: bool,
2684        escalate: impl std::future::Future<Output = ()>,
2685    ) {
2686        tokio::pin!(escalate);
2687        let mut escalated = already_escalated;
2688        if let Some(forwarding) = &self.forwarding {
2689            let reason = CloseReason::new(
2690                "daemon_shutdown",
2691                "the daemon is exiting after its shutdown notice and drain",
2692            );
2693            if escalated {
2694                // The operator asked to stop waiting: queue the GOODBYEs but
2695                // do not wait for them to be written.
2696                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2697            } else {
2698                tokio::select! {
2699                    biased;
2700                    _ = escalate.as_mut() => {
2701                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2702                        escalated = true;
2703                    }
2704                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2705                }
2706            }
2707            let closed = forwarding.close_all_connections(&reason);
2708            debug!(closed, "closed established connections for daemon shutdown");
2709        }
2710        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2711        // already completed and must not be polled again; the child shutdown
2712        // wait is told it is escalated and gets a future that never fires.
2713        let escalated_here = escalated && !already_escalated;
2714        let remaining_escalate = async move {
2715            if escalated_here {
2716                std::future::pending::<()>().await;
2717            } else {
2718                escalate.await;
2719            }
2720        };
2721        crate::child_roster::end_children_for_daemon_shutdown(
2722            &self.child_roster,
2723            escalated,
2724            remaining_escalate,
2725        )
2726        .await;
2727    }
2728
2729    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2730        Self {
2731            registry,
2732            restart_policy,
2733            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2734            connection_file_path: None,
2735            capture_logs_dir: None,
2736            forwarding: None,
2737            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2738            supervisor_handle: None,
2739            health: HealthConfig::default(),
2740            daemon_start_clock: crate::clock::StartClock::capture(),
2741            terminal_journal: None,
2742            spawn_events: SpawnEventFeed::default(),
2743            provenance_probe: ExecutableIdentityProbe::default(),
2744            child_roster: ChildRoster::default(),
2745            #[cfg(target_os = "linux")]
2746            cgroup_placement: None,
2747            #[cfg(test)]
2748            test_after_first_spawn: AfterFirstSpawnHook::default(),
2749        }
2750    }
2751
2752    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2753        self.drain_timeout = drain_timeout;
2754        self
2755    }
2756
2757    pub fn with_process_liveness(
2758        mut self,
2759        process_liveness: Arc<SupervisorProcessLiveness>,
2760    ) -> Self {
2761        self.process_liveness = process_liveness;
2762        self
2763    }
2764
2765    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2766        self.connection_file_path = Some(connection_file_path.into());
2767        self
2768    }
2769
2770    /// Enables daemon-owned capture files for supervised stdout and stderr.
2771    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2772        self.capture_logs_dir = Some(logs_dir.into());
2773        self
2774    }
2775
2776    /// Names this daemon lifetime in spawn events, independently of whether a
2777    /// terminal journal is configured.
2778    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2779        // A millisecond start stamp can repeat after clock rollback or a rapid
2780        // restart. Use the connection file's random daemon_id instead: it already
2781        // identifies this daemon lifetime independently of the wall clock.
2782        self.spawn_events.configure_incarnation(daemon_incarnation);
2783        self
2784    }
2785
2786    /// Enables best-effort history shared by every supervised module. Without
2787    /// it, terminal history is kept only in each module's in-memory ring.
2788    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2789        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2790        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2791            path,
2792            daemon_incarnation,
2793        )));
2794        this
2795    }
2796
2797    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2798        self.forwarding = Some(forwarding);
2799        self
2800    }
2801
2802    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2803        self.spawn_events = supervisor_handle.spawn_events.clone();
2804        self.supervisor_handle = Some(supervisor_handle);
2805        self
2806    }
2807
2808    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2809        self.health = health;
2810        self
2811    }
2812
2813    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2814    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2815    /// record is kept.
2816    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2817        self.child_roster.record_to(path.into());
2818        self
2819    }
2820
2821    #[cfg(target_os = "linux")]
2822    pub fn with_cgroup_placement(
2823        mut self,
2824        cgroup_placement: Option<subc_cgroup::Placement>,
2825    ) -> Self {
2826        self.cgroup_placement = cgroup_placement;
2827        self
2828    }
2829
2830    /// Spawn `spec.program` and start monitoring it.
2831    ///
2832    /// The child is expected to parse `--subc <connection-file-path>`, read the
2833    /// TCP+key connection file, authenticate to the already-running listener, and
2834    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2835    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2836        validate_spec(&spec)?;
2837        self.establish_identity(&spec);
2838
2839        let runtime = self.runtime_config();
2840        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2841        let spawned = spawn_child(
2842            &spec,
2843            runtime.connection_file_path.as_deref(),
2844            self.supervisor_handle.as_ref(),
2845            &runtime.stderr_ring,
2846            runtime.capture_logs_dir.as_deref(),
2847            &runtime.child_roster,
2848            #[cfg(target_os = "linux")]
2849            runtime.cgroup_placement.as_ref(),
2850        );
2851        #[cfg(test)]
2852        self.test_after_first_spawn.run(&spec.module_id);
2853        let child = match spawned {
2854            Ok(child) => child,
2855            Err(err) => {
2856                // Unlike the configured paths, a failed `spawn` leaves nothing
2857                // on the roster, so the module must not stay marked configured.
2858                self.abandon_unrostered(&spec.module_id);
2859                return Err(err);
2860            }
2861        };
2862        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2863
2864        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2865    }
2866
2867    /// Make `spec`'s module count as configured, with its identity gates
2868    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2869    /// exists.
2870    ///
2871    /// Every path that takes on a new module calls this before `spawn_child`.
2872    /// The order is the point: the child can connect, register, sync its
2873    /// scopes and ask about them as soon as it is spawned, and the module is
2874    /// only put on the roster after `spawn_child` returns. Were the mark set
2875    /// with the roster entry, a fast child would see its own owner reported
2876    /// as not configured, and a scoped `route.open` in that window would be
2877    /// refused as terminal `scope_not_live` ("will never sync") instead of
2878    /// retryable `scope_not_synced`.
2879    fn establish_identity(&self, spec: &ModuleSpec) {
2880        if let Some(supervisor_handle) = &self.supervisor_handle {
2881            supervisor_handle.apply_identity_configuration(spec);
2882            supervisor_handle.mark_configured(&spec.module_id);
2883        }
2884    }
2885
2886    /// Take back [`Self::establish_identity`]'s configured mark when the
2887    /// module will not be put on the roster after all.
2888    fn abandon_unrostered(&self, module_id: &str) {
2889        if let Some(supervisor_handle) = &self.supervisor_handle {
2890            supervisor_handle.unmark_configured_unless_rostered(module_id);
2891        }
2892    }
2893
2894    /// Record a freshly spawned first process as running. On failure the
2895    /// module never reaches the roster, so its configured mark is taken back.
2896    fn mark_first_process_running(
2897        &self,
2898        spec: &ModuleSpec,
2899        runtime: &SupervisorRuntimeConfig,
2900        snapshot: &SharedSnapshot,
2901        child: &SupervisedChild,
2902    ) -> Result<(), SuperviseError> {
2903        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2904            self.abandon_unrostered(&spec.module_id);
2905            return Err(err);
2906        }
2907        self.process_liveness
2908            .track(spec.module_id.clone(), Arc::clone(snapshot));
2909        Ok(())
2910    }
2911
2912    /// Start supervising a module declared in daemon configuration.
2913    ///
2914    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2915    /// failures in the supervisor handle so operator-facing `supervisor.list`
2916    /// reflects every configured module while daemon startup continues.
2917    pub fn supervise_configured(
2918        &self,
2919        spec: ModuleSpec,
2920        enabled: bool,
2921    ) -> Result<SupervisedModule, SuperviseError> {
2922        validate_spec(&spec)?;
2923        self.establish_identity(&spec);
2924
2925        let runtime = self.runtime_config();
2926        if !enabled {
2927            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2928            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2929        }
2930
2931        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2932        let spawned = spawn_child(
2933            &spec,
2934            runtime.connection_file_path.as_deref(),
2935            self.supervisor_handle.as_ref(),
2936            &runtime.stderr_ring,
2937            runtime.capture_logs_dir.as_deref(),
2938            &runtime.child_roster,
2939            #[cfg(target_os = "linux")]
2940            runtime.cgroup_placement.as_ref(),
2941        );
2942        #[cfg(test)]
2943        self.test_after_first_spawn.run(&spec.module_id);
2944        match spawned {
2945            Ok(child) => {
2946                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2947                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2948            }
2949            Err(err) => {
2950                error!(
2951                    module_id = %spec.module_id,
2952                    program = %spec.program.display(),
2953                    error = %err,
2954                    "configured module failed to spawn; marking failed and continuing"
2955                );
2956                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2957                Ok(self.supervised_module(spec, runtime, snapshot, None))
2958            }
2959        }
2960    }
2961
2962    /// Supervise a configured module with its own health, drain, and crash
2963    /// budget. The restart policy is per-module because the config file is:
2964    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2965    /// module that is expensive to restart should not be forced onto the same
2966    /// budget as one that is cheap.
2967    pub fn supervise_configured_with_health(
2968        &self,
2969        spec: ModuleSpec,
2970        enabled: bool,
2971        health: HealthConfig,
2972        drain_timeout_ms: Option<u64>,
2973        restart_policy: RestartPolicy,
2974    ) -> Result<SupervisedModule, SuperviseError> {
2975        validate_spec(&spec)?;
2976        self.establish_identity(&spec);
2977
2978        let mut runtime = self.runtime_config();
2979        runtime.health = health.clone();
2980        runtime.restart_policy = restart_policy;
2981        if let Some(ms) = drain_timeout_ms {
2982            runtime.drain_timeout = Duration::from_millis(ms);
2983            *runtime
2984                .effective_drain_timeout
2985                .lock()
2986                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2987        }
2988        if !enabled {
2989            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2990            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2991        }
2992
2993        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2994        let spawned = spawn_child(
2995            &spec,
2996            runtime.connection_file_path.as_deref(),
2997            self.supervisor_handle.as_ref(),
2998            &runtime.stderr_ring,
2999            runtime.capture_logs_dir.as_deref(),
3000            &runtime.child_roster,
3001            #[cfg(target_os = "linux")]
3002            runtime.cgroup_placement.as_ref(),
3003        );
3004        #[cfg(test)]
3005        self.test_after_first_spawn.run(&spec.module_id);
3006        match spawned {
3007            Ok(child) => {
3008                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3009                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3010            }
3011            Err(err) => {
3012                if health.critical {
3013                    error!(
3014                        module_id = %spec.module_id,
3015                        program = %spec.program.display(),
3016                        error = %err,
3017                        "critical configured module failed to spawn; marking failed and alerting"
3018                    );
3019                } else {
3020                    error!(
3021                        module_id = %spec.module_id,
3022                        program = %spec.program.display(),
3023                        error = %err,
3024                        "configured module failed to spawn; marking failed and continuing"
3025                    );
3026                }
3027                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3028                Ok(self.supervised_module(spec, runtime, snapshot, None))
3029            }
3030        }
3031    }
3032
3033    fn runtime_config(&self) -> SupervisorRuntimeConfig {
3034        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3035        SupervisorRuntimeConfig {
3036            scheduled_respawn: Arc::default(),
3037            deferred_reload_reply: Arc::default(),
3038            restart_policy: self.restart_policy,
3039            drain_timeout: self.drain_timeout,
3040            // Shared with this module's roster copy: daemon shutdown waits on
3041            // each child for the module's own drain budget, as resolved now.
3042            child_roster: self
3043                .child_roster
3044                .for_module(Arc::clone(&effective_drain_timeout)),
3045            effective_drain_timeout,
3046            default_drain_timeout: self.drain_timeout,
3047            health: self.health.clone(),
3048            connection_file_path: self.connection_file_path.clone(),
3049            capture_logs_dir: self.capture_logs_dir.clone(),
3050            forwarding: self.forwarding.clone(),
3051            supervisor_handle: self.supervisor_handle.clone(),
3052            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3053            terminal_ring: Arc::new(Mutex::new(
3054                TerminalRing::new(
3055                    TerminalRingConfig::default(),
3056                    self.daemon_start_clock.started_at_ms(),
3057                )
3058                .with_start_clock(self.daemon_start_clock)
3059                .with_journal(self.terminal_journal.clone())
3060                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3061            )),
3062            spawn_events: self.spawn_events.clone(),
3063            #[cfg(target_os = "linux")]
3064            cgroup_placement: self.cgroup_placement.clone(),
3065            #[cfg(test)]
3066            test_seed_stale_facts_before_enable_spawn: false,
3067            #[cfg(test)]
3068            test_reload_exit_record_gate: None,
3069        }
3070    }
3071
3072    fn supervised_module(
3073        &self,
3074        spec: ModuleSpec,
3075        runtime: SupervisorRuntimeConfig,
3076        snapshot: SharedSnapshot,
3077        child: Option<SupervisedChild>,
3078    ) -> SupervisedModule {
3079        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3080            spec: spec.clone(),
3081            health: runtime.health.clone(),
3082        }));
3083        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3084        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3085        // The module's OWN policy, which may be its per-module config rather than
3086        // the supervisor-wide one; status must report the budget the supervise
3087        // loop actually enforces.
3088        let restart_policy = runtime.restart_policy;
3089        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3090        let (tx, rx) = mpsc::channel(4);
3091        let monitor = tokio::spawn(supervise_loop(
3092            spec.clone(),
3093            runtime,
3094            Arc::clone(&self.registry),
3095            Arc::clone(&self.process_liveness),
3096            Arc::clone(&snapshot),
3097            child,
3098            rx,
3099        ));
3100
3101        let module_id = spec.module_id.clone();
3102        let module = SupervisedModule {
3103            inner: Arc::new(SupervisedModuleInner {
3104                module_id: module_id.clone(),
3105                registry: Arc::clone(&self.registry),
3106                snapshot,
3107                configuration,
3108                stderr_ring,
3109                terminal_ring,
3110                commands: tx,
3111                monitor: Mutex::new(Some(monitor)),
3112                restart_policy,
3113                effective_drain_timeout,
3114                provenance_probe: self.provenance_probe.clone(),
3115            }),
3116        };
3117        // The identity gates and the configured mark were set by
3118        // `establish_identity` before any process was spawned; only the roster
3119        // entry waits for the module handle, which needs the spawned child.
3120        if let Some(supervisor_handle) = &self.supervisor_handle {
3121            supervisor_handle.insert(module.clone());
3122        }
3123        module
3124    }
3125}
3126
3127impl Default for Supervisor {
3128    fn default() -> Self {
3129        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3130    }
3131}
3132
3133/// Handle to one supervised child process.
3134#[derive(Clone)]
3135pub struct SupervisedModule {
3136    inner: Arc<SupervisedModuleInner>,
3137}
3138
3139struct SupervisedModuleInner {
3140    module_id: String,
3141    registry: Arc<Registry>,
3142    snapshot: SharedSnapshot,
3143    configuration: Arc<Mutex<SupervisedConfiguration>>,
3144    stderr_ring: Arc<Mutex<StderrRing>>,
3145    terminal_ring: Arc<Mutex<TerminalRing>>,
3146    commands: mpsc::Sender<SupervisorCommand>,
3147    monitor: Mutex<Option<JoinHandle<()>>>,
3148    /// Copied from the supervisor's runtime config at spawn so `status()` can
3149    /// report the restart budget without reaching back into the supervisor. The
3150    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3151    restart_policy: RestartPolicy,
3152    effective_drain_timeout: Arc<Mutex<Duration>>,
3153    provenance_probe: ExecutableIdentityProbe,
3154}
3155
3156impl fmt::Debug for SupervisedModule {
3157    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3158        f.debug_struct("SupervisedModule")
3159            .field("module_id", &self.inner.module_id)
3160            .field("status", &self.status())
3161            .finish_non_exhaustive()
3162    }
3163}
3164
3165impl SupervisedModule {
3166    pub fn module_id(&self) -> &str {
3167        &self.inner.module_id
3168    }
3169
3170    /// Test-only: put one probe miss on the streak, the way
3171    /// `handle_health_probe_failure` does, so tests can assert what a later
3172    /// event does to the streak without driving the whole probe loop.
3173    #[cfg(test)]
3174    pub(crate) fn record_health_probe_failure_for_test(
3175        &self,
3176        detail: &str,
3177    ) -> Result<(), SuperviseError> {
3178        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3179            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3180            state.health.detail = Some(detail.to_string());
3181        })
3182    }
3183
3184    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3185        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3186    }
3187
3188    /// The module's retained stderr, newest lines last.
3189    ///
3190    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3191    /// module, `supervisor.list` renders every module, and putting it in the
3192    /// shared snapshot would make each status read carry a payload almost nobody
3193    /// asked for. Callers that want the text ask for it.
3194    pub fn stderr_tail(
3195        &self,
3196        max_lines: Option<usize>,
3197        max_bytes: Option<usize>,
3198    ) -> StderrTailSnapshot {
3199        self.inner
3200            .stderr_ring
3201            .lock()
3202            .unwrap_or_else(|poisoned| poisoned.into_inner())
3203            .snapshot(max_lines, max_bytes)
3204    }
3205
3206    /// The module's bounded terminal history, oldest retained exit first.
3207    ///
3208    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3209    /// daemon whose in-memory history was necessarily reset.
3210    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3211        self.inner
3212            .terminal_ring
3213            .lock()
3214            .unwrap_or_else(|poisoned| poisoned.into_inner())
3215            .snapshot()
3216    }
3217
3218    /// Retained observations from the current ring and all journal generations.
3219    ///
3220    /// Blocking: this reads the journal files. Async callers use
3221    /// [`Self::read_durable_terminal_history`].
3222    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3223        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3224    }
3225
3226    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3227    /// read (up to every retained generation) never occupies a runtime worker.
3228    /// Fails only if the blocking task could not finish (runtime shutdown or a
3229    /// panic in the read).
3230    pub(crate) async fn read_durable_terminal_history(
3231        &self,
3232    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3233        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3234        let module_id = self.inner.module_id.clone();
3235        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3236            .await
3237    }
3238
3239    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3240        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3241            .map(|(status, _)| status)
3242    }
3243
3244    pub(crate) fn record_deliberate_severance(
3245        &self,
3246        identity: ProcessIdentity,
3247    ) -> Result<bool, SuperviseError> {
3248        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3249        if snapshot.pid != Some(identity.pid)
3250            || snapshot.process_start_time != Some(identity.start_time)
3251        {
3252            return Ok(false);
3253        }
3254        snapshot.deliberate_severance = Some(identity);
3255        Ok(true)
3256    }
3257
3258    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3259    ///
3260    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3261    /// does not produce reader-observability logs.
3262    pub(crate) fn status_for_control(
3263        &self,
3264        caller: &'static str,
3265    ) -> Result<ModuleStatus, SuperviseError> {
3266        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3267            .map(|(status, _)| status)
3268    }
3269
3270    fn status_with_snapshot_lock(
3271        &self,
3272        snapshot: &SharedSnapshot,
3273        caller: Option<&'static str>,
3274    ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3275        let mut guard = match caller {
3276            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3277            None => lock_snapshot(snapshot)?,
3278        };
3279        // Read the budget through the pruning path so a reader sees the same
3280        // in-window count the restart decision would use, not a stale total.
3281        let restart_count =
3282            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3283        let snapshot = guard.clone();
3284        drop(guard);
3285        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3286            SuperviseError::StatePoisoned {
3287                module_id: Some(self.inner.module_id.clone()),
3288            }
3289        })?;
3290        let registration_active = self
3291            .inner
3292            .registry
3293            .get_module(&self.inner.module_id)
3294            .map_err(SuperviseError::Registry)?
3295            .is_some();
3296        let protocol = snapshot
3297            .spawned_protocol
3298            .unwrap_or(self.declared_protocol()?);
3299        let running_process =
3300            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3301        // Registration is the difference between the two protocols and the only
3302        // one: a subc module that has not registered cannot serve a request even
3303        // though its process is up, and a `none` module never registers at all,
3304        // so requiring it there would pin `live` to false for the whole life of
3305        // a perfectly healthy process.
3306        let live = match protocol {
3307            ModuleProtocol::Subc => running_process && registration_active,
3308            ModuleProtocol::None => running_process,
3309        };
3310
3311        Ok((
3312            ModuleStatus {
3313                module_id: self.inner.module_id.clone(),
3314                state: snapshot.state,
3315                enabled: snapshot.enabled,
3316                process_alive: snapshot.process_alive,
3317                registration_active,
3318                protocol,
3319                live,
3320                restart_count,
3321                lifetime_restarts: snapshot.lifetime_restarts,
3322                spawn_generation: snapshot.spawn_generation,
3323                max_restarts: self.inner.restart_policy.max_restarts,
3324                restart_window: self.inner.restart_policy.window,
3325                drain_timeout,
3326                restart_backoff: self.inner.restart_policy.backoff,
3327                restart_max_backoff: self.inner.restart_policy.max_backoff,
3328                pid: snapshot.reported_pid(),
3329                spawned_at_ms: snapshot.spawned_at_ms,
3330                spawned_from: snapshot.spawned_from,
3331                process_start_time: snapshot.process_start_time,
3332                last_exit: snapshot.last_exit,
3333                health: snapshot.health,
3334            },
3335            snapshot.spawned_file_identity,
3336        ))
3337    }
3338
3339    #[cfg(test)]
3340    pub(crate) fn hold_snapshot_for_test(
3341        &self,
3342        acquired: std::sync::mpsc::Sender<()>,
3343        hold: Duration,
3344    ) -> std::thread::JoinHandle<()> {
3345        let snapshot = Arc::clone(&self.inner.snapshot);
3346        std::thread::spawn(move || {
3347            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3348            acquired
3349                .send(())
3350                .expect("test receiver waits for snapshot lock");
3351            std::thread::sleep(hold);
3352        })
3353    }
3354
3355    /// The status and the running-image check for `supervisor.provenance`,
3356    /// taken from one status read. The exec acknowledgement can land between
3357    /// two separate reads, and the reply would then pair "no pid yet" with an
3358    /// image observed after the module started, which describes no single
3359    /// moment.
3360    pub(crate) async fn status_and_running_image_agreement(
3361        &self,
3362    ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3363        let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3364        let image = self
3365            .inner
3366            .provenance_probe
3367            .observe(
3368                status.pid,
3369                status.spawned_from.as_deref(),
3370                identity,
3371                status.process_start_time,
3372            )
3373            .await;
3374        Ok((status, image))
3375    }
3376
3377    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3378        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3379            Ok(snapshot) => snapshot.clone(),
3380            Err(_) => {
3381                return subc_control::RunningImageAgreement::Unavailable {
3382                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3383                };
3384            }
3385        };
3386        self.inner
3387            .provenance_probe
3388            .observe(
3389                snapshot.reported_pid(),
3390                snapshot.spawned_from.as_deref(),
3391                snapshot.spawned_file_identity,
3392                snapshot.process_start_time,
3393            )
3394            .await
3395    }
3396
3397    /// Memory and CPU time of the module's current process, read now. Only the
3398    /// process the supervisor spawned is read, not processes it has started.
3399    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3400        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3401            Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3402            Err(_) => {
3403                return subc_control::ChildResourceUsage::Unavailable {
3404                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3405                }
3406            }
3407        };
3408        crate::child_resources::read(pid, start_time)
3409    }
3410
3411    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3412        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3413        Ok(match snapshot.state {
3414            ModuleState::Restarting => true,
3415            ModuleState::Failed | ModuleState::Disabled => false,
3416            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3417        })
3418    }
3419
3420    #[cfg(test)]
3421    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3422        self.is_warming_with_snapshot_lock(None)
3423    }
3424
3425    pub(crate) fn is_warming_for_control(
3426        &self,
3427        caller: &'static str,
3428    ) -> Result<bool, SuperviseError> {
3429        self.is_warming_with_snapshot_lock(Some(caller))
3430    }
3431
3432    fn is_warming_with_snapshot_lock(
3433        &self,
3434        caller: Option<&'static str>,
3435    ) -> Result<bool, SuperviseError> {
3436        let snapshot = match caller {
3437            Some(caller) => {
3438                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3439            }
3440            None => lock_snapshot(&self.inner.snapshot)?,
3441        }
3442        .clone();
3443        Ok(matches!(
3444            snapshot.state,
3445            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3446        ))
3447    }
3448
3449    /// Drain the module and stop monitoring it.
3450    pub async fn drain(&self) -> Result<(), SuperviseError> {
3451        self.stop().await
3452    }
3453
3454    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3455        match self.state()? {
3456            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3457            ModuleState::Starting
3458            | ModuleState::Running
3459            | ModuleState::Unresponsive
3460            | ModuleState::Restarting
3461            | ModuleState::Draining
3462            | ModuleState::Disabled => {}
3463        }
3464
3465        let (reply_tx, reply_rx) = oneshot::channel();
3466        self.inner
3467            .commands
3468            .send(SupervisorCommand::Retire { reply: reply_tx })
3469            .await
3470            .map_err(|_| SuperviseError::CommandClosed {
3471                module_id: self.inner.module_id.clone(),
3472            })?;
3473        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3474            module_id: self.inner.module_id.clone(),
3475        })?
3476    }
3477
3478    pub async fn stop(&self) -> Result<(), SuperviseError> {
3479        match self.state()? {
3480            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3481            ModuleState::Starting
3482            | ModuleState::Running
3483            | ModuleState::Unresponsive
3484            | ModuleState::Restarting
3485            | ModuleState::Draining
3486            | ModuleState::Disabled => {}
3487        }
3488
3489        let (reply_tx, reply_rx) = oneshot::channel();
3490        self.inner
3491            .commands
3492            .send(SupervisorCommand::Drain { reply: reply_tx })
3493            .await
3494            .map_err(|_| SuperviseError::CommandClosed {
3495                module_id: self.inner.module_id.clone(),
3496            })?;
3497        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3498            module_id: self.inner.module_id.clone(),
3499        })?
3500    }
3501
3502    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3503        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3504        let (reply_tx, reply_rx) = oneshot::channel();
3505        self.inner
3506            .commands
3507            .send(SupervisorCommand::Restart {
3508                drain_timeout_ms,
3509                received_at_generation,
3510                queued_at: Instant::now(),
3511                reply: reply_tx,
3512            })
3513            .await
3514            .map_err(|_| SuperviseError::CommandClosed {
3515                module_id: self.inner.module_id.clone(),
3516            })?;
3517        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3518            module_id: self.inner.module_id.clone(),
3519        })?
3520    }
3521
3522    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3523    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3524    /// process then drains in the background of the supervise loop) or has
3525    /// failed, leaving the old process serving.
3526    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3527        let (reply_tx, reply_rx) = oneshot::channel();
3528        self.inner
3529            .commands
3530            .send(SupervisorCommand::Swap {
3531                ready_timeout,
3532                reply: reply_tx,
3533            })
3534            .await
3535            .map_err(|_| SuperviseError::CommandClosed {
3536                module_id: self.inner.module_id.clone(),
3537            })?;
3538        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3539            module_id: self.inner.module_id.clone(),
3540        })?
3541    }
3542
3543    pub async fn reload(&self) -> Result<(), SuperviseError> {
3544        let (reply_tx, reply_rx) = oneshot::channel();
3545        self.inner
3546            .commands
3547            .send(SupervisorCommand::Reload { reply: reply_tx })
3548            .await
3549            .map_err(|_| SuperviseError::CommandClosed {
3550                module_id: self.inner.module_id.clone(),
3551            })?;
3552        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3553            module_id: self.inner.module_id.clone(),
3554        })?
3555    }
3556
3557    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3558        let (reply_tx, reply_rx) = oneshot::channel();
3559        self.inner
3560            .commands
3561            .send(SupervisorCommand::SetEnabled {
3562                enabled,
3563                reply: reply_tx,
3564            })
3565            .await
3566            .map_err(|_| SuperviseError::CommandClosed {
3567                module_id: self.inner.module_id.clone(),
3568            })?;
3569        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3570            module_id: self.inner.module_id.clone(),
3571        })?
3572    }
3573
3574    /// The current process's protocol, or the configured protocol when down.
3575    /// A rescan stores the next launch spec without changing how an existing
3576    /// process registers, serves routes, is probed, or exits.
3577    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3578        let configured = self
3579            .inner
3580            .configuration
3581            .lock()
3582            .map_err(|_| SuperviseError::StatePoisoned {
3583                module_id: Some(self.inner.module_id.clone()),
3584            })?
3585            .spec
3586            .protocol;
3587        let state = lock_snapshot(&self.inner.snapshot)?;
3588        Ok(state.spawned_protocol.unwrap_or(configured))
3589    }
3590
3591    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3592        let configuration =
3593            self.inner
3594                .configuration
3595                .lock()
3596                .map_err(|_| SuperviseError::StatePoisoned {
3597                    module_id: Some(self.inner.module_id.clone()),
3598                })?;
3599        Ok((configuration.spec.clone(), configuration.health.clone()))
3600    }
3601
3602    /// Replace this module's launch spec, keeping its health and drain policy,
3603    /// the way a rescan does for a changed config entry. The running process is
3604    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3605    #[cfg(any(test, feature = "test-support"))]
3606    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3607        let (_, health) = self.configuration()?;
3608        let drain_timeout_ms = u64::try_from(
3609            self.inner
3610                .effective_drain_timeout
3611                .lock()
3612                .unwrap_or_else(|poisoned| poisoned.into_inner())
3613                .as_millis(),
3614        )
3615        .ok();
3616        self.update_configuration(spec, health, drain_timeout_ms)
3617            .await
3618    }
3619
3620    pub(crate) async fn update_configuration(
3621        &self,
3622        spec: ModuleSpec,
3623        health: HealthConfig,
3624        drain_timeout_ms: Option<u64>,
3625    ) -> Result<(), SuperviseError> {
3626        if spec.module_id != self.inner.module_id {
3627            return Err(SuperviseError::InvalidSpec {
3628                reason: "a supervised module's module_id cannot be changed".to_string(),
3629            });
3630        }
3631        validate_spec(&spec)?;
3632        let (reply_tx, reply_rx) = oneshot::channel();
3633        self.inner
3634            .commands
3635            .send(SupervisorCommand::UpdateConfiguration {
3636                spec: spec.clone(),
3637                health: health.clone(),
3638                drain_timeout_ms,
3639                reply: reply_tx,
3640            })
3641            .await
3642            .map_err(|_| SuperviseError::CommandClosed {
3643                module_id: self.inner.module_id.clone(),
3644            })?;
3645        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3646            module_id: self.inner.module_id.clone(),
3647        })?;
3648        let mut configuration =
3649            self.inner
3650                .configuration
3651                .lock()
3652                .map_err(|_| SuperviseError::StatePoisoned {
3653                    module_id: Some(self.inner.module_id.clone()),
3654                })?;
3655        configuration.spec = spec;
3656        configuration.health = health;
3657        Ok(())
3658    }
3659}
3660
3661impl Drop for SupervisedModuleInner {
3662    fn drop(&mut self) {
3663        let Ok(mut monitor) = self.monitor.lock() else {
3664            return;
3665        };
3666        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3667            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3668                state.state = ModuleState::Stopped;
3669                clear_current_process_facts(state);
3670            });
3671            monitor.abort();
3672        }
3673        let _ = monitor.take();
3674    }
3675}
3676
3677#[derive(Debug)]
3678enum SupervisorCommand {
3679    Drain {
3680        reply: oneshot::Sender<Result<(), SuperviseError>>,
3681    },
3682    Retire {
3683        reply: oneshot::Sender<Result<(), SuperviseError>>,
3684    },
3685    Restart {
3686        /// Operator override for this one restart's drain budget, in ms. `None`
3687        /// uses the module's configured/default budget; `Some(0)` cuts
3688        /// immediately (wedge bounce: a stuck request never settles, so
3689        /// waiting only delays recovery).
3690        drain_timeout_ms: Option<u64>,
3691        /// The module's `spawn_generation` when the request was received, before
3692        /// it waited in the command queue. A queued restart whose module has
3693        /// since spawned a newer process is already satisfied (see the handler).
3694        received_at_generation: u64,
3695        /// When the request entered the command queue, so the handler can log
3696        /// how long it waited behind the loop's other work.
3697        queued_at: Instant,
3698        reply: oneshot::Sender<Result<(), SuperviseError>>,
3699    },
3700    Reload {
3701        reply: oneshot::Sender<Result<(), SuperviseError>>,
3702    },
3703    SetEnabled {
3704        enabled: bool,
3705        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3706    },
3707    UpdateConfiguration {
3708        spec: ModuleSpec,
3709        health: HealthConfig,
3710        /// Per-module drain override from the new config; `None` re-resolves to
3711        /// the supervisor-wide default.
3712        drain_timeout_ms: Option<u64>,
3713        reply: oneshot::Sender<()>,
3714    },
3715    Swap {
3716        /// How long the candidate may take to register and declare itself
3717        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3718        ready_timeout: Option<Duration>,
3719        /// Answered at cutover or failure; the incumbent's drain follows.
3720        reply: oneshot::Sender<Result<(), SuperviseError>>,
3721    },
3722}
3723
3724#[derive(Debug)]
3725pub enum SuperviseError {
3726    InvalidSpec {
3727        reason: String,
3728    },
3729    Spawn {
3730        program: PathBuf,
3731        source: io::Error,
3732        cgroup_path: Option<PathBuf>,
3733    },
3734    Cgroup {
3735        module_id: String,
3736        source: io::Error,
3737    },
3738    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3739    /// than spawn a reserved module without its identity binding.
3740    LaunchNonce {
3741        reason: String,
3742    },
3743    Wait {
3744        module_id: String,
3745        source: io::Error,
3746    },
3747    Kill {
3748        module_id: String,
3749        source: io::Error,
3750    },
3751    Forwarding(ForwardingError),
3752    Registry(RegistryError),
3753    ReloadUnavailable {
3754        module_id: String,
3755        reason: String,
3756    },
3757    /// An operator restart/reload was requested for a module that is currently
3758    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3759    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3760    /// by a restart, so these commands are rejected instead of re-enabling it.
3761    Disabled {
3762        module_id: String,
3763    },
3764    ReloadFailed {
3765        module_id: String,
3766        reason: String,
3767    },
3768    RegistrationStillActive {
3769        module_id: String,
3770        waited: Duration,
3771    },
3772    StatePoisoned {
3773        module_id: Option<String>,
3774    },
3775    CommandClosed {
3776        module_id: String,
3777    },
3778    /// A restart or reload arrived while a swap's candidate was warming. The
3779    /// swap owns the module until it cuts over or fails; a stop or disable
3780    /// would have aborted it instead.
3781    SwapInProgress {
3782        module_id: String,
3783    },
3784    /// A swap was refused before anything was spawned.
3785    SwapRefused {
3786        module_id: String,
3787        reason: SwapRefusal,
3788    },
3789    /// A swap spawned a candidate and gave up on it. The candidate has been
3790    /// killed and its slot freed; the incumbent was left serving and was never
3791    /// drained, except in the one `CutoverLost` case described on that arm.
3792    SwapFailed {
3793        module_id: String,
3794        arm: SwapFailureArm,
3795        detail: String,
3796        /// How the candidate exited, when it exited on its own before the
3797        /// supervisor gave up on it.
3798        candidate_exit: Option<ExitReport>,
3799    },
3800}
3801
3802/// Why a swap was refused before a candidate was spawned.
3803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3804pub enum SwapRefusal {
3805    /// The module's config does not declare `overlap: "safe"`.
3806    OverlapExclusive,
3807    /// The module is not registered, so there is no incumbent to keep serving
3808    /// and nothing a swap would improve on; a plain restart is the tool.
3809    NotRegistered,
3810    /// The module does not speak the subc wire, so a candidate could never
3811    /// register or declare itself ready.
3812    ProtocolNone,
3813    /// The supervisor lacks the forwarding table (to cut routes over) or the
3814    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3815    NotConfigured,
3816    /// A swap is already open for this module.
3817    AlreadySwapping,
3818}
3819
3820impl SwapRefusal {
3821    pub fn as_str(self) -> &'static str {
3822        match self {
3823            Self::OverlapExclusive => "overlap_exclusive",
3824            Self::NotRegistered => "not_registered",
3825            Self::ProtocolNone => "protocol_none",
3826            Self::NotConfigured => "not_configured",
3827            Self::AlreadySwapping => "already_swapping",
3828        }
3829    }
3830}
3831
3832/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3833/// serving and undrained; see `CutoverLost`.
3834#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3835pub enum SwapFailureArm {
3836    /// The candidate process could not be started.
3837    SpawnFailed,
3838    /// The candidate did not register within the readiness budget.
3839    NeverRegistered,
3840    /// The candidate registered but did not declare itself ready in time.
3841    NeverReady,
3842    /// The candidate exited before cutover.
3843    CandidateExited,
3844    /// The candidate declared itself ready but failed its health probe.
3845    CandidateUnhealthy,
3846    /// An operator stop, disable or retire arrived while the candidate warmed.
3847    /// The candidate was killed and the operator's command then carried out on
3848    /// the incumbent.
3849    Interrupted,
3850    /// The candidate's connection closed at the moment of cutover. If it
3851    /// closed before forwarding moved, the incumbent is untouched. If it closed
3852    /// between the forwarding and registry halves of cutover, forwarding can no
3853    /// longer route to the incumbent, so the module is restarted plainly.
3854    CutoverLost,
3855}
3856
3857impl SwapFailureArm {
3858    pub fn as_str(self) -> &'static str {
3859        match self {
3860            Self::SpawnFailed => "spawn_failed",
3861            Self::NeverRegistered => "never_registered",
3862            Self::NeverReady => "never_ready",
3863            Self::CandidateExited => "candidate_exited",
3864            Self::CandidateUnhealthy => "candidate_unhealthy",
3865            Self::Interrupted => "interrupted",
3866            Self::CutoverLost => "cutover_lost",
3867        }
3868    }
3869}
3870
3871impl fmt::Display for SuperviseError {
3872    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3873        match self {
3874            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3875            Self::Spawn {
3876                program,
3877                source,
3878                cgroup_path: Some(cgroup_path),
3879            } => write!(
3880                f,
3881                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3882                cgroup_path.display(),
3883                program.display()
3884            ),
3885            Self::Spawn {
3886                program,
3887                source,
3888                cgroup_path: None,
3889            } => write!(
3890                f,
3891                "failed to spawn module '{}': {source}",
3892                program.display()
3893            ),
3894            Self::Cgroup { module_id, source } => {
3895                write!(
3896                    f,
3897                    "failed to prepare cgroup for module '{module_id}': {source}"
3898                )
3899            }
3900            Self::LaunchNonce { reason } => {
3901                write!(
3902                    f,
3903                    "failed to generate reserved-module launch nonce: {reason}"
3904                )
3905            }
3906            Self::Wait { module_id, source } => {
3907                write!(f, "failed to wait for module '{module_id}': {source}")
3908            }
3909            Self::Kill { module_id, source } => {
3910                write!(f, "failed to kill module '{module_id}': {source}")
3911            }
3912            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3913            Self::Registry(err) => write!(f, "registry error: {err}"),
3914            Self::ReloadUnavailable { module_id, reason } => {
3915                write!(f, "reload unavailable for module '{module_id}': {reason}")
3916            }
3917            Self::Disabled { module_id } => {
3918                write!(
3919                    f,
3920                    "module '{module_id}' is disabled; enable it before restart or reload"
3921                )
3922            }
3923            Self::ReloadFailed { module_id, reason } => {
3924                write!(f, "reload failed for module '{module_id}': {reason}")
3925            }
3926            Self::RegistrationStillActive { module_id, waited } => write!(
3927                f,
3928                "module '{module_id}' registration remained active after waiting {waited:?}"
3929            ),
3930            Self::StatePoisoned { module_id } => match module_id {
3931                Some(module_id) => {
3932                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3933                }
3934                None => write!(f, "supervisor state was poisoned"),
3935            },
3936            Self::CommandClosed { module_id } => {
3937                write!(
3938                    f,
3939                    "supervisor command channel for module '{module_id}' is closed"
3940                )
3941            }
3942            Self::SwapInProgress { module_id } => write!(
3943                f,
3944                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3945            ),
3946            Self::SwapRefused { module_id, reason } => match reason {
3947                SwapRefusal::OverlapExclusive => write!(
3948                    f,
3949                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3950                ),
3951                SwapRefusal::NotRegistered => write!(
3952                    f,
3953                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3954                ),
3955                SwapRefusal::ProtocolNone => write!(
3956                    f,
3957                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3958                ),
3959                SwapRefusal::NotConfigured => write!(
3960                    f,
3961                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3962                ),
3963                SwapRefusal::AlreadySwapping => {
3964                    write!(f, "module '{module_id}' is already being swapped")
3965                }
3966            },
3967            Self::SwapFailed {
3968                module_id,
3969                arm,
3970                detail,
3971                ..
3972            } => write!(
3973                f,
3974                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3975                arm.as_str()
3976            ),
3977        }
3978    }
3979}
3980
3981impl Error for SuperviseError {
3982    fn source(&self) -> Option<&(dyn Error + 'static)> {
3983        match self {
3984            Self::Spawn { source, .. }
3985            | Self::Cgroup { source, .. }
3986            | Self::Wait { source, .. }
3987            | Self::Kill { source, .. } => Some(source),
3988            Self::Forwarding(err) => Some(err),
3989            Self::Registry(err) => Some(err),
3990            Self::LaunchNonce { .. }
3991            | Self::InvalidSpec { .. }
3992            | Self::ReloadUnavailable { .. }
3993            | Self::Disabled { .. }
3994            | Self::ReloadFailed { .. }
3995            | Self::RegistrationStillActive { .. }
3996            | Self::StatePoisoned { .. }
3997            | Self::CommandClosed { .. }
3998            | Self::SwapInProgress { .. }
3999            | Self::SwapRefused { .. }
4000            | Self::SwapFailed { .. } => None,
4001        }
4002    }
4003}
4004
4005pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4006    if spec.module_id.trim().is_empty() {
4007        return Err(SuperviseError::InvalidSpec {
4008            reason: "module_id must not be empty".to_string(),
4009        });
4010    }
4011
4012    Ok(())
4013}
4014
4015#[derive(Debug, Default)]
4016struct HealthProbeRuntime {
4017    configured_health: Option<HealthConfig>,
4018    registered_connection: Option<crate::ConnectionId>,
4019    advertised: bool,
4020    next_probe_at: Option<Instant>,
4021    probe_index: u64,
4022}
4023
4024fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4025    lock_snapshot(snapshot)
4026        .ok()
4027        .and_then(|state| state.spawned_protocol)
4028        .unwrap_or(spec.protocol)
4029}
4030
4031impl HealthProbeRuntime {
4032    fn refresh_registration(
4033        &mut self,
4034        spec: &ModuleSpec,
4035        runtime: &SupervisorRuntimeConfig,
4036        registry: &Registry,
4037        snapshot: &SharedSnapshot,
4038    ) {
4039        if self.configured_health.as_ref() != Some(&runtime.health) {
4040            self.configured_health = Some(runtime.health.clone());
4041            self.next_probe_at = None;
4042            self.registered_connection = None;
4043            self.probe_index = 0;
4044        }
4045        // A non-wire process never registers. Only an explicitly configured
4046        // HTTP endpoint can arm its health probe; an absent HELLO is not a
4047        // health failure for that kind of process.
4048        if running_protocol(spec, snapshot) == ModuleProtocol::None {
4049            self.registered_connection = None;
4050            self.advertised = runtime.health.http.is_some();
4051            if !self.advertised {
4052                self.next_probe_at = None;
4053                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4054                    state.health = ModuleHealthStatus::default();
4055                });
4056            } else if self.next_probe_at.is_none() {
4057                self.next_probe_at = Some(
4058                    Instant::now()
4059                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4060                );
4061            }
4062            return;
4063        }
4064
4065        let registration = match registry.get_module(&spec.module_id) {
4066            Ok(registration) => registration,
4067            Err(err) => {
4068                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4069                self.advertised = false;
4070                self.next_probe_at = None;
4071                return;
4072            }
4073        };
4074
4075        let Some(registration) = registration else {
4076            self.registered_connection = None;
4077            self.advertised = false;
4078            self.next_probe_at = None;
4079            return;
4080        };
4081
4082        let advertised = registration
4083            .control_ops
4084            .iter()
4085            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4086        if !advertised {
4087            self.registered_connection = Some(registration.connection_id);
4088            self.advertised = false;
4089            self.next_probe_at = None;
4090            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4091                state.health.status = SupervisorHealthStatus::Unknown;
4092                state.health.consecutive_failures = 0;
4093                state.health.last_probe_ms = None;
4094                state.health.detail = None;
4095                state.health.metrics = None;
4096            });
4097            return;
4098        }
4099
4100        let reregistered = self.registered_connection != Some(registration.connection_id);
4101        self.registered_connection = Some(registration.connection_id);
4102        self.advertised = true;
4103        if reregistered || self.next_probe_at.is_none() {
4104            self.probe_index = 0;
4105            self.next_probe_at = Some(
4106                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4107            );
4108            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4109                state.health.status = SupervisorHealthStatus::Unknown;
4110                state.health.consecutive_failures = 0;
4111                state.health.detail = None;
4112                state.health.metrics = None;
4113            });
4114        }
4115    }
4116
4117    fn wake_after(&self) -> Duration {
4118        if !self.advertised {
4119            return REGISTRY_RELEASE_POLL;
4120        }
4121        self.next_probe_at
4122            .map(|next| next.saturating_duration_since(Instant::now()))
4123            .unwrap_or(REGISTRY_RELEASE_POLL)
4124    }
4125
4126    fn due(&self) -> bool {
4127        self.advertised
4128            && self
4129                .next_probe_at
4130                .is_some_and(|next| Instant::now() >= next)
4131    }
4132
4133    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4134        self.probe_index = self.probe_index.wrapping_add(1);
4135        self.next_probe_at = Some(
4136            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4137        );
4138    }
4139}
4140
4141/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4142///
4143/// This was a struct with a single `message: String`, and every one of the
4144/// fifteen construction sites collapsed into it. Each site knows exactly what it
4145/// saw -- the lane is gone, the module did not answer in time, the module
4146/// answered with the wrong thing -- and `handle_health_probe_failure` then
4147/// treated all of them identically: increment a counter, compare to a threshold,
4148/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4149/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4150///
4151/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4152///
4153/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4154///   answer on it again.
4155/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4156///   AND with a perfectly healthy one that lost a CPU race -- which is what
4157///   happens under machine load, and is how this supervisor killed a healthy
4158///   module three times in one day.
4159/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4160///   Restarting on it is defensible, but it is not the silence case and should
4161///   never be counted as one.
4162/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4163///   anything, so it cannot be evidence about the module at all.
4164///
4165/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4166/// one that fires most often, and while every variant collapsed into one string
4167/// it carried the same weight as the strongest.
4168///
4169/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4170/// DESIGN and a reader stopping at it gets the build backwards: the restart
4171/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4172/// probes still increment the failure streak and drive escalation at the
4173/// threshold (see `is_proof_of_death` below for why that is deliberate and
4174/// what gates the change). Absence of evidence restarts modules today.
4175#[derive(Debug)]
4176enum HealthProbeEvidence {
4177    /// The module's control lane is gone. Proof of death.
4178    LaneDead,
4179    /// No reply within the deadline. Proves nothing about the module's state.
4180    NoAnswer,
4181    /// The module replied, but not with a usable health report. Proves it is alive.
4182    BadAnswer,
4183    /// The daemon could not ask. Says nothing about the module.
4184    Misconfigured,
4185}
4186
4187#[derive(Debug)]
4188struct HealthProbeError {
4189    evidence: HealthProbeEvidence,
4190    message: String,
4191}
4192
4193impl HealthProbeError {
4194    fn lane_dead(message: impl Into<String>) -> Self {
4195        Self::with(HealthProbeEvidence::LaneDead, message)
4196    }
4197
4198    fn no_answer(message: impl Into<String>) -> Self {
4199        Self::with(HealthProbeEvidence::NoAnswer, message)
4200    }
4201
4202    fn bad_answer(message: impl Into<String>) -> Self {
4203        Self::with(HealthProbeEvidence::BadAnswer, message)
4204    }
4205
4206    fn misconfigured(message: impl Into<String>) -> Self {
4207        Self::with(HealthProbeEvidence::Misconfigured, message)
4208    }
4209
4210    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4211        Self {
4212            evidence,
4213            message: message.into(),
4214        }
4215    }
4216
4217    /// Whether this observation is proof the module cannot serve.
4218    ///
4219    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4220    /// variant that fires under CPU starvation, and treating it as proof is the
4221    /// defect this enum exists to make impossible to reintroduce silently.
4222    ///
4223    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4224    /// to restart also needs a bound for the case it excludes -- a genuinely
4225    /// wedged module, alive but never answering -- and that bound must come from
4226    /// the distribution of real late-answer latencies, which nothing measures
4227    /// yet. Landing the classification first makes the later change a one-line
4228    /// decision against evidence that already exists, rather than two unproven
4229    /// changes at once.
4230    #[allow(dead_code)]
4231    fn is_proof_of_death(&self) -> bool {
4232        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4233    }
4234
4235    /// Short stable label for logs and the health snapshot.
4236    ///
4237    /// An operator reading `ck health` currently cannot tell "the module is gone"
4238    /// from "the module did not answer in five seconds", because both render as
4239    /// prose in the same field. These labels are what make the two
4240    /// distinguishable at a glance, and they are what a later restart-policy
4241    /// change will be argued from.
4242    fn label(&self) -> &'static str {
4243        match self.evidence {
4244            HealthProbeEvidence::LaneDead => "lane-dead",
4245            HealthProbeEvidence::NoAnswer => "no-answer",
4246            HealthProbeEvidence::BadAnswer => "bad-answer",
4247            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4248        }
4249    }
4250}
4251
4252impl fmt::Display for HealthProbeError {
4253    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4254        f.write_str(&self.message)
4255    }
4256}
4257
4258async fn run_health_probe_cycle(
4259    spec: &ModuleSpec,
4260    runtime: &SupervisorRuntimeConfig,
4261    registry: &Registry,
4262    process_liveness: &SupervisorProcessLiveness,
4263    snapshot: &SharedSnapshot,
4264    child: &mut Option<SupervisedChild>,
4265) {
4266    let now_ms = unix_ms_now();
4267    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4268        .then_some(runtime.health.http.as_deref())
4269        .flatten();
4270    let result = match http {
4271        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4272        None => probe_module_health(&spec.module_id, runtime, None).await,
4273    };
4274    match result {
4275        Ok(report) => {
4276            handle_health_report(
4277                spec,
4278                runtime,
4279                registry,
4280                process_liveness,
4281                snapshot,
4282                child,
4283                report,
4284                now_ms,
4285            )
4286            .await;
4287        }
4288        Err(err) => {
4289            if http.is_some() {
4290                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4291                    state.health.status = SupervisorHealthStatus::Failing;
4292                });
4293            }
4294            handle_health_probe_failure(
4295                spec,
4296                runtime,
4297                registry,
4298                process_liveness,
4299                snapshot,
4300                child,
4301                err,
4302                now_ms,
4303            )
4304            .await;
4305        }
4306    }
4307}
4308
4309pub(crate) struct HttpProbeTarget<'a> {
4310    address: std::net::SocketAddr,
4311    localhost: bool,
4312    authority: &'a str,
4313    path: String,
4314}
4315
4316/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4317/// or TLS. A URL cannot turn a local health check into an outbound connection.
4318pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4319    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4320        return Err("must not contain whitespace, controls, or a fragment".into());
4321    }
4322    let rest = url
4323        .strip_prefix("http://")
4324        .ok_or("must use plain http://")?;
4325    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4326    let (authority, suffix) = rest.split_at(split);
4327    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4328        ("::1", rest)
4329    } else {
4330        let split = authority.find(':').unwrap_or(authority.len());
4331        authority.split_at(split)
4332    };
4333    let ip: std::net::IpAddr = match host {
4334        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4335        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4336        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4337    };
4338    let port = if port.is_empty() {
4339        80
4340    } else {
4341        port.strip_prefix(':')
4342            .and_then(|p| p.parse::<u16>().ok())
4343            .filter(|p| *p > 0)
4344            .ok_or("must have a valid nonzero TCP port")?
4345    };
4346    let path = if suffix.is_empty() {
4347        "/".into()
4348    } else if suffix.starts_with('?') {
4349        format!("/{suffix}")
4350    } else {
4351        suffix.into()
4352    };
4353    Ok(HttpProbeTarget {
4354        address: std::net::SocketAddr::new(ip, port),
4355        localhost: host == "localhost",
4356        authority,
4357        path,
4358    })
4359}
4360
4361async fn probe_http_health(
4362    url: &str,
4363    deadline: Duration,
4364) -> Result<HealthReport, HealthProbeError> {
4365    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4366    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4367    // Keep partial diagnostics outside the timed future so cancellation does
4368    // not discard a status line or body bytes already received.
4369    let mut response_status = String::new();
4370    let mut body = Vec::new();
4371    let probe = async {
4372        // Resolve localhost ourselves so a hosts-file override cannot turn
4373        // this into an outbound request, while IPv6-only local servers work.
4374        let connection = match tokio::net::TcpStream::connect(target.address).await {
4375            Err(_) if target.localhost => {
4376                tokio::net::TcpStream::connect((
4377                    std::net::Ipv6Addr::LOCALHOST,
4378                    target.address.port(),
4379                ))
4380                .await
4381            }
4382            result => result,
4383        };
4384        let mut stream = connection.map_err(|error| {
4385            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4386        })?;
4387        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4388            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4389        let mut reader = BufReader::new(stream);
4390        let mut budget = 16 * 1024;
4391        let status = http_line(&mut reader, &mut budget).await?;
4392        let mut words = status.split_ascii_whitespace();
4393        let version = words.next();
4394        let code = words
4395            .next()
4396            .filter(|word| word.len() == 3)
4397            .and_then(|word| word.parse::<u16>().ok());
4398        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4399            || !code.is_some_and(|code| (100..600).contains(&code))
4400        {
4401            return Err(HealthProbeError::bad_answer(format!(
4402                "invalid HTTP status: {status}"
4403            )));
4404        }
4405        let code = code.expect("validated status code");
4406        response_status = status.clone();
4407        let mut length = None;
4408        let mut chunked = false;
4409        loop {
4410            let line = http_line(&mut reader, &mut budget).await?;
4411            if line.is_empty() {
4412                break;
4413            }
4414            if let Some((name, value)) = line.split_once(':') {
4415                if name.eq_ignore_ascii_case("content-length") {
4416                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4417                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4418                    })?);
4419                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4420                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4421                }
4422            }
4423        }
4424        if chunked {
4425            while body.len() < 200 {
4426                let line = http_line(&mut reader, &mut budget).await?;
4427                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4428                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4429                if size == 0 {
4430                    break;
4431                }
4432                let count = size.min((200 - body.len()) as u64) as usize;
4433                let start = body.len();
4434                (&mut reader)
4435                    .take(count as u64)
4436                    .read_to_end(&mut body)
4437                    .await
4438                    .map_err(|error| {
4439                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4440                    })?;
4441                if body.len() - start != count {
4442                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4443                }
4444                if size > count as u64 || body.len() == 200 {
4445                    break;
4446                }
4447                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4448                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4449                }
4450            }
4451        } else {
4452            reader
4453                .take(length.unwrap_or(200).min(200))
4454                .read_to_end(&mut body)
4455                .await
4456                .map_err(|error| {
4457                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4458                })?;
4459        }
4460        if (200..300).contains(&code) {
4461            Ok(HealthReport::ok())
4462        } else {
4463            Err(HealthProbeError::bad_answer(
4464                "HTTP health endpoint returned non-2xx",
4465            ))
4466        }
4467    };
4468    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4469        Err(HealthProbeError::no_answer(format!(
4470            "HTTP probe timed out after {deadline:?}"
4471        )))
4472    });
4473    if let Err(error) = &mut result {
4474        if !response_status.is_empty() {
4475            error.message = format!(
4476                "{}; {response_status}: {}",
4477                error.message,
4478                String::from_utf8_lossy(&body)
4479            );
4480        }
4481    }
4482    result
4483}
4484
4485async fn http_line(
4486    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4487    remaining: &mut usize,
4488) -> Result<String, HealthProbeError> {
4489    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4490    let mut line = Vec::new();
4491    (&mut *reader)
4492        .take(*remaining as u64)
4493        .read_until(b'\n', &mut line)
4494        .await
4495        .map_err(|error| {
4496            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4497        })?;
4498    *remaining -= line.len();
4499    if !line.ends_with(b"\r\n") {
4500        return Err(HealthProbeError::bad_answer(
4501            "HTTP headers are incomplete or exceed 16 KiB",
4502        ));
4503    }
4504    line.truncate(line.len() - 2);
4505    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4506}
4507
4508async fn probe_module_health(
4509    module_id: &str,
4510    runtime: &SupervisorRuntimeConfig,
4511    drain_deadline: Option<Instant>,
4512) -> Result<HealthReport, HealthProbeError> {
4513    let Some(forwarding) = runtime.forwarding.as_ref() else {
4514        return Err(HealthProbeError::misconfigured(
4515            "supervisor was not configured with a forwarding table",
4516        ));
4517    };
4518    let probe_started_at = Instant::now();
4519    let mut deadline = probe_started_at + runtime.health.deadline;
4520    if let Some(drain_deadline) = drain_deadline {
4521        deadline = deadline.min(drain_deadline);
4522    }
4523    let pending = if drain_deadline.is_some() {
4524        forwarding.begin_drain_health_probe_rpc_for(
4525            module_id,
4526            MODULE_CONTROL_OP_HEALTH_CHECK,
4527            probe_started_at,
4528            deadline,
4529        )
4530    } else {
4531        forwarding.begin_health_probe_rpc_for(
4532            module_id,
4533            MODULE_CONTROL_OP_HEALTH_CHECK,
4534            probe_started_at,
4535            deadline,
4536        )
4537    }
4538    .map_err(|err| {
4539        // The endpoint is not registered, so there is no live control lane to
4540        // ask. That is the module being absent, not slow.
4541        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4542    })?;
4543    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4544}
4545
4546/// [`probe_module_health`] for one endpoint rather than the id's active one.
4547///
4548/// A swap probes two processes that no by-id lookup reaches: its candidate
4549/// before cutover, and its superseded incumbent (for busy gauges) while the
4550/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4551/// bounds the by-id drain probe.
4552async fn probe_endpoint_health(
4553    endpoint: crate::ModuleEndpointId,
4554    runtime: &SupervisorRuntimeConfig,
4555    deadline_cap: Option<Instant>,
4556) -> Result<HealthReport, HealthProbeError> {
4557    let Some(forwarding) = runtime.forwarding.as_ref() else {
4558        return Err(HealthProbeError::misconfigured(
4559            "supervisor was not configured with a forwarding table",
4560        ));
4561    };
4562    let probe_started_at = Instant::now();
4563    let mut deadline = probe_started_at + runtime.health.deadline;
4564    if let Some(cap) = deadline_cap {
4565        deadline = deadline.min(cap);
4566    }
4567    let pending = forwarding
4568        .begin_endpoint_health_probe_rpc_for(
4569            endpoint,
4570            MODULE_CONTROL_OP_HEALTH_CHECK,
4571            probe_started_at,
4572            deadline,
4573        )
4574        .map_err(|err| {
4575            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4576        })?;
4577    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4578}
4579
4580/// Send a begun health probe and classify its answer.
4581async fn await_health_probe(
4582    forwarding: &ForwardingTable,
4583    pending: PendingModuleControlRpc,
4584    deadline: Instant,
4585    probe_budget: Duration,
4586) -> Result<HealthReport, HealthProbeError> {
4587    let PendingModuleControlRpc {
4588        endpoint,
4589        module_sink,
4590        negotiated_ver,
4591        corr,
4592        receiver,
4593    } = pending;
4594    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4595        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4596    })?;
4597    let frame = Frame::build_with_version(
4598        negotiated_ver,
4599        FrameType::Request,
4600        control_flags(),
4601        0,
4602        0,
4603        corr,
4604        body,
4605    )
4606    .map_err(|err| {
4607        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4608    })?;
4609
4610    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4611    // blocks waiting for capacity when the module's egress queue is full, and an
4612    // unbounded await here freezes the whole supervision actor (it stops polling
4613    // Child::wait and supervisor commands), making the module unrecoverable
4614    // in-band. On timeout the probe fails like any transport failure.
4615    match timeout_at(deadline, module_sink.send(frame)).await {
4616        Ok(Ok(())) => {}
4617        Ok(Err(err)) => {
4618            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4619            // A closed sink means the module's egress channel is gone -- the
4620            // receiving half is dropped when its connection tears down. Proof.
4621            return Err(HealthProbeError::lane_dead(format!(
4622                "failed to send health.check: {err}"
4623            )));
4624        }
4625        Err(_elapsed) => {
4626            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4627            // A full egress queue means the module is not draining its socket, which
4628            // is consistent with a wedged module AND with one whose reader is merely
4629            // starved. Silence, not proof.
4630            return Err(HealthProbeError::no_answer(
4631                "health.check send timed out before enqueue (module egress full)",
4632            ));
4633        }
4634    }
4635
4636    match timeout_at(deadline, receiver).await {
4637        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4638        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4639        // and those prove it is alive even though the probe failed.
4640        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4641            response.health_report().ok_or_else(|| {
4642                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4643            })
4644        }
4645        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4646            format!("health.check rejected: {}", body.message),
4647        )),
4648        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4649            Err(HealthProbeError::lane_dead(message))
4650        }
4651        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4652            Err(HealthProbeError::bad_answer(message))
4653        }
4654        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4655            Err(HealthProbeError::bad_answer(format!(
4656                "expected module-control op '{expected}', got '{actual}'"
4657            )))
4658        }
4659        // A reply that crosses the deadline before this waiter observes it is
4660        // still proof of life. The forwarding path records its end-to-end latency
4661        // before delivering this classification.
4662        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4663            "module answered health.check after its daemon deadline",
4664        )),
4665        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4666            "health.check waiter was canceled before the module responded",
4667        )),
4668        Err(_) => {
4669            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4670            Err(HealthProbeError::no_answer(format!(
4671                "module did not answer health.check within {probe_budget:?}"
4672            )))
4673        }
4674    }
4675}
4676
4677#[allow(clippy::too_many_arguments)]
4678async fn handle_health_report(
4679    spec: &ModuleSpec,
4680    runtime: &SupervisorRuntimeConfig,
4681    registry: &Registry,
4682    process_liveness: &SupervisorProcessLiveness,
4683    snapshot: &SharedSnapshot,
4684    child: &mut Option<SupervisedChild>,
4685    report: HealthReport,
4686    now_ms: u64,
4687) {
4688    let status = supervisor_health_status(report.status);
4689    let detail = report.detail.clone();
4690    let metrics = truncate_health_metrics(report.metrics);
4691    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4692        state.health.status = status;
4693        state.health.last_probe_ms = Some(now_ms);
4694        state.health.detail = detail.clone();
4695        state.health.metrics = metrics.clone();
4696        state.health.consecutive_failures = 0;
4697    });
4698
4699    let action = match report.status {
4700        HealthStatus::Ok => return,
4701        HealthStatus::Degraded => runtime.health.on_degraded,
4702        HealthStatus::Failing => runtime.health.on_failing,
4703    };
4704    apply_l3_health_action(
4705        spec,
4706        runtime,
4707        registry,
4708        process_liveness,
4709        snapshot,
4710        child,
4711        status,
4712        detail.as_deref(),
4713        action,
4714        now_ms,
4715    )
4716    .await;
4717}
4718
4719#[allow(clippy::too_many_arguments)]
4720async fn handle_health_probe_failure(
4721    spec: &ModuleSpec,
4722    runtime: &SupervisorRuntimeConfig,
4723    registry: &Registry,
4724    process_liveness: &SupervisorProcessLiveness,
4725    snapshot: &SharedSnapshot,
4726    child: &mut Option<SupervisedChild>,
4727    err: HealthProbeError,
4728    now_ms: u64,
4729) {
4730    let threshold = runtime.health.failure_threshold.max(1);
4731    let mut failures = 0;
4732    // Carry the evidence class into the operator-visible detail. Without it,
4733    // "module did not answer within 5s" and "the control lane is gone" are two
4734    // prose strings in the same field, and the reader has to know the codebase to
4735    // tell which one is proof of anything.
4736    let detail = format!("[{}] {err}", err.label());
4737    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4738        // A failed wire probe invalidates the last report, even before the
4739        // restart threshold. HTTP probes already mark failures as Failing.
4740        if state.spawned_protocol.unwrap_or(spec.protocol) == ModuleProtocol::Subc {
4741            state.health.status = SupervisorHealthStatus::Unknown;
4742        }
4743        state.health.last_probe_ms = Some(now_ms);
4744        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4745        state.health.detail = Some(detail.clone());
4746        state.health.metrics = None;
4747        failures = state.health.consecutive_failures;
4748    });
4749
4750    if failures < threshold {
4751        warn!(
4752            module_id = %spec.module_id,
4753            consecutive_failures = failures,
4754            threshold,
4755            evidence = err.label(),
4756            detail = %detail,
4757            "health.check probe failed"
4758        );
4759        return;
4760    }
4761
4762    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4763        state.state = ModuleState::Unresponsive;
4764        state.health.status = SupervisorHealthStatus::Unresponsive;
4765    });
4766    // The evidence class is logged at the kill site because this is the line an
4767    // operator reads after an unexplained restart. A streak of `no-answer` under
4768    // machine load is the known false-positive shape; a `lane-dead` is not.
4769    if runtime.health.critical {
4770        error!(
4771            module_id = %spec.module_id,
4772            status = "unresponsive",
4773            evidence = err.label(),
4774            detail = %detail,
4775            "critical module health alert"
4776        );
4777    } else {
4778        warn!(
4779            module_id = %spec.module_id,
4780            status = "unresponsive",
4781            evidence = err.label(),
4782            detail = %detail,
4783            "module health threshold breached"
4784        );
4785    }
4786    if let Err(err) = health_restart_child(
4787        spec,
4788        runtime,
4789        registry,
4790        process_liveness,
4791        snapshot,
4792        child,
4793        SupervisorHealthStatus::Unresponsive,
4794        Some(&detail),
4795        now_ms,
4796    )
4797    .await
4798    {
4799        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4800    }
4801}
4802
4803#[allow(clippy::too_many_arguments)]
4804async fn apply_l3_health_action(
4805    spec: &ModuleSpec,
4806    runtime: &SupervisorRuntimeConfig,
4807    registry: &Registry,
4808    process_liveness: &SupervisorProcessLiveness,
4809    snapshot: &SharedSnapshot,
4810    child: &mut Option<SupervisedChild>,
4811    status: SupervisorHealthStatus,
4812    detail: Option<&str>,
4813    action: HealthAction,
4814    now_ms: u64,
4815) {
4816    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4817    match action {
4818        HealthAction::Report => {
4819            info!(
4820                module_id = %spec.module_id,
4821                status = ?status,
4822                detail,
4823                "module reported non-ok health"
4824            );
4825        }
4826        HealthAction::Alert => {
4827            error!(
4828                module_id = %spec.module_id,
4829                status = ?status,
4830                detail,
4831                "module health alert"
4832            );
4833        }
4834        HealthAction::Restart => {
4835            if let Err(err) = health_restart_child(
4836                spec,
4837                runtime,
4838                registry,
4839                process_liveness,
4840                snapshot,
4841                child,
4842                status,
4843                detail,
4844                now_ms,
4845            )
4846            .await
4847            {
4848                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4849            }
4850        }
4851    }
4852}
4853
4854#[allow(clippy::too_many_arguments)]
4855async fn health_restart_child(
4856    spec: &ModuleSpec,
4857    runtime: &SupervisorRuntimeConfig,
4858    registry: &Registry,
4859    process_liveness: &SupervisorProcessLiveness,
4860    snapshot: &SharedSnapshot,
4861    child: &mut Option<SupervisedChild>,
4862    status: SupervisorHealthStatus,
4863    detail: Option<&str>,
4864    now_ms: u64,
4865) -> Result<(), SuperviseError> {
4866    let (enabled, schedule) = {
4867        let mut state = lock_snapshot(snapshot)?;
4868        let enabled = state.enabled;
4869        let schedule = if enabled {
4870            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4871        } else {
4872            None
4873        };
4874        (enabled, schedule)
4875    };
4876
4877    if !enabled {
4878        return Err(SuperviseError::Disabled {
4879            module_id: spec.module_id.clone(),
4880        });
4881    }
4882
4883    if schedule.is_none() {
4884        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4885        error!(
4886            module_id = %spec.module_id,
4887            status = ?status,
4888            detail,
4889            max_restarts = runtime.restart_policy.max_restarts,
4890            window_secs = runtime.restart_policy.window.as_secs(),
4891            reason = %runtime.restart_policy.budget_exhausted_detail(),
4892            "health restart budget exhausted; marking module failed"
4893        );
4894        let stop_notice = begin_forwarding_drain_if_configured(
4895            spec,
4896            runtime,
4897            registry,
4898            snapshot,
4899            Some(true),
4900            RouteCloseReason::Disable,
4901        )
4902        .await?;
4903        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4904            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4905        })?;
4906        drain_optional_child(
4907            &spec.module_id,
4908            spec.protocol,
4909            stop_notice,
4910            registry,
4911            runtime.forwarding.as_deref(),
4912            snapshot,
4913            &runtime.terminal_ring,
4914            &runtime.spawn_events,
4915            child,
4916            runtime.drain_timeout,
4917            ModuleState::Failed,
4918            Some(true),
4919        )
4920        .await?;
4921        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4922        return Ok(());
4923    }
4924
4925    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4926    let mut restart_count = 0;
4927    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4928        restart_count = state.crash_restarts.len();
4929        state.state = ModuleState::Unresponsive;
4930        state.health.status = status;
4931        state.health.last_action = Some(HealthAction::Restart.to_string());
4932        state.health.last_action_ms = Some(now_ms);
4933    })?;
4934    warn!(
4935        module_id = %spec.module_id,
4936        status = ?status,
4937        detail,
4938        restart_count,
4939        restart_in_window = schedule.restart_in_window,
4940        delay_ms = schedule.delay.as_millis() as u64,
4941        "health-triggered module restart"
4942    );
4943
4944    let stop_notice = begin_forwarding_drain_if_configured(
4945        spec,
4946        runtime,
4947        registry,
4948        snapshot,
4949        Some(true),
4950        RouteCloseReason::Restart,
4951    )
4952    .await?;
4953    drain_optional_child(
4954        &spec.module_id,
4955        spec.protocol,
4956        stop_notice,
4957        registry,
4958        runtime.forwarding.as_deref(),
4959        snapshot,
4960        &runtime.terminal_ring,
4961        &runtime.spawn_events,
4962        child,
4963        runtime.drain_timeout,
4964        ModuleState::Restarting,
4965        Some(true),
4966    )
4967    .await?;
4968    schedule_respawn(
4969        runtime,
4970        snapshot,
4971        &spec.module_id,
4972        schedule.delay,
4973        RespawnKind::Spawn,
4974    )
4975}
4976
4977fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4978    if let Some(reply) = runtime
4979        .deferred_reload_reply
4980        .lock()
4981        .unwrap_or_else(|p| p.into_inner())
4982        .take()
4983    {
4984        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4985            module_id: module_id.to_string(),
4986            reason: reason.to_string(),
4987        }));
4988    }
4989}
4990
4991fn schedule_respawn(
4992    runtime: &SupervisorRuntimeConfig,
4993    snapshot: &SharedSnapshot,
4994    module_id: &str,
4995    delay: Duration,
4996    kind: RespawnKind,
4997) -> Result<(), SuperviseError> {
4998    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4999    update_snapshot(snapshot, Some(module_id), |state| {
5000        state.respawn_pending = true
5001    })?;
5002    *runtime
5003        .scheduled_respawn
5004        .lock()
5005        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5006        deadline: Instant::now() + delay,
5007        kind,
5008    });
5009    Ok(())
5010}
5011
5012fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5013    let _ = update_snapshot(snapshot, Some(module_id), |state| {
5014        state.health.last_action = Some(action);
5015        state.health.last_action_ms = Some(now_ms);
5016    });
5017}
5018
5019fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5020    match status {
5021        HealthStatus::Ok => SupervisorHealthStatus::Ok,
5022        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5023        HealthStatus::Failing => SupervisorHealthStatus::Failing,
5024    }
5025}
5026
5027/// Caps the metrics blob stored in the cached supervisor snapshot, which is
5028/// returned to every `supervisor.list` and `supervisor.health` caller.
5029///
5030/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
5031/// path: that request exists to return a module's complete metrics object, and
5032/// `ck health <module-id>` documents it as the way to see what the cached view
5033/// truncates. The asymmetry is the feature.
5034///
5035/// So a new caller must decide which side it is on rather than assume the cap is
5036/// universal. Reaching for it on a fresh-probe path would silently reintroduce
5037/// the truncation that path exists to avoid.
5038fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5039    let metrics = metrics?;
5040    match serde_json::to_vec(&metrics) {
5041        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5042            "truncated": true,
5043            "original_bytes": encoded.len(),
5044        })),
5045        Ok(_) | Err(_) => Some(metrics),
5046    }
5047}
5048
5049/// Spread health probes so a fleet-wide restart does not converge them.
5050///
5051/// The delay is derived from the module id and probe index rather than a random
5052/// source, so it is deterministic per module: a module keeps its own offset
5053/// across daemon restarts instead of re-rolling into a collision.
5054fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5055    if cadence.is_zero() {
5056        return Duration::ZERO;
5057    }
5058    let cadence_ms = cadence.as_millis() as u64;
5059    // This early return is REDUNDANT, deliberately, and a mutation run will show
5060    // it surviving removal. Recording why here so the next person to notice does
5061    // not have to re-derive it:
5062    //
5063    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
5064    //   a zero cadence and builds the Duration from whole milliseconds, so a
5065    //   sub-millisecond cadence cannot come from config.
5066    // - Even if reached it changes no answer. The `.max(1)` below makes the span
5067    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
5068    //   -- exactly what this returns.
5069    //
5070    // Kept as a guard against a future widening of the config parser (accepting
5071    // microseconds, say), which would make the sub-millisecond case reachable.
5072    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
5073    // divides by zero. Remove this and nothing changes.
5074    if cadence_ms == 0 {
5075        return cadence;
5076    }
5077    // Note that this never returns less than one cadence, including for the FIRST
5078    // probe. So a freshly registered module reports health `unknown` for a full
5079    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5080    // ready to answer.
5081    //
5082    // That is a property of the supervisor's schedule, not of any module: an
5083    // operator watching a restart sees `unknown` and cannot tell it from a module
5084    // that is slow to warm. Measured on two unrelated modules, both flipping to
5085    // `ok` between 22s and 32s after restart.
5086    //
5087    // Left as-is because spreading the first probe is what keeps a fleet-wide
5088    // restart from firing fourteen simultaneous probes into a cold machine. The
5089    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5090    // that thundering herd for a faster first reading.
5091    let jitter_span = (cadence_ms / 10).max(1);
5092    let hash = module_id.as_bytes().iter().fold(
5093        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5094        |acc, byte| {
5095            acc.wrapping_mul(1099511628211)
5096                .wrapping_add(u64::from(*byte))
5097        },
5098    );
5099    cadence + Duration::from_millis(hash % jitter_span)
5100}
5101
5102#[cfg(test)]
5103mod tests {
5104    use super::*;
5105
5106    #[test]
5107    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5108        let handle = SupervisorHandle::new();
5109        let module_id = "readded-tombstone";
5110        handle.record_rescan_removal(module_id);
5111        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5112
5113        handle.apply_identity_configuration(&ModuleSpec {
5114            module_id: module_id.to_string(),
5115            program: PathBuf::from("/test/module"),
5116            args: Vec::new(),
5117            env: Vec::new(),
5118            reserved: false,
5119            reserved_prefixes: Vec::new(),
5120            protocol: ModuleProtocol::Subc,
5121            overlap: Default::default(),
5122        });
5123
5124        assert!(
5125            handle.removal_tombstone_age_ms(module_id).is_none(),
5126            "a re-added module must not retain a stale removal tombstone"
5127        );
5128    }
5129
5130    /// What one module's owner looked like from the control plane at the
5131    /// instant after its first process was spawned.
5132    #[derive(Debug, PartialEq, Eq)]
5133    struct OwnerInSpawnWindow {
5134        module_id: String,
5135        configured: bool,
5136        on_roster: bool,
5137        admission_refusal: Option<&'static str>,
5138    }
5139
5140    /// A supervised module's process can connect, register, sync its scopes
5141    /// and describe them as soon as it is spawned, which is BEFORE the
5142    /// supervisor puts the module on the roster. In that window the owner must
5143    /// already read as configured, so a scoped `route.open` against it is
5144    /// refused as retryable `scope_not_synced` and not as terminal
5145    /// `scope_not_live` ("will never sync").
5146    ///
5147    /// The hook runs in exactly that window on every path that takes on a new
5148    /// module, so no race with a real child is needed: `on_roster: false`
5149    /// proves each observation was taken before the roster insert.
5150    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5151    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5152        use crate::scopes::ScopeTable;
5153        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5154
5155        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5156        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5157            module_id: module_id.to_string(),
5158            program,
5159            args: Vec::new(),
5160            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5161                .into_iter()
5162                .map(|key| (key.to_string(), dir.path().display().to_string()))
5163                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5164                .collect(),
5165            reserved: false,
5166            reserved_prefixes: Vec::new(),
5167            protocol: ModuleProtocol::Subc,
5168            overlap: Default::default(),
5169        };
5170        let live = super::terminal_history_tests::fake_aft_stub_path();
5171        let missing = dir.path().join("definitely-missing-module");
5172
5173        let handle = SupervisorHandle::new();
5174        let mut supervisor =
5175            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5176                .with_handle(handle.clone());
5177        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5178        let hook_handle = handle.clone();
5179        let hook_observed = Arc::clone(&observed);
5180        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5181            // Exactly what the control plane computes for a scoped route.open
5182            // naming this module as the owner of a scope it has not synced.
5183            let configured = hook_handle.is_configured(module_id);
5184            let selector = ScopeSelector {
5185                owner: Principal::Reserved {
5186                    module_id: module_id.to_string(),
5187                },
5188                scope_ref: "s".to_string(),
5189                scope_epoch: Some(1),
5190            };
5191            let carrier = Principal::Reserved {
5192                module_id: "carrier".to_string(),
5193            };
5194            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5195                .admit(&carrier, module_id, &selector, configured)
5196            {
5197                Ok(_) => None,
5198                Err(refusal) => Some(refusal.code),
5199            };
5200            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5201                module_id: module_id.to_string(),
5202                configured,
5203                on_roster: hook_handle.get(module_id).is_some(),
5204                admission_refusal,
5205            });
5206        })));
5207
5208        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5209        let configured = supervisor
5210            .supervise_configured(stub("configured", live.clone()), true)
5211            .unwrap();
5212        let with_health = supervisor
5213            .supervise_configured_with_health(
5214                stub("with-health", live.clone()),
5215                true,
5216                HealthConfig::default(),
5217                None,
5218                RestartPolicy::default(),
5219            )
5220            .unwrap();
5221        // The failed-spawn path still puts the module on the roster (as
5222        // failed), so it is configured throughout.
5223        let failed = supervisor
5224            .supervise_configured_with_health(
5225                stub("failed-spawn", missing.clone()),
5226                true,
5227                HealthConfig::default(),
5228                None,
5229                RestartPolicy::default(),
5230            )
5231            .unwrap();
5232        // A failed plain `spawn` puts nothing on the roster, so its mark is
5233        // taken back once the spawn has failed.
5234        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5235
5236        let expected = [
5237            "plain",
5238            "configured",
5239            "with-health",
5240            "failed-spawn",
5241            "spawn-error",
5242        ]
5243        .into_iter()
5244        .map(|module_id| OwnerInSpawnWindow {
5245            module_id: module_id.to_string(),
5246            configured: true,
5247            on_roster: false,
5248            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5249        })
5250        .collect::<Vec<_>>();
5251        assert_eq!(*observed.lock().unwrap(), expected);
5252
5253        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5254            assert!(
5255                handle.get(module_id).is_some(),
5256                "{module_id} is on the roster"
5257            );
5258            assert!(
5259                handle.is_configured(module_id),
5260                "{module_id} stays configured"
5261            );
5262        }
5263        assert!(handle.get("spawn-error").is_none());
5264        assert!(
5265            !handle.is_configured("spawn-error"),
5266            "a plain spawn that failed must not leave its module marked configured"
5267        );
5268
5269        // Leaving the roster clears the mark with it.
5270        handle.retire("failed-spawn");
5271        assert!(!handle.is_configured("failed-spawn"));
5272
5273        for module in [plain, configured, with_health] {
5274            module.stop().await.unwrap();
5275        }
5276        drop(failed);
5277    }
5278
5279    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5280        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5281        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5282            snapshot.process_alive = true;
5283            snapshot.pid = Some(41);
5284            snapshot.spawned_at_ms = Some(42);
5285            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5286            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5287                device: 43,
5288                inode: 44,
5289            });
5290        })
5291        .unwrap();
5292        snapshot
5293    }
5294
5295    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5296        let snapshot = lock_snapshot(snapshot).unwrap();
5297        assert!(!snapshot.process_alive);
5298        assert_eq!(snapshot.pid, None);
5299        assert_eq!(snapshot.spawned_at_ms, None);
5300        assert_eq!(snapshot.spawned_from, None);
5301        assert_eq!(snapshot.spawned_file_identity, None);
5302    }
5303
5304    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5305    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5306        let supervisor =
5307            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5308        let mut runtime = supervisor.runtime_config();
5309        runtime.test_seed_stale_facts_before_enable_spawn = true;
5310        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5311        let mut child = None;
5312        let spec = ModuleSpec {
5313            module_id: "failed-enable-clears-facts".to_string(),
5314            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5315            args: Vec::new(),
5316            env: Vec::new(),
5317            reserved: false,
5318            reserved_prefixes: Vec::new(),
5319            protocol: ModuleProtocol::Subc,
5320            overlap: Default::default(),
5321        };
5322
5323        let result = set_child_enabled(
5324            &spec,
5325            &runtime,
5326            &supervisor.registry,
5327            &supervisor.process_liveness,
5328            &snapshot,
5329            &mut child,
5330            true,
5331        )
5332        .await;
5333
5334        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5335        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5336        assert_snapshot_process_facts_cleared(&snapshot);
5337    }
5338
5339    #[tokio::test]
5340    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5341        let supervisor =
5342            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5343        let runtime = supervisor.runtime_config();
5344        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5345            ModuleState::Restarting,
5346            true,
5347        )));
5348        let spec = ModuleSpec {
5349            module_id: "start-stranded-restarting".to_string(),
5350            program: super::terminal_history_tests::fake_aft_stub_path(),
5351            args: Vec::new(),
5352            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5353            reserved: false,
5354            reserved_prefixes: Vec::new(),
5355            protocol: ModuleProtocol::None,
5356            overlap: Default::default(),
5357        };
5358        let mut child = None;
5359        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5360        assert!(!super::set_child_enabled(
5361            &spec,
5362            &runtime,
5363            &Registry::default(),
5364            &supervisor.process_liveness,
5365            &snapshot,
5366            &mut child,
5367            true
5368        )
5369        .await
5370        .unwrap());
5371        assert!(child.is_none());
5372        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5373        assert!(super::set_child_enabled(
5374            &spec,
5375            &runtime,
5376            &Registry::default(),
5377            &supervisor.process_liveness,
5378            &snapshot,
5379            &mut child,
5380            true
5381        )
5382        .await
5383        .unwrap());
5384        assert_eq!(
5385            lock_snapshot(&snapshot).unwrap().state,
5386            ModuleState::Running
5387        );
5388        let mut child = child.unwrap();
5389        child.start_kill().unwrap();
5390        child.wait().await.unwrap();
5391    }
5392
5393    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5394    async fn failed_reload_spawn_clears_current_process_facts() {
5395        let supervisor =
5396            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5397        let mut runtime = supervisor.runtime_config();
5398        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5399        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5400        let mut child = None;
5401        let spec = ModuleSpec {
5402            module_id: "failed-reload-clears-facts".to_string(),
5403            program: PathBuf::from("/unused/failed-reload-module"),
5404            args: Vec::new(),
5405            env: Vec::new(),
5406            reserved: false,
5407            reserved_prefixes: Vec::new(),
5408            protocol: ModuleProtocol::Subc,
5409            overlap: Default::default(),
5410        };
5411
5412        let result = handle_reload_spawn_failure(
5413            &spec,
5414            &runtime,
5415            &supervisor.process_liveness,
5416            &snapshot,
5417            &mut child,
5418            "forced reload spawn failure".to_string(),
5419        )
5420        .await;
5421
5422        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5423        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5424        assert_snapshot_process_facts_cleared(&snapshot);
5425    }
5426
5427    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5428    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5429        let supervisor =
5430            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5431        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5432        let module = supervisor.supervised_module(
5433            ModuleSpec {
5434                module_id: "drop-clears-facts".to_string(),
5435                program: PathBuf::from("/unused/drop-module"),
5436                args: Vec::new(),
5437                env: Vec::new(),
5438                reserved: false,
5439                reserved_prefixes: Vec::new(),
5440                protocol: ModuleProtocol::Subc,
5441                overlap: Default::default(),
5442            },
5443            supervisor.runtime_config(),
5444            Arc::clone(&snapshot),
5445            None,
5446        );
5447        assert!(!module
5448            .inner
5449            .monitor
5450            .lock()
5451            .unwrap()
5452            .as_ref()
5453            .unwrap()
5454            .is_finished());
5455
5456        drop(module);
5457
5458        assert_eq!(
5459            lock_snapshot(&snapshot).unwrap().state,
5460            ModuleState::Stopped
5461        );
5462        assert_snapshot_process_facts_cleared(&snapshot);
5463    }
5464
5465    #[cfg(unix)]
5466    #[tokio::test]
5467    async fn rescan_preserves_running_protocol_until_respawn() {
5468        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5469        let initial = ModuleSpec {
5470            module_id: "rescan-protocol".into(),
5471            program: PathBuf::from("/bin/sleep"),
5472            args: vec!["60".into()],
5473            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5474                .into_iter()
5475                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5476                .collect(),
5477            reserved: false,
5478            reserved_prefixes: vec![],
5479            protocol: ModuleProtocol::None,
5480            overlap: Default::default(),
5481        };
5482        let supervisor =
5483            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5484        let module = supervisor.spawn(initial.clone()).unwrap();
5485        assert!(module.status().unwrap().live);
5486        let mut next = initial;
5487        next.protocol = ModuleProtocol::Subc;
5488        module
5489            .update_configuration(next.clone(), HealthConfig::default(), None)
5490            .await
5491            .unwrap();
5492        assert!(
5493            module.status().unwrap().live,
5494            "rescan must not require HELLO from the old non-wire process"
5495        );
5496        let runtime = supervisor.runtime_config();
5497        let action = on_child_exit(
5498            &next,
5499            RestartPolicy::default(),
5500            &supervisor.registry,
5501            &module.inner.snapshot,
5502            &runtime.terminal_ring,
5503            &runtime.spawn_events,
5504            &runtime.child_roster,
5505            ExitReport {
5506                kind: ExitKind::Clean,
5507                code: Some(0),
5508                signal: None,
5509                at_ms: unix_ms_now(),
5510            },
5511        )
5512        .await;
5513        assert!(
5514            matches!(action, NextAction::Restart { .. }),
5515            "the old non-wire process's clean exit must restart"
5516        );
5517        module.drain().await.unwrap();
5518    }
5519
5520    #[cfg(unix)]
5521    #[tokio::test]
5522    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5523        use std::os::unix::fs::PermissionsExt;
5524        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5525        let script = dir.join("module.sh");
5526        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5527        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5528        let record_path = dir.join("live-children.json");
5529        let supervisor =
5530            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5531                .with_live_children_record(&record_path);
5532        for (program, args) in [
5533            (PathBuf::from("sleep"), vec!["60".into()]),
5534            (script, vec![]),
5535        ] {
5536            let spec = ModuleSpec {
5537                module_id: "image-identity".into(),
5538                program,
5539                args,
5540                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5541                    .into_iter()
5542                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5543                    .collect(),
5544                reserved: false,
5545                reserved_prefixes: vec![],
5546                protocol: ModuleProtocol::None,
5547                overlap: Default::default(),
5548            };
5549            let module = supervisor.spawn(spec).unwrap();
5550            #[cfg(target_os = "macos")]
5551            {
5552                // SETEXEC confirmation is asynchronous; the orphan record must
5553                // identify the final image, never the intermediate trampoline.
5554                let deadline = Instant::now() + Duration::from_secs(5);
5555                while crate::live_children::read_record(&record_path)
5556                    .unwrap()
5557                    .iter()
5558                    .all(|entry| entry.executable.is_none())
5559                {
5560                    assert!(Instant::now() < deadline, "module image was not confirmed");
5561                    tokio::time::sleep(Duration::from_millis(5)).await;
5562                }
5563            }
5564            let entry = crate::live_children::read_record(&record_path)
5565                .unwrap()
5566                .pop()
5567                .unwrap();
5568            let observed = subc_os::Process::open(entry.pid)
5569                .unwrap()
5570                .unwrap()
5571                .observe()
5572                .unwrap();
5573            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5574            module.drain().await.unwrap();
5575            assert_eq!(verdict, crate::live_children::IdentityVerdict::Matches);
5576        }
5577    }
5578
5579    #[cfg(unix)]
5580    fn http_fixture(
5581        dir: &std::path::Path,
5582        url: &str,
5583        threshold: u32,
5584    ) -> crate::daemon_config::ConfiguredModule {
5585        let path = dir.join("subc.jsonc");
5586        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5587            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5588            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5589            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5590            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5591        }}}).to_string()).unwrap();
5592        crate::daemon_config::load(&path)
5593            .unwrap()
5594            .unwrap()
5595            .modules
5596            .pop()
5597            .unwrap()
5598    }
5599
5600    #[cfg(unix)]
5601    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5602        timeout(Duration::from_secs(5), async {
5603            loop {
5604                if module.status().unwrap().health.status == status {
5605                    break;
5606                }
5607                sleep(Duration::from_millis(5)).await;
5608            }
5609        })
5610        .await
5611        .unwrap_or_else(|_| {
5612            panic!(
5613                "expected {status:?}, got {:?}",
5614                module.status().unwrap().health
5615            )
5616        });
5617    }
5618
5619    #[cfg(unix)]
5620    #[tokio::test]
5621    async fn http_health_status_flips_ok_failing_ok() {
5622        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5623        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5624        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5625        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5626        let serving_status = status.clone();
5627        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5628        let server = tokio::spawn(async move {
5629            loop {
5630                let (mut stream, _) = listener.accept().await.unwrap();
5631                let mut request = [0u8; 2048];
5632                let count = stream.read(&mut request).await.unwrap();
5633                assert!(count > 0, "a probe must send an HTTP request");
5634                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5635                let body = if code == 200 {
5636                    "ready"
5637                } else {
5638                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5639                };
5640                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5641                let _ = stream.write_all(response.as_bytes()).await;
5642            }
5643        });
5644        let configured = http_fixture(&dir, &url, 1000);
5645        let module =
5646            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5647                .supervise_configured_with_health(
5648                    configured.module_spec(),
5649                    true,
5650                    configured.health,
5651                    configured.drain_timeout_ms,
5652                    configured.restart,
5653                )
5654                .unwrap();
5655        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5656        status.store(503, std::sync::atomic::Ordering::SeqCst);
5657        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5658        assert!(module
5659            .status()
5660            .unwrap()
5661            .health
5662            .detail
5663            .unwrap()
5664            .contains("scratch failure"));
5665        status.store(200, std::sync::atomic::Ordering::SeqCst);
5666        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5667        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5668        let before = module.status().unwrap();
5669        let (spec, mut health) = module.configuration().unwrap();
5670        health.http = None;
5671        module
5672            .update_configuration(spec.clone(), health.clone(), Some(10))
5673            .await
5674            .unwrap();
5675        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5676        health.http = Some(url);
5677        module
5678            .update_configuration(spec, health, Some(10))
5679            .await
5680            .unwrap();
5681        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5682        assert_eq!(
5683            module.status().unwrap().pid,
5684            before.pid,
5685            "changing a probe must apply live, not restart its process"
5686        );
5687        let (spec, mut health) = module.configuration().unwrap();
5688        health.failure_threshold = 2;
5689        module
5690            .update_configuration(spec, health, Some(10))
5691            .await
5692            .unwrap();
5693        status.store(503, std::sync::atomic::Ordering::SeqCst);
5694        timeout(Duration::from_secs(5), async {
5695            while module.status().unwrap().spawn_generation == before.spawn_generation {
5696                sleep(Duration::from_millis(5)).await;
5697            }
5698        })
5699        .await
5700        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5701        module.drain().await.unwrap();
5702        server.abort();
5703    }
5704
5705    #[cfg(unix)]
5706    #[tokio::test]
5707    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5708        let dir = subc_test_support::TestTempDir::new("http-refused");
5709        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5710        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5711        drop(unused);
5712        let configured = http_fixture(&dir, &url, 2);
5713        let module =
5714            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5715                .supervise_configured_with_health(
5716                    configured.module_spec(),
5717                    true,
5718                    configured.health,
5719                    configured.drain_timeout_ms,
5720                    configured.restart,
5721                )
5722                .unwrap();
5723        let before = module.status().unwrap().spawn_generation;
5724        timeout(Duration::from_secs(5), async {
5725            loop {
5726                let status = module.status().unwrap();
5727                if status.spawn_generation > before {
5728                    assert!(status.lifetime_restarts > 0);
5729                    break;
5730                }
5731                sleep(Duration::from_millis(5)).await;
5732            }
5733        })
5734        .await
5735        .expect("sustained HTTP refusal must trigger the health restart policy");
5736        module.drain().await.unwrap();
5737    }
5738
5739    #[cfg(unix)]
5740    #[tokio::test]
5741    async fn http_health_timeout_honours_deadline() {
5742        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5743        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5744        let server = tokio::spawn(async move {
5745            let _held = listener.accept().await.unwrap();
5746            std::future::pending::<()>().await;
5747        });
5748        let error = timeout(
5749            Duration::from_secs(1),
5750            probe_http_health(&url, Duration::from_millis(10)),
5751        )
5752        .await
5753        .expect("the probe must enforce its own deadline")
5754        .unwrap_err();
5755        server.abort();
5756        assert!(error.to_string().contains("timed out"));
5757    }
5758
5759    #[cfg(unix)]
5760    #[tokio::test]
5761    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5762        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5763        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5764        let url = format!(
5765            "http://localhost:{}/healthz",
5766            listener.local_addr().unwrap().port()
5767        );
5768        let server = tokio::spawn(async move {
5769            let (mut stream, _) = listener.accept().await.unwrap();
5770            let mut request = [0u8; 2048];
5771            assert!(stream.read(&mut request).await.unwrap() > 0);
5772            stream
5773                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5774                .await
5775                .unwrap();
5776        });
5777        // The deadline only bounds a hang. A probe that never tried the IPv6
5778        // address would be refused on 127.0.0.1 and fail at once, so a longer
5779        // deadline does not weaken the assertion; one second timed out under a
5780        // loaded parallel test run.
5781        assert_eq!(
5782            probe_http_health(&url, Duration::from_secs(10))
5783                .await
5784                .unwrap()
5785                .status,
5786            HealthStatus::Ok
5787        );
5788        server.await.unwrap();
5789    }
5790
5791    #[cfg(unix)]
5792    #[tokio::test]
5793    async fn http_health_timeout_keeps_partial_status_and_body() {
5794        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5795        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5796        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5797        let server = tokio::spawn(async move {
5798            let (mut stream, _) = listener.accept().await.unwrap();
5799            let mut request = [0u8; 2048];
5800            assert!(stream.read(&mut request).await.unwrap() > 0);
5801            stream
5802                .write_all(
5803                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5804                )
5805                .await
5806                .unwrap();
5807            std::future::pending::<()>().await;
5808        });
5809        let error = probe_http_health(&url, Duration::from_secs(1))
5810            .await
5811            .unwrap_err()
5812            .to_string();
5813        server.abort();
5814        assert!(
5815            error.contains("timed out")
5816                && error.contains("503 Unavailable")
5817                && error.contains("partial diagnostic"),
5818            "{error}"
5819        );
5820    }
5821
5822    #[cfg(unix)]
5823    #[tokio::test]
5824    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5825        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5826        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5827        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5828        let server = tokio::spawn(async move {
5829            let (mut stream, _) = listener.accept().await.unwrap();
5830            let mut request = [0u8; 2048];
5831            assert!(stream.read(&mut request).await.unwrap() > 0);
5832            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5833            let response = format!(
5834                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5835                body.len()
5836            );
5837            stream.write_all(response.as_bytes()).await.unwrap();
5838        });
5839        let error = probe_http_health(&url, Duration::from_secs(1))
5840            .await
5841            .unwrap_err()
5842            .to_string();
5843        server.await.unwrap();
5844        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5845        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5846        assert!(!error.contains("not-in-diagnostic"));
5847    }
5848
5849    #[cfg(unix)]
5850    #[tokio::test]
5851    async fn http_health_real_nats_server_monitoring() {
5852        if std::process::Command::new("nats-server")
5853            .arg("--version")
5854            .env("XDG_DATA_HOME", std::env::temp_dir())
5855            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5856            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5857            .output()
5858            .is_err()
5859        {
5860            eprintln!(
5861                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5862            );
5863            return;
5864        }
5865        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5866        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5867        let port = monitor.local_addr().unwrap().port();
5868        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5869        let client_port = client.local_addr().unwrap().port();
5870        let config = dir.join("server.conf");
5871        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5872        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5873        configured.program = PathBuf::from("nats-server");
5874        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5875        drop(monitor);
5876        drop(client);
5877        let module =
5878            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5879                .supervise_configured_with_health(
5880                    configured.module_spec(),
5881                    true,
5882                    configured.health,
5883                    configured.drain_timeout_ms,
5884                    configured.restart,
5885                )
5886                .unwrap();
5887        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5888        module.drain().await.unwrap();
5889    }
5890
5891    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5892    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5893        let supervisor =
5894            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5895        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5896        let initial = ModuleSpec {
5897            module_id: "rescan-preserves-spawn-facts".to_string(),
5898            program: PathBuf::from("/spawned/module"),
5899            args: Vec::new(),
5900            env: Vec::new(),
5901            reserved: false,
5902            reserved_prefixes: Vec::new(),
5903            protocol: ModuleProtocol::Subc,
5904            overlap: Default::default(),
5905        };
5906        let module = supervisor.supervised_module(
5907            initial.clone(),
5908            supervisor.runtime_config(),
5909            snapshot,
5910            None,
5911        );
5912        let before = module.status().unwrap();
5913        let mut replacement = initial;
5914        replacement.program = PathBuf::from("/rescanned/replacement-module");
5915
5916        module
5917            .update_configuration(replacement, HealthConfig::default(), None)
5918            .await
5919            .unwrap();
5920
5921        let after = module.status().unwrap();
5922        assert_eq!(after.pid, before.pid);
5923        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5924        assert_eq!(after.spawned_from, before.spawned_from);
5925        drop(module);
5926    }
5927}
5928
5929fn unix_ms_now() -> u64 {
5930    SystemTime::now()
5931        .duration_since(UNIX_EPOCH)
5932        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5933        .unwrap_or(0)
5934}
5935
5936async fn supervise_loop(
5937    mut spec: ModuleSpec,
5938    mut runtime: SupervisorRuntimeConfig,
5939    registry: Arc<Registry>,
5940    process_liveness: Arc<SupervisorProcessLiveness>,
5941    snapshot: SharedSnapshot,
5942    mut child: Option<SupervisedChild>,
5943    mut commands: mpsc::Receiver<SupervisorCommand>,
5944) {
5945    let mut health_probe = HealthProbeRuntime::default();
5946    // All restart backoffs run here, including health and operator requests.
5947    // While one is pending the loop serves commands, so disable or drain can
5948    // cancel the replacement without spawning a process just to stop it.
5949    let mut pending_respawn: Option<PendingRespawn> = None;
5950    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5951    // before anything else so a stop that interrupted a swap runs at once.
5952    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5953    loop {
5954        #[cfg(target_os = "macos")]
5955        if let Some(active) = child.as_mut() {
5956            active.confirm_privacy_exec().await;
5957        }
5958        if let Some(scheduled) = runtime
5959            .scheduled_respawn
5960            .lock()
5961            .unwrap_or_else(|p| p.into_inner())
5962            .take()
5963        {
5964            pending_respawn = Some(scheduled);
5965        }
5966        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5967            pending_respawn = None;
5968            cancel_deferred_reload(
5969                &runtime,
5970                &spec.module_id,
5971                "respawn cancelled by a supervisor command",
5972            );
5973        }
5974        if child.is_none() && pending_respawn.is_none() {
5975            cancel_deferred_reload(
5976                &runtime,
5977                &spec.module_id,
5978                "respawn cancelled before a replacement was spawned",
5979            );
5980            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5981                state.respawn_pending = false;
5982                state.coalesced_restart_pending = false;
5983                if matches!(
5984                    state.state,
5985                    ModuleState::Restarting
5986                        | ModuleState::Starting
5987                        | ModuleState::Draining
5988                        | ModuleState::Unresponsive
5989                ) {
5990                    error!(module_id = %spec.module_id, state = ?state.state,
5991                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5992                    state.state = ModuleState::Failed;
5993                    clear_current_process_facts(state);
5994                }
5995            });
5996        }
5997        if let Some(command) = requeued.pop_front() {
5998            if !handle_supervisor_command(
5999                command,
6000                &mut spec,
6001                &mut runtime,
6002                &registry,
6003                &process_liveness,
6004                &snapshot,
6005                &mut child,
6006                &mut commands,
6007                &mut requeued,
6008            )
6009            .await
6010            {
6011                return;
6012            }
6013            if child.is_some() || !respawn_still_pending(&snapshot) {
6014                pending_respawn = None;
6015                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6016                    state.respawn_pending = false
6017                });
6018            }
6019            continue;
6020        }
6021        if child.is_some() {
6022            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
6023            let probe_sleep = sleep(health_probe.wake_after());
6024            tokio::pin!(probe_sleep);
6025            let active_child = child.as_mut().expect("child checked above");
6026            tokio::select! {
6027                wait_result = active_child.wait() => {
6028                    // Every arm below that gives up on the CHILD must keep the
6029                    // supervision task itself alive (child = None, loop
6030                    // continues into command-serving mode). Returning here
6031                    // closes the command channel, which makes the module
6032                    // permanently unrestartable in-band: a clean child exit
6033                    // of an enabled module once wedged the fleet this way
6034                    // ('supervisor command channel is closed') and required a
6035                    // full daemon restart to recover.
6036                    let exit_report = match wait_result {
6037                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6038                        Err(err) => {
6039                            active_child.drain_stderr(&spec.module_id).await;
6040                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6041                            // Every other exit path (on_child_exit's Clean/Crash arms,
6042                            // the reload-registration-failure path) records a terminal
6043                            // before moving on. Without one here, a module whose wait()
6044                            // itself errored (e.g. already reaped) leaves no terminal
6045                            // record at all -- an empty ring reads as "nothing died".
6046                            record_wait_error_terminal(
6047                                &spec.module_id,
6048                                &runtime.terminal_ring,
6049                                &runtime.spawn_events,
6050                            );
6051                            untrack_if_registration_released(
6052                                &process_liveness,
6053                                &registry,
6054                                &spec.module_id,
6055                                &snapshot,
6056                            );
6057                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6058                            child = None;
6059                            continue;
6060                        }
6061                    };
6062                    active_child.drain_stderr(&spec.module_id).await;
6063
6064                    let next = on_child_exit(
6065                        &spec,
6066                        runtime.restart_policy,
6067                        &registry,
6068                        &snapshot,
6069                        &runtime.terminal_ring,
6070                        &runtime.spawn_events,
6071                        &runtime.child_roster,
6072                        exit_report,
6073                    ).await;
6074                    // The exit is recorded, so a daemon shutdown may stop
6075                    // waiting for this child (see `SupervisedChild::wait`).
6076                    active_child.release_roster();
6077                    match next {
6078                        NextAction::Stop { registration_released } => {
6079                            if registration_released {
6080                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6081                            }
6082                            child = None;
6083                        }
6084                        NextAction::Restart { schedule } => {
6085                            let delay = schedule.map_or(
6086                                runtime.restart_policy.delay_for_restart(0),
6087                                |schedule| schedule.delay,
6088                            );
6089                            if let Some(schedule) = schedule {
6090                                log_crash_respawn(&spec.module_id, schedule);
6091                            }
6092                            // The exited child is fully recorded at this point,
6093                            // so release it and count the backoff down in the
6094                            // command-serving branch below rather than sleeping
6095                            // here: commands cannot be received from inside this
6096                            // select arm, and an operator disable or drain that
6097                            // arrives during the backoff must cancel the pending
6098                            // respawn instead of waiting for it to spawn first.
6099                            child = None;
6100                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6101                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6102                        }
6103                    }
6104                }
6105                command = commands.recv() => {
6106                    let Some(command) = command else {
6107                        return;
6108                    };
6109                    if !handle_supervisor_command(
6110                        command,
6111                        &mut spec,
6112                        &mut runtime,
6113                        &registry,
6114                        &process_liveness,
6115                        &snapshot,
6116                        &mut child,
6117                        &mut commands,
6118                        &mut requeued,
6119                    ).await {
6120                        return;
6121                    }
6122                }
6123                _ = &mut probe_sleep => {
6124                    if health_probe.due() {
6125                        run_health_probe_cycle(
6126                            &spec,
6127                            &runtime,
6128                            &registry,
6129                            &process_liveness,
6130                            &snapshot,
6131                            &mut child,
6132                        ).await;
6133                        if child.is_some() {
6134                            health_probe.schedule_next(&spec, runtime.health.cadence);
6135                        }
6136                    }
6137                }
6138            }
6139        } else if let Some(pending) = pending_respawn {
6140            tokio::select! {
6141                _ = sleep_until(pending.deadline) => {
6142                    pending_respawn = None;
6143                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6144                    // A command handled below while the backoff elapsed may
6145                    // have stopped the module; never respawn past an operator's
6146                    // disable or drain.
6147                    if !respawn_still_pending(&snapshot) {
6148                        continue;
6149                    }
6150                    // The daemon began shutting down during the backoff: the
6151                    // spawn would be refused anyway, and refusing it here
6152                    // leaves the module stopped instead of reporting a
6153                    // failed restart.
6154                    if runtime.child_roster.is_closed() {
6155                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6156                            state.state = ModuleState::Stopped;
6157                        });
6158                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6159                        continue;
6160                    }
6161                    if let Err(err) = release_dead_registration(
6162                        &registry,
6163                        runtime.forwarding.as_deref(),
6164                        &snapshot,
6165                        &spec.module_id,
6166                    ).await {
6167                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6168                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6169                        continue;
6170                    }
6171
6172                    if matches!(pending.kind, RespawnKind::Reload) {
6173                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6174                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6175                        if let Some(reply) = reply { let _ = reply.send(result); }
6176                        continue;
6177                    }
6178                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6179                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6180                        Ok(next_child) => {
6181                            child = Some(next_child);
6182                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6183                        }
6184                        Err(err) => {
6185                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6186                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6187                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6188                        }
6189                    }
6190                }
6191                command = commands.recv() => {
6192                    let Some(command) = command else {
6193                        return;
6194                    };
6195                    if !handle_supervisor_command(
6196                        command,
6197                        &mut spec,
6198                        &mut runtime,
6199                        &registry,
6200                        &process_liveness,
6201                        &snapshot,
6202                        &mut child,
6203                        &mut commands,
6204                        &mut requeued,
6205                    ).await {
6206                        return;
6207                    }
6208                    // Reconcile the pending respawn with what the command did:
6209                    // a start may already have spawned a fresh child,
6210                    // while a disable or drain moved the snapshot out of the
6211                    // state the respawn was counting down from.
6212                    if child.is_some() || !respawn_still_pending(&snapshot) {
6213                        pending_respawn = None;
6214                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6215                    }
6216                }
6217            }
6218        } else {
6219            let Some(command) = commands.recv().await else {
6220                return;
6221            };
6222            if !handle_supervisor_command(
6223                command,
6224                &mut spec,
6225                &mut runtime,
6226                &registry,
6227                &process_liveness,
6228                &snapshot,
6229                &mut child,
6230                &mut commands,
6231                &mut requeued,
6232            )
6233            .await
6234            {
6235                return;
6236            }
6237        }
6238    }
6239}
6240
6241fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6242    info!(
6243        module_id,
6244        restart_in_window = schedule.restart_in_window,
6245        delay_ms = schedule.delay.as_millis() as u64,
6246        "respawning after crash"
6247    );
6248}
6249
6250/// Whether the respawn a backoff was counting down to is still wanted. A
6251/// disable or drain handled while the backoff elapsed moves the snapshot out
6252/// of `Restarting`, and the operator's stop must win over the pending respawn,
6253/// so every sleep-then-spawn path re-validates against the live snapshot
6254/// instead of assuming the state it left behind still holds.
6255fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6256    matches!(
6257        lock_snapshot(snapshot),
6258        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6259    )
6260}
6261
6262enum NextAction {
6263    Stop {
6264        registration_released: bool,
6265    },
6266    Restart {
6267        schedule: Option<CrashRestartSchedule>,
6268    },
6269}
6270
6271#[allow(clippy::too_many_arguments)]
6272async fn handle_supervisor_command(
6273    command: SupervisorCommand,
6274    spec: &mut ModuleSpec,
6275    runtime: &mut SupervisorRuntimeConfig,
6276    registry: &Arc<Registry>,
6277    process_liveness: &SupervisorProcessLiveness,
6278    snapshot: &SharedSnapshot,
6279    child: &mut Option<SupervisedChild>,
6280    commands: &mut mpsc::Receiver<SupervisorCommand>,
6281    requeued: &mut VecDeque<SupervisorCommand>,
6282) -> bool {
6283    match command {
6284        SupervisorCommand::Drain { reply } => {
6285            // A plain stop runs no forwarding drain, so nothing reaches the
6286            // module over its connection before the wait: ask by signal.
6287            let result = drain_optional_child(
6288                &spec.module_id,
6289                spec.protocol,
6290                StopNotice::NotSent,
6291                registry,
6292                runtime.forwarding.as_deref(),
6293                snapshot,
6294                &runtime.terminal_ring,
6295                &runtime.spawn_events,
6296                child,
6297                runtime.drain_timeout,
6298                ModuleState::Stopped,
6299                None,
6300            )
6301            .await;
6302            let registration_released = result.is_ok();
6303            let _ = reply.send(result);
6304            if registration_released {
6305                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6306            }
6307            false
6308        }
6309        SupervisorCommand::Retire { reply } => {
6310            let result = async {
6311                let stop_notice = begin_forwarding_drain_if_configured(
6312                    spec,
6313                    runtime,
6314                    registry,
6315                    snapshot,
6316                    None,
6317                    RouteCloseReason::Disable,
6318                )
6319                .await?;
6320                drain_optional_child(
6321                    &spec.module_id,
6322                    spec.protocol,
6323                    stop_notice,
6324                    registry,
6325                    runtime.forwarding.as_deref(),
6326                    snapshot,
6327                    &runtime.terminal_ring,
6328                    &runtime.spawn_events,
6329                    child,
6330                    runtime.drain_timeout,
6331                    ModuleState::Stopped,
6332                    None,
6333                )
6334                .await
6335            }
6336            .await;
6337            let registration_released = result.is_ok();
6338            let _ = reply.send(result);
6339            if registration_released {
6340                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6341            }
6342            false
6343        }
6344        SupervisorCommand::Restart {
6345            drain_timeout_ms,
6346            received_at_generation,
6347            queued_at,
6348            reply,
6349        } => {
6350            // Without this line a restart that waited in the queue (behind a
6351            // health probe cycle or another command) was invisible: the log
6352            // showed only the drain timing out, minutes after the operator's call.
6353            info!(
6354                module_id = %spec.module_id,
6355                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6356                "restart command dequeued"
6357            );
6358            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6359            // caller whose own request lane rides the module being restarted: the
6360            // caller's in-flight request keeps the drain from quiescing, the drain
6361            // keeps the restart from completing, and the completion keeps the reply
6362            // from releasing the caller — so the drain always timed out and cut the
6363            // initiator with a GOODBYE, even on a healthy module. Replying once the
6364            // restart is validated lets a self-lane caller settle, which is exactly
6365            // what makes the drain succeed. Completion is observable via
6366            // supervisor.list / module status; a post-ack failure lands the module
6367            // in a visible terminal state below rather than in a reply nobody can
6368            // receive.
6369            let validation = match lock_snapshot(snapshot) {
6370                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6371                    module_id: spec.module_id.clone(),
6372                }),
6373                Ok(_) => Ok(()),
6374                Err(err) => Err(err),
6375            };
6376            let initiated = validation.is_ok();
6377            let _ = reply.send(validation);
6378            // A restart asks for a fresh process. Commands run one at a time,
6379            // so a restart queued behind another restart (two operator calls
6380            // in quick succession) is dequeued the moment the first one has
6381            // spawned its replacement -- before that process has sent HELLO.
6382            // Running it would drain and kill the process the first restart
6383            // just produced, which is the opposite of what both callers asked
6384            // for. If a process spawned after this request was received is
6385            // still supervised, the request is already satisfied. Not when the
6386            // configuration changed since that spawn: then the newer process
6387            // predates the spec this restart may exist to apply.
6388            let satisfied_by_generation = if initiated && child.is_some() {
6389                lock_snapshot(snapshot).ok().and_then(|state| {
6390                    (state.spawn_generation > received_at_generation
6391                        && !state.configuration_updated_since_spawn)
6392                        .then_some(state.spawn_generation)
6393                })
6394            } else {
6395                None
6396            };
6397            let satisfied_by_pending = initiated
6398                && child.is_none()
6399                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6400                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6401                    if pending {
6402                        state.coalesced_restart_pending = true;
6403                    }
6404                    pending
6405                });
6406            if satisfied_by_pending {
6407                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6408            } else if let Some(generation) = satisfied_by_generation {
6409                info!(
6410                    module_id = %spec.module_id,
6411                    received_at_generation,
6412                    "restart already satisfied by generation {generation}; not restarting again"
6413                );
6414            } else if initiated {
6415                // Precedence: this restart's operator override, else the module's
6416                // configured budget (already resolved into the runtime).
6417                let drain_timeout = drain_timeout_ms
6418                    .map(Duration::from_millis)
6419                    .unwrap_or(runtime.drain_timeout);
6420                if let Err(err) = restart_child(
6421                    spec,
6422                    runtime,
6423                    registry,
6424                    process_liveness,
6425                    snapshot,
6426                    child,
6427                    drain_timeout,
6428                )
6429                .await
6430                {
6431                    warn!(
6432                        module_id = %spec.module_id,
6433                        error = %err,
6434                        "operator restart failed after initiation ack; module state carries the outcome"
6435                    );
6436                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6437                        state.state = ModuleState::Failed;
6438                        clear_current_process_facts(state);
6439                    });
6440                }
6441            }
6442            true
6443        }
6444        SupervisorCommand::Reload { reply } => {
6445            let result =
6446                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6447            if result.is_ok()
6448                && runtime
6449                    .scheduled_respawn
6450                    .lock()
6451                    .unwrap_or_else(|p| p.into_inner())
6452                    .as_ref()
6453                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6454            {
6455                *runtime
6456                    .deferred_reload_reply
6457                    .lock()
6458                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6459            } else {
6460                let _ = reply.send(result);
6461            }
6462            true
6463        }
6464        SupervisorCommand::SetEnabled { enabled, reply } => {
6465            let result = set_child_enabled(
6466                spec,
6467                runtime,
6468                registry,
6469                process_liveness,
6470                snapshot,
6471                child,
6472                enabled,
6473            )
6474            .await;
6475            let _ = reply.send(result);
6476            true
6477        }
6478        SupervisorCommand::UpdateConfiguration {
6479            spec: next_spec,
6480            health,
6481            drain_timeout_ms,
6482            reply,
6483        } => {
6484            if let Some(handle) = &runtime.supervisor_handle {
6485                handle.apply_identity_configuration(&next_spec);
6486            }
6487            *spec = next_spec;
6488            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6489                state.configuration_updated_since_spawn = true;
6490            });
6491            let health_changed = runtime.health != health;
6492            runtime.health = health;
6493            // Reset the cadence and old endpoint's failure streak on a live
6494            // health-policy change rather than waiting for its old deadline.
6495            if health_changed {
6496                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6497                    state.health = ModuleHealthStatus::default();
6498                });
6499            }
6500            runtime.drain_timeout = drain_timeout_ms
6501                .map(Duration::from_millis)
6502                .unwrap_or(runtime.default_drain_timeout);
6503            *runtime
6504                .effective_drain_timeout
6505                .lock()
6506                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6507            let _ = reply.send(());
6508            true
6509        }
6510        SupervisorCommand::Swap {
6511            ready_timeout,
6512            reply,
6513        } => {
6514            let end = swap::run_swap(
6515                spec,
6516                runtime,
6517                registry,
6518                process_liveness,
6519                snapshot,
6520                child,
6521                commands,
6522                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6523                reply,
6524            )
6525            .await;
6526            requeued.extend(end.requeue);
6527            true
6528        }
6529    }
6530}
6531
6532async fn restart_child(
6533    spec: &ModuleSpec,
6534    runtime: &SupervisorRuntimeConfig,
6535    registry: &Registry,
6536    process_liveness: &SupervisorProcessLiveness,
6537    snapshot: &SharedSnapshot,
6538    child: &mut Option<SupervisedChild>,
6539    drain_timeout: Duration,
6540) -> Result<(), SuperviseError> {
6541    // Restart cycles a running module; it must not silently start a disabled one.
6542    if !lock_snapshot(snapshot)?.enabled {
6543        return Err(SuperviseError::Disabled {
6544            module_id: spec.module_id.clone(),
6545        });
6546    }
6547    let stop_notice = begin_forwarding_drain_with_timeout(
6548        spec,
6549        runtime,
6550        registry,
6551        snapshot,
6552        None,
6553        RouteCloseReason::Restart,
6554        drain_timeout,
6555    )
6556    .await?;
6557
6558    if child.is_some() {
6559        drain_optional_child(
6560            &spec.module_id,
6561            spec.protocol,
6562            stop_notice,
6563            registry,
6564            runtime.forwarding.as_deref(),
6565            snapshot,
6566            &runtime.terminal_ring,
6567            &runtime.spawn_events,
6568            child,
6569            drain_timeout,
6570            ModuleState::Restarting,
6571            Some(true),
6572        )
6573        .await?;
6574    } else {
6575        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6576            state.enabled = true;
6577            state.state = ModuleState::Restarting;
6578            clear_current_process_facts(state);
6579        })?;
6580        release_dead_registration(
6581            registry,
6582            runtime.forwarding.as_deref(),
6583            snapshot,
6584            &spec.module_id,
6585        )
6586        .await?;
6587    }
6588
6589    reset_restart_count(snapshot, &spec.module_id)?;
6590    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6591    schedule_respawn(
6592        runtime,
6593        snapshot,
6594        &spec.module_id,
6595        runtime.restart_policy.backoff,
6596        RespawnKind::Spawn,
6597    )
6598}
6599
6600async fn reload_child(
6601    spec: &ModuleSpec,
6602    runtime: &SupervisorRuntimeConfig,
6603    registry: &Registry,
6604    process_liveness: &SupervisorProcessLiveness,
6605    snapshot: &SharedSnapshot,
6606    child: &mut Option<SupervisedChild>,
6607) -> Result<(), SuperviseError> {
6608    // Reload cycles a running module; it must not silently start a disabled one.
6609    if !lock_snapshot(snapshot)?.enabled {
6610        return Err(SuperviseError::Disabled {
6611            module_id: spec.module_id.clone(),
6612        });
6613    }
6614    let stop_notice = begin_forwarding_drain(
6615        spec,
6616        runtime,
6617        registry,
6618        snapshot,
6619        Some(true),
6620        RouteCloseReason::Reload,
6621    )
6622    .await?;
6623
6624    if child.is_some() {
6625        drain_optional_child(
6626            &spec.module_id,
6627            spec.protocol,
6628            stop_notice,
6629            registry,
6630            runtime.forwarding.as_deref(),
6631            snapshot,
6632            &runtime.terminal_ring,
6633            &runtime.spawn_events,
6634            child,
6635            runtime.drain_timeout,
6636            ModuleState::Restarting,
6637            Some(true),
6638        )
6639        .await?;
6640    } else {
6641        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6642            state.enabled = true;
6643            state.state = ModuleState::Restarting;
6644            clear_current_process_facts(state);
6645        })?;
6646        release_dead_registration(
6647            registry,
6648            runtime.forwarding.as_deref(),
6649            snapshot,
6650            &spec.module_id,
6651        )
6652        .await?;
6653    }
6654
6655    reset_restart_count(snapshot, &spec.module_id)?;
6656    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6657    schedule_respawn(
6658        runtime,
6659        snapshot,
6660        &spec.module_id,
6661        runtime.restart_policy.backoff,
6662        RespawnKind::Reload,
6663    )
6664}
6665
6666async fn finish_reload_child(
6667    spec: &ModuleSpec,
6668    runtime: &SupervisorRuntimeConfig,
6669    registry: &Registry,
6670    process_liveness: &SupervisorProcessLiveness,
6671    snapshot: &SharedSnapshot,
6672    child: &mut Option<SupervisedChild>,
6673) -> Result<(), SuperviseError> {
6674    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6675    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6676        Ok(next_child) => next_child,
6677        Err(err) => {
6678            return handle_reload_spawn_failure(
6679                spec,
6680                runtime,
6681                process_liveness,
6682                snapshot,
6683                child,
6684                format!("new child failed to spawn: {err}"),
6685            )
6686            .await;
6687        }
6688    };
6689    *child = Some(next_child);
6690
6691    let wait_outcome = {
6692        let active_child = child.as_mut().expect("new reload child was just stored");
6693        wait_for_registration_after_reload(
6694            registry,
6695            &spec.module_id,
6696            snapshot,
6697            active_child,
6698            REGISTRY_RELEASE_TIMEOUT,
6699        )
6700        .await?
6701    };
6702
6703    match wait_outcome {
6704        RegistrationWaitOutcome::Registered => {
6705            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6706            Ok(())
6707        }
6708        RegistrationWaitOutcome::Exited(exit_report) => {
6709            if let Some(active_child) = child.as_mut() {
6710                active_child.drain_stderr(&spec.module_id).await;
6711            }
6712            // Keep the reaped child's roster guard until its terminal is written.
6713            // Shutdown waits on that guard, not on the child Option used for respawn.
6714            let mut exited_child = child.take().expect("exited reload child is still stored");
6715            #[cfg(test)]
6716            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6717                gate.reached.notify_one();
6718                gate.resume.notified().await;
6719            }
6720            let result = handle_reload_child_registration_failure(
6721                spec,
6722                runtime,
6723                registry,
6724                process_liveness,
6725                snapshot,
6726                child,
6727                ReloadRegistrationFailure {
6728                    exit_report: registration_failure_exit_report(exit_report),
6729                    reason: exited_child
6730                        .spawn_failure
6731                        .clone()
6732                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6733                },
6734            )
6735            .await;
6736            exited_child.release_roster();
6737            result
6738        }
6739        RegistrationWaitOutcome::TimedOut => {
6740            let mut timed_out_child = child
6741                .take()
6742                .expect("timed-out reload child is still running");
6743            timed_out_child
6744                .start_kill()
6745                .map_err(|source| SuperviseError::Kill {
6746                    module_id: spec.module_id.clone(),
6747                    source,
6748                })?;
6749            let status = timed_out_child
6750                .wait()
6751                .await
6752                .map_err(|source| SuperviseError::Wait {
6753                    module_id: spec.module_id.clone(),
6754                    source,
6755                })?;
6756            timed_out_child.drain_stderr(&spec.module_id).await;
6757            handle_reload_child_registration_failure(
6758                spec,
6759                runtime,
6760                registry,
6761                process_liveness,
6762                snapshot,
6763                child,
6764                ReloadRegistrationFailure {
6765                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6766                        snapshot,
6767                        &timed_out_child,
6768                        &status,
6769                    )),
6770                    reason: format!(
6771                        "new child did not register within {:?}",
6772                        REGISTRY_RELEASE_TIMEOUT
6773                    ),
6774                },
6775            )
6776            .await
6777        }
6778    }
6779}
6780
6781async fn set_child_enabled(
6782    spec: &ModuleSpec,
6783    runtime: &SupervisorRuntimeConfig,
6784    registry: &Registry,
6785    process_liveness: &SupervisorProcessLiveness,
6786    snapshot: &SharedSnapshot,
6787    child: &mut Option<SupervisedChild>,
6788    enabled: bool,
6789) -> Result<bool, SuperviseError> {
6790    let (current_enabled, current_state, respawn_pending) = {
6791        let state = lock_snapshot(snapshot)?;
6792        (state.enabled, state.state, state.respawn_pending)
6793    };
6794    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6795    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6796    // clean (Stopped) has no live process and no other in-band recovery — the
6797    // operator's start is the explicit recovery act and resets the budget. Without
6798    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6799    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6800    // the one providing every agent's shell.
6801    let revive_terminal = enabled
6802        && current_enabled
6803        && child.is_none()
6804        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6805            || (current_state == ModuleState::Restarting && !respawn_pending));
6806    if current_enabled == enabled && !revive_terminal {
6807        return Ok(false);
6808    }
6809
6810    if enabled {
6811        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6812            state.enabled = true;
6813            state.state = ModuleState::Starting;
6814            clear_current_process_facts(state);
6815        })?;
6816        #[cfg(test)]
6817        if runtime.test_seed_stale_facts_before_enable_spawn {
6818            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6819                state.process_alive = true;
6820                state.pid = Some(41);
6821                state.spawned_at_ms = Some(42);
6822                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6823                state.spawned_file_identity = Some(SpawnedFileIdentity {
6824                    device: 43,
6825                    inode: 44,
6826                });
6827            })?;
6828        }
6829        release_dead_registration(
6830            registry,
6831            runtime.forwarding.as_deref(),
6832            snapshot,
6833            &spec.module_id,
6834        )
6835        .await?;
6836        reset_restart_count(snapshot, &spec.module_id)?;
6837        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6838        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6839            Ok(next_child) => next_child,
6840            Err(err) => {
6841                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6842                    state.state = ModuleState::Failed;
6843                    clear_current_process_facts(state);
6844                }) {
6845                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6846                }
6847                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6848                return Err(err);
6849            }
6850        };
6851        *child = Some(next_child);
6852        debug!(module_id = %spec.module_id, "supervised module enabled");
6853        Ok(true)
6854    } else {
6855        let stop_notice = begin_forwarding_drain_if_configured(
6856            spec,
6857            runtime,
6858            registry,
6859            snapshot,
6860            Some(false),
6861            RouteCloseReason::Disable,
6862        )
6863        .await?;
6864        drain_optional_child(
6865            &spec.module_id,
6866            spec.protocol,
6867            stop_notice,
6868            registry,
6869            runtime.forwarding.as_deref(),
6870            snapshot,
6871            &runtime.terminal_ring,
6872            &runtime.spawn_events,
6873            child,
6874            runtime.drain_timeout,
6875            ModuleState::Disabled,
6876            Some(false),
6877        )
6878        .await?;
6879        debug!(module_id = %spec.module_id, "supervised module disabled");
6880        Ok(true)
6881    }
6882}
6883
6884#[allow(clippy::too_many_arguments)]
6885async fn on_child_exit(
6886    spec: &ModuleSpec,
6887    policy: RestartPolicy,
6888    registry: &Registry,
6889    snapshot: &SharedSnapshot,
6890    terminal_ring: &Arc<Mutex<TerminalRing>>,
6891    spawn_events: &SpawnEventFeed,
6892    roster: &ChildRoster,
6893    exit_report: ExitReport,
6894) -> NextAction {
6895    // Once the daemon has begun shutting down, no exit is a crash to recover
6896    // from: the module is exiting because the daemon is going away (EOF on its
6897    // connection, or a service manager signalling the whole cgroup). Record it
6898    // as such and never schedule a respawn, which would only start a process
6899    // for the shutdown to end again.
6900    if roster.is_closed() {
6901        return on_child_exit_during_daemon_shutdown(
6902            spec,
6903            registry,
6904            snapshot,
6905            terminal_ring,
6906            spawn_events,
6907            exit_report,
6908        )
6909        .await;
6910    }
6911    // Every stop the supervisor itself asks for (operator stop, disable,
6912    // restart, reload, swap, a health restart, a drain that runs out of budget)
6913    // takes the child out of the supervise loop and reaps it in
6914    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
6915    // that reaches this point was not requested by the daemon.
6916    //
6917    // For a subc-wire module a clean exit is still a stop: those modules are
6918    // written to re-raise SIGTERM, so a stray outside signal already reads as a
6919    // crash, and exiting 0 is a deliberate choice the module made. A
6920    // `protocol: "none"` module is a stock program we cannot change, and many
6921    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
6922    // stop would leave the module down for good after any stray signal, so it
6923    // goes through the crash path instead: it spends restart budget, respawns
6924    // with the crash backoff, and ends `failed` when the budget runs out.
6925    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6926        && running_protocol(spec, snapshot) == ModuleProtocol::None;
6927    match exit_report.kind {
6928        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6929            info!(
6930                module_id = %spec.module_id,
6931                exit_code = ?exit_report.code,
6932                exit_signal = ?exit_report.signal,
6933                "supervised module exited cleanly"
6934            );
6935            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6936                state.state = ModuleState::Stopped;
6937                clear_current_process_facts(state);
6938                state.last_exit = Some(exit_report.clone());
6939            }) {
6940                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6941            }
6942            record_terminal(
6943                &spec.module_id,
6944                terminal_ring,
6945                spawn_events,
6946                &exit_report,
6947                TerminalDisposition::Stopped,
6948            );
6949            let registration_released = match wait_for_registration_release(
6950                registry,
6951                &spec.module_id,
6952                REGISTRY_RELEASE_TIMEOUT,
6953            )
6954            .await
6955            {
6956                Ok(()) => true,
6957                Err(err) => {
6958                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6959                    false
6960                }
6961            };
6962            NextAction::Stop {
6963                registration_released,
6964            }
6965        }
6966        ExitKind::Clean | ExitKind::Crash => {
6967            if unrequested_clean_exit_of_protocol_none {
6968                warn!(
6969                    module_id = %spec.module_id,
6970                    exit_code = ?exit_report.code,
6971                    exit_signal = ?exit_report.signal,
6972                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
6973                );
6974            } else {
6975                warn!(
6976                    module_id = %spec.module_id,
6977                    exit_code = ?exit_report.code,
6978                    exit_signal = ?exit_report.signal,
6979                    "supervised module exited abnormally (crash)"
6980                );
6981            }
6982            let mut restart_schedule = None;
6983            let mut disposition = TerminalDisposition::Disabled;
6984            // Set only when the budget is what stopped the module, so the
6985            // terminal record says which limit was hit rather than leaving
6986            // `failed` to be read as "crashed once, badly".
6987            let mut disposition_detail = lock_snapshot(snapshot)
6988                .ok()
6989                .and_then(|mut state| state.spawn_failure.take());
6990            let now = Instant::now();
6991            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6992                clear_current_process_facts(state);
6993                state.last_exit = Some(exit_report.clone());
6994                if state.enabled {
6995                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
6996                        state.state = ModuleState::Restarting;
6997                        restart_schedule = Some(schedule);
6998                        disposition = TerminalDisposition::Restarting;
6999                    } else {
7000                        state.state = ModuleState::Failed;
7001                        disposition = TerminalDisposition::Failed;
7002                        let budget = policy.budget_exhausted_detail();
7003                        disposition_detail =
7004                            Some(disposition_detail.take().map_or_else(
7005                                || budget.clone(),
7006                                |cause| format!("{cause}; {budget}"),
7007                            ));
7008                    }
7009                } else {
7010                    state.state = ModuleState::Disabled;
7011                    disposition = TerminalDisposition::Disabled;
7012                }
7013            }) {
7014                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7015                return NextAction::Stop {
7016                    registration_released: false,
7017                };
7018            }
7019            if disposition == TerminalDisposition::Failed {
7020                // The window is in the message, not only in the fields: this line
7021                // is read in a scrollback where a bare `max_restarts=3` reads as a
7022                // lifetime cap and sends the operator looking for three crashes
7023                // that never happened together.
7024                error!(
7025                    module_id = %spec.module_id,
7026                    max_restarts = policy.max_restarts,
7027                    window_secs = policy.window.as_secs(),
7028                    "module stopped: {}",
7029                    policy.budget_exhausted_detail()
7030                );
7031            }
7032            record_terminal_with_detail(
7033                &spec.module_id,
7034                terminal_ring,
7035                spawn_events,
7036                &exit_report,
7037                disposition,
7038                disposition_detail,
7039            );
7040
7041            if let Some(schedule) = restart_schedule {
7042                NextAction::Restart {
7043                    schedule: Some(schedule),
7044                }
7045            } else {
7046                let registration_released = match wait_for_registration_release(
7047                    registry,
7048                    &spec.module_id,
7049                    REGISTRY_RELEASE_TIMEOUT,
7050                )
7051                .await
7052                {
7053                    Ok(()) => true,
7054                    Err(err) => {
7055                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7056                        false
7057                    }
7058                };
7059                NextAction::Stop {
7060                    registration_released,
7061                }
7062            }
7063        }
7064        ExitKind::DeliberateSeverance => {
7065            warn!(
7066                module_id = %spec.module_id,
7067                exit_code = ?exit_report.code,
7068                exit_signal = ?exit_report.signal,
7069                "supervised module exited after deliberate connection severance"
7070            );
7071            let mut should_restart = false;
7072            let mut disposition = TerminalDisposition::Disabled;
7073            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7074                clear_current_process_facts(state);
7075                state.last_exit = Some(exit_report.clone());
7076                state.lifetime_restarts += 1;
7077                if state.enabled {
7078                    state.state = ModuleState::Restarting;
7079                    should_restart = true;
7080                    disposition = TerminalDisposition::Restarting;
7081                } else {
7082                    state.state = ModuleState::Disabled;
7083                }
7084            }) {
7085                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7086                return NextAction::Stop {
7087                    registration_released: false,
7088                };
7089            }
7090            record_terminal(
7091                &spec.module_id,
7092                terminal_ring,
7093                spawn_events,
7094                &exit_report,
7095                disposition,
7096            );
7097
7098            if should_restart {
7099                NextAction::Restart { schedule: None }
7100            } else {
7101                let registration_released = match wait_for_registration_release(
7102                    registry,
7103                    &spec.module_id,
7104                    REGISTRY_RELEASE_TIMEOUT,
7105                )
7106                .await
7107                {
7108                    Ok(()) => true,
7109                    Err(err) => {
7110                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7111                        false
7112                    }
7113                };
7114                NextAction::Stop {
7115                    registration_released,
7116                }
7117            }
7118        }
7119    }
7120}
7121
7122async fn on_child_exit_during_daemon_shutdown(
7123    spec: &ModuleSpec,
7124    registry: &Registry,
7125    snapshot: &SharedSnapshot,
7126    terminal_ring: &Arc<Mutex<TerminalRing>>,
7127    spawn_events: &SpawnEventFeed,
7128    exit_report: ExitReport,
7129) -> NextAction {
7130    info!(
7131        module_id = %spec.module_id,
7132        exit_code = ?exit_report.code,
7133        exit_signal = ?exit_report.signal,
7134        exit_kind = ?exit_report.kind,
7135        "supervised module exited during daemon shutdown; not restarting it"
7136    );
7137    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7138        state.state = ModuleState::Stopped;
7139        clear_current_process_facts(state);
7140        state.last_exit = Some(exit_report.clone());
7141    }) {
7142        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7143    }
7144    record_terminal(
7145        &spec.module_id,
7146        terminal_ring,
7147        spawn_events,
7148        &exit_report,
7149        TerminalDisposition::DaemonShutdown,
7150    );
7151    let registration_released =
7152        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7153            .await
7154            .is_ok();
7155    NextAction::Stop {
7156        registration_released,
7157    }
7158}
7159
7160fn record_wait_error_terminal(
7161    module_id: &str,
7162    terminal_ring: &Arc<Mutex<TerminalRing>>,
7163    spawn_events: &SpawnEventFeed,
7164) {
7165    record_terminal(
7166        module_id,
7167        terminal_ring,
7168        spawn_events,
7169        &wait_error_exit_report(),
7170        TerminalDisposition::Failed,
7171    );
7172}
7173
7174fn record_terminal(
7175    module_id: &str,
7176    terminal_ring: &Arc<Mutex<TerminalRing>>,
7177    spawn_events: &SpawnEventFeed,
7178    exit_report: &ExitReport,
7179    disposition: TerminalDisposition,
7180) {
7181    record_terminal_with_detail(
7182        module_id,
7183        terminal_ring,
7184        spawn_events,
7185        exit_report,
7186        disposition,
7187        None,
7188    );
7189}
7190
7191/// The ring lock is held only to capture the read (see
7192/// `TerminalJournal::capture_read`), so this module's exits keep recording
7193/// while the journal files are read. Blocking: it reads files.
7194fn durable_terminal_history_of(
7195    terminal_ring: &Mutex<TerminalRing>,
7196    module_id: &str,
7197) -> subc_control::TerminalHistory {
7198    let read = terminal_ring
7199        .lock()
7200        .unwrap_or_else(|p| p.into_inner())
7201        .capture_durable_history();
7202    read.read(module_id)
7203}
7204
7205fn record_terminal_with_detail(
7206    module_id: &str,
7207    terminal_ring: &Arc<Mutex<TerminalRing>>,
7208    spawn_events: &SpawnEventFeed,
7209    exit_report: &ExitReport,
7210    disposition: TerminalDisposition,
7211    disposition_detail: Option<String>,
7212) {
7213    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7214    let record = TerminalRecord {
7215        exit_code: exit_report.code,
7216        exit_signal: exit_report.signal,
7217        at_ms: exit_report.at_ms,
7218        disposition,
7219        exit_kind: exit_report.kind.into(),
7220        disposition_detail,
7221    };
7222    terminal_ring
7223        .lock()
7224        .unwrap_or_else(|poisoned| poisoned.into_inner())
7225        .record_exit(module_id, record);
7226}
7227
7228fn untrack_if_registration_released(
7229    process_liveness: &SupervisorProcessLiveness,
7230    registry: &Registry,
7231    module_id: &str,
7232    snapshot: &SharedSnapshot,
7233) {
7234    match registry.get_module(module_id) {
7235        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7236        Ok(Some(_)) => {}
7237        Err(err) => {
7238            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7239        }
7240    }
7241}
7242
7243/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7244/// then apply the module's configured entries minus daemon-private capture keys.
7245///
7246/// Separated from `spawn_child` only so it can be asserted without spawning a
7247/// process — a duplicate of this logic in a test would pass while the real one
7248/// drifted, which is the defect class this function exists to avoid.
7249/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7250/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7251/// either and the argument would stop a stock binary from starting at all.
7252/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7253///
7254/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7255/// through [`apply_wire_spawn_args_for_role`].
7256#[cfg(test)]
7257fn apply_wire_spawn_args(
7258    command: &mut Command,
7259    spec: &ModuleSpec,
7260    connection_file_path: Option<&std::path::Path>,
7261    handle: Option<&SupervisorHandle>,
7262) -> Result<Option<NonceHandoff>, SuperviseError> {
7263    apply_wire_spawn_args_for_role(
7264        command,
7265        spec,
7266        connection_file_path,
7267        handle,
7268        SpawnRole::Plain,
7269    )
7270}
7271
7272/// The read end of a spawn's launch-nonce pipe, prepared by
7273/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7274/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7275/// handoff and keeps only the environment copy.
7276#[cfg(unix)]
7277type NonceHandoff = subc_os::LaunchNonceHandoff;
7278#[cfg(not(unix))]
7279type NonceHandoff = std::convert::Infallible;
7280
7281/// Prepare wire identity for a plain spawn or a swap candidate.
7282///
7283/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7284/// a separate candidate token so the still-serving incumbent and its consumers
7285/// keep their nonce. Both records are installed before the process exists, so
7286/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7287///
7288/// On Unix the nonce is delivered only through a pipe. It is written into
7289/// a pipe whose read end the child gets as descriptor 3, named by
7290/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7291/// process of the same user cannot read it with `ps eww`. That handoff is
7292/// returned rather than installed here, because installing it replaces
7293/// whatever the child has at descriptor 3 and so must be the last pre-exec
7294/// step, after the Linux cgroup placement that the caller registers later.
7295/// Windows retains the environment handoff until restricted handle inheritance
7296/// can be implemented outside std's process primitives.
7297fn apply_wire_spawn_args_for_role(
7298    command: &mut Command,
7299    spec: &ModuleSpec,
7300    connection_file_path: Option<&std::path::Path>,
7301    handle: Option<&SupervisorHandle>,
7302    role: SpawnRole,
7303) -> Result<Option<NonceHandoff>, SuperviseError> {
7304    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7305    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7306    // included: a daemon started from a module's process tree inherits it,
7307    // and passing it on would point the child at a descriptor it does not
7308    // have.
7309    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7310    // Remove inherited or configured copies too: withholding must mean absent.
7311    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7312    if spec.protocol == ModuleProtocol::None {
7313        return Ok(None);
7314    }
7315    if let Some(connection_file_path) = connection_file_path {
7316        command.arg(SUBC_ARG).arg(connection_file_path);
7317    }
7318
7319    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7320    // route.open attestation. Reserved modules additionally use the same nonce
7321    // for HELLO id-squatting protection. A respawn rotates both records.
7322    let nonce = generate_launch_nonce()?;
7323    if let Some(handle) = handle {
7324        match role {
7325            SpawnRole::Plain => {
7326                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7327                if spec.reserved {
7328                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7329                }
7330            }
7331            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7332        }
7333    }
7334    #[cfg(unix)]
7335    let handoff = {
7336        let handoff =
7337            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7338                program: spec.program.clone(),
7339                source,
7340                cgroup_path: None,
7341            })?;
7342        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7343        Some(handoff)
7344    };
7345    #[cfg(not(unix))]
7346    let handoff = None;
7347    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7348    // handle to this child without leaking it to concurrently spawned processes.
7349    #[cfg(not(unix))]
7350    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7351    Ok(handoff)
7352}
7353
7354fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7355    command.env_remove(CK_LOG_ENV);
7356    // The spawn role is the supervisor's to set, and only on a swap candidate
7357    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7358    // it, is what makes it absent on a plain spawn: the daemon's own
7359    // environment could carry it, and so could a spec built outside daemon
7360    // config (config refuses it as an `env` key). A module reading it on a
7361    // plain restart would pick the long swap budget and leave callers waiting.
7362    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7363    for (key, value) in &spec.env {
7364        // cortexkit-log currently exposes retention only as a Rust struct, not
7365        // environment names. These values are daemon-private sink metadata and
7366        // must never become a public child-process contract by being inherited.
7367        if matches!(
7368            key.as_str(),
7369            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7370        ) || key == SUBC_SPAWN_ROLE_ENV
7371        {
7372            continue;
7373        }
7374        command.env(key, value);
7375    }
7376}
7377
7378/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7379/// of a blue/green swap.
7380#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7381enum SpawnRole {
7382    Plain,
7383    SwapCandidate,
7384}
7385
7386/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7387/// `apply_child_env` has already removed the variable for every spawn.
7388fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7389    if role == SpawnRole::SwapCandidate {
7390        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7391    }
7392}
7393
7394fn spawn_child(
7395    spec: &ModuleSpec,
7396    connection_file_path: Option<&std::path::Path>,
7397    handle: Option<&SupervisorHandle>,
7398    ring: &Arc<Mutex<StderrRing>>,
7399    capture_logs_dir: Option<&std::path::Path>,
7400    roster: &ChildRoster,
7401    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7402) -> Result<SupervisedChild, SuperviseError> {
7403    spawn_child_in_slot(
7404        spec,
7405        connection_file_path,
7406        handle,
7407        ring,
7408        capture_logs_dir,
7409        roster,
7410        #[cfg(target_os = "linux")]
7411        cgroup_placement,
7412        SpawnRole::Plain,
7413        false,
7414    )
7415}
7416
7417/// Spawn one process of `spec` into a slot.
7418///
7419/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7420/// A swap candidate needs a different cgroup from the process it is replacing,
7421/// which is still alive: in the same cgroup the two would be one kill domain,
7422/// and killing a failed candidate could take the incumbent with it.
7423///
7424/// The stderr capture file is `<module_id>.stderr.log` for every process of
7425/// the module, whichever slot it is in, because that is the one file
7426/// `ck module logs` reads. During a swap's overlap both processes append to it;
7427/// the daemon writes whole lines, so the two interleave by line, which is also
7428/// the merged view an operator wants while a swap runs.
7429#[allow(clippy::too_many_arguments)]
7430fn spawn_child_in_slot(
7431    spec: &ModuleSpec,
7432    connection_file_path: Option<&std::path::Path>,
7433    handle: Option<&SupervisorHandle>,
7434    ring: &Arc<Mutex<StderrRing>>,
7435    capture_logs_dir: Option<&std::path::Path>,
7436    roster: &ChildRoster,
7437    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7438    role: SpawnRole,
7439    alternate_slot: bool,
7440) -> Result<SupervisedChild, SuperviseError> {
7441    if roster.is_closed() {
7442        return Err(SuperviseError::Spawn {
7443            program: spec.program.clone(),
7444            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7445            cgroup_path: None,
7446        });
7447    }
7448    #[cfg(target_os = "linux")]
7449    let cgroup_name = {
7450        // Slot names alone are not kill domains: a retired incumbent may still
7451        // be draining when a later enable/restart spawns into the same slot.
7452        // Decimal entropy keeps the suffix unambiguous; Placement performs
7453        // the module-id escaping and constructs the filesystem path.
7454        if cgroup_placement.is_none() {
7455            swap::cgroup_name(&spec.module_id, alternate_slot)
7456        } else {
7457            let nonce = generate_launch_nonce()?;
7458            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7459            // Leave room for byte escaping and the suffix under NAME_MAX. The
7460            // label is only for humans; the nonce identifies the kill domain.
7461            let mut end = spec.module_id.len().min(64);
7462            while !spec.module_id.is_char_boundary(end) {
7463                end -= 1;
7464            }
7465            format!(
7466                "{}_{suffix}",
7467                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7468            )
7469        }
7470    };
7471    #[cfg(not(target_os = "linux"))]
7472    let _ = alternate_slot;
7473    #[cfg(target_os = "macos")]
7474    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7475    #[cfg(not(target_os = "macos"))]
7476    let mut command = Command::new(&spec.program);
7477    command.args(&spec.args);
7478    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7479    // that is the whole of the intent, so remove that one key rather than the
7480    // environment.
7481    //
7482    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7483    // and took the POSIX environment with it. Modules spawned that way had no
7484    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7485    // logging:
7486    //
7487    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7488    //     both unset it fell back to the temp dir alone and `ck` could not find
7489    //     a daemon running on the same machine from inside any module's process
7490    //     tree — reporting a path the file has never lived at, which reads as
7491    //     "the daemon did not write its file".
7492    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7493    //     the RELATIVE `.local/share`, so a module deriving its own store path
7494    //     resolved it against its own CWD. That is the store-fragmentation
7495    //     defect the daemon already refuses in config (`parse_doc` rejects a
7496    //     relative `storage.data_home`) arriving by derivation instead.
7497    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7498    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7499    //     quietly rather than erroring.
7500    //
7501    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7502    // offered one candidate under /tmp while the file sat in /run/user/1000.
7503    //
7504    // A configured module is unaffected either way: `module_spec()` puts the
7505    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7506    // wins over anything ambient.
7507    apply_child_env(&mut command, spec);
7508    apply_spawn_role(&mut command, role);
7509    let nonce_handoff =
7510        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7511
7512    #[cfg(target_os = "linux")]
7513    let cgroup_path = cgroup_placement
7514        .map(|placement| placement.module_path(&cgroup_name))
7515        .transpose()
7516        .map_err(|source| SuperviseError::Cgroup {
7517            module_id: spec.module_id.clone(),
7518            source,
7519        })?;
7520    #[cfg(not(target_os = "linux"))]
7521    let cgroup_path: Option<PathBuf> = None;
7522    #[cfg(target_os = "linux")]
7523    if let Some(path) = &cgroup_path {
7524        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7525            if let Some(placement) = cgroup_placement {
7526                remove_module_cgroup(placement, &cgroup_name);
7527            }
7528            return Err(error);
7529        }
7530    }
7531
7532    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7533        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7534        match ChildOutputSink::open(&path, capture_retention(spec)) {
7535            Ok(sink) => sink,
7536            Err(error) => {
7537                warn!(
7538                    module_id = %spec.module_id,
7539                    path = %path.display(),
7540                    error = %error,
7541                    "could not open child output capture file; forwarding to stderr"
7542                );
7543                ChildOutputSink::Stderr
7544            }
7545        }
7546    } else {
7547        ChildOutputSink::Stderr
7548    };
7549
7550    command.stdout(Stdio::piped());
7551    command.stderr(Stdio::piped());
7552    command.kill_on_drop(true);
7553    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7554    // before exec). In the daemon's group, a service manager that kills the
7555    // job's process group when the daemon exits (launchd's default) killed
7556    // every module at the same moment its control connection closed, so no
7557    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7558    // module is reached only by the daemon: the EOF it sees when its
7559    // connection closes, and the bounded stop in `child_roster` for anything
7560    // still running after that. On Linux this composes with the cgroup
7561    // placement above: that is a pre_exec write to cgroup.procs, std performs
7562    // setpgid in the child before running pre_exec callbacks, and the two
7563    // change independent process attributes.
7564    //
7565    // stdin is /dev/null because a process outside the terminal's foreground
7566    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7567    // by hand would otherwise hand down. Under a service manager stdin is
7568    // already /dev/null.
7569    #[cfg(unix)]
7570    command.process_group(0);
7571    command.stdin(Stdio::null());
7572    // The LAST pre-exec step, after the cgroup placement above: installing the
7573    // nonce at descriptor 3 replaces whatever the child had there, which could
7574    // be the descriptor an earlier step writes through.
7575    #[cfg(unix)]
7576    if let Some(handoff) = nonce_handoff {
7577        handoff.install_last(command.as_std_mut());
7578    }
7579    #[cfg(not(unix))]
7580    let _ = nonce_handoff;
7581
7582    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7583    // cannot run a single instruction -- and therefore cannot spawn a
7584    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7585    // other two steps and why the window matters.
7586    #[cfg(windows)]
7587    subc_jobobject::suspend_on_create_async(&mut command);
7588    let mut child = match command.spawn() {
7589        Ok(child) => child,
7590        Err(source) => {
7591            #[cfg(target_os = "linux")]
7592            if let Some(placement) = cgroup_placement {
7593                remove_module_cgroup(placement, &cgroup_name);
7594            }
7595            return Err(SuperviseError::Spawn {
7596                program: spec.program.clone(),
7597                source,
7598                cgroup_path,
7599            });
7600        }
7601    };
7602    // The parent must close its writer now: the acknowledgement pipe reports EOF
7603    // only when every writer is gone, and the child's copy closes when the
7604    // trampoline replaces itself with the module. Command holds only an integer
7605    // in its pre_exec callback, not another writer.
7606    #[cfg(target_os = "macos")]
7607    drop(exec_ack);
7608
7609    // Containment, steps 2 and 3: assign while suspended, then resume.
7610    #[cfg(windows)]
7611    let job = contain_spawned_child(&child, spec)?;
7612    let spawned_at_ms = unix_ms_now();
7613    let spawned_from = spec.program.clone();
7614    let spawned_file_identity = spawned_file_identity(&spawned_from);
7615    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7616        program: spec.program.clone(),
7617        source: io::Error::other("spawned child exposed no live pid"),
7618        cgroup_path: cgroup_path.clone(),
7619    })?;
7620    let process_start_time = crate::provenance::process_start_time(pid);
7621    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7622    // Unix spawn returns after exec's error pipe closes. The kernel image is
7623    // therefore the executable to compare during a future orphan sweep: PATH
7624    // lookup and shebang interpretation may select a different file from the
7625    // configured program. Keep the literal program's identity for provenance,
7626    // but never use it as proof that a recorded pid may be signalled.
7627    let recorded_image = observe_spawned_image(pid);
7628    // spawn() confirms only the first exec, into the trampoline. Never persist
7629    // the trampoline image; the asynchronous acknowledgement publishes the
7630    // module image once the trampoline has replaced itself with the module.
7631    #[cfg(target_os = "macos")]
7632    let recorded_image = if privacy_exec.is_some() {
7633        None
7634    } else {
7635        recorded_image
7636    };
7637    #[cfg(target_os = "linux")]
7638    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7639    #[cfg(not(target_os = "linux"))]
7640    let recorded_cgroup_name = None;
7641    let roster_guard = roster.admit(
7642        spec.module_id.clone(),
7643        pid,
7644        spec.protocol,
7645        process_start_time,
7646        crate::child_roster::RecordedIdentity {
7647            start_time: recorded_image.map(|image| image.start_time),
7648            executable: recorded_image
7649                .and_then(|image| image.executable)
7650                .map(crate::live_children::ExecutableIdentity::from),
7651            cgroup_name: recorded_cgroup_name,
7652            #[cfg(target_os = "linux")]
7653            cgroup_placement: cgroup_placement.cloned(),
7654        },
7655    );
7656    // The check at the top of this function can pass just before daemon
7657    // shutdown begins, and the process is only in the roster from here on.
7658    // The shutdown stop returns as soon as it finds the roster empty, so a
7659    // process admitted after that look would outlive the daemon. The roster
7660    // is closed before the stop first reads it and admission happens under
7661    // the roster's lock, so either the stop sees this process or this check
7662    // sees the roster closed: end the process now rather than start a module
7663    // the daemon is about to stop.
7664    if roster.is_closed() {
7665        // This child was never admitted, so there is no module protocol shutdown to wait for.
7666        #[cfg(target_os = "linux")]
7667        kill_module_cgroup(cgroup_placement, &cgroup_name);
7668        if let Err(error) = child.start_kill() {
7669            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7670        }
7671        #[cfg(target_os = "linux")]
7672        if let Some(placement) = cgroup_placement {
7673            // This spawn was never admitted, so shutdown has no roster entry
7674            // to await. Do not detach its cleanup: the runtime could exit
7675            // before that task reaps the rejected child and removes its group.
7676            while matches!(child.try_wait(), Ok(None)) {
7677                std::thread::yield_now();
7678            }
7679            if matches!(
7680                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7681                subc_cgroup::KillOutcome::Killed
7682            ) {
7683                if let Ok(path) = placement.module_path(&cgroup_name) {
7684                    while std::fs::read_to_string(path.join("cgroup.events"))
7685                        .ok()
7686                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7687                    {
7688                        std::thread::yield_now();
7689                    }
7690                }
7691            }
7692            remove_module_cgroup(placement, &cgroup_name);
7693        }
7694        drop(roster_guard);
7695        return Err(SuperviseError::Spawn {
7696            program: spec.program.clone(),
7697            source: io::Error::other(
7698                "the daemon began shutting down while this process was starting; ended it",
7699            ),
7700            cgroup_path,
7701        });
7702    }
7703
7704    let stdout_pump = match child.stdout.take() {
7705        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7706        None => {
7707            warn!(
7708                module_id = %spec.module_id,
7709                "spawned child exposed no stdout pipe; file capture will be incomplete"
7710            );
7711            None
7712        }
7713    };
7714    let stderr_pump = match child.stderr.take() {
7715        Some(stderr) => {
7716            let generation = ring
7717                .lock()
7718                .unwrap_or_else(|poisoned| poisoned.into_inner())
7719                .begin_process();
7720            Some(StderrPump {
7721                task: tokio::spawn(pump_stderr_to(
7722                    stderr,
7723                    Arc::clone(ring),
7724                    generation,
7725                    output_sink,
7726                )),
7727                generation,
7728            })
7729        }
7730        None => {
7731            // Spawning succeeded but the pipe did not materialise. Recording it as
7732            // uncaptured keeps the tail honest: the alternative is an empty tail
7733            // that reads as a module which printed nothing.
7734            ring.lock()
7735                .unwrap_or_else(|poisoned| poisoned.into_inner())
7736                .mark_not_captured("stderr pipe was not available on spawn");
7737            warn!(
7738                module_id = %spec.module_id,
7739                "spawned child exposed no stderr pipe; tail will be unavailable"
7740            );
7741            None
7742        }
7743    };
7744
7745    Ok(SupervisedChild {
7746        child,
7747        protocol: spec.protocol,
7748        #[cfg(target_os = "linux")]
7749        module_id: cgroup_name,
7750        #[cfg(target_os = "linux")]
7751        cgroup_placement: cgroup_placement.cloned(),
7752        #[cfg(windows)]
7753        job,
7754        stdout_pump,
7755        stderr_pump,
7756        stderr_ring: Arc::clone(ring),
7757        spawned_at_ms,
7758        spawned_from,
7759        spawned_file_identity,
7760        process_start_time,
7761        process_identity,
7762        pid,
7763        roster_guard: Some(roster_guard),
7764        #[cfg(target_os = "macos")]
7765        privacy_exec,
7766        #[cfg(target_os = "macos")]
7767        report_ready: Arc::new(OnceLock::new()),
7768        spawn_failure: None,
7769    })
7770}
7771
7772#[cfg(target_os = "linux")]
7773pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7774    use subc_cgroup::KillOutcome;
7775    match subc_cgroup::kill_module(placement, module_id) {
7776        KillOutcome::Killed => {}
7777        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7778            debug!(
7779                module_id,
7780                "cgroup tree kill unavailable; using direct-child kill"
7781            );
7782        }
7783        KillOutcome::IoError { path, error } => {
7784            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7785        }
7786    }
7787}
7788
7789/// Contain a freshly spawned Windows child and start it.
7790///
7791/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7792/// child assigned **while it is still suspended** (step 1 is
7793/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7794///
7795/// A child that is never resumed hangs forever holding a pid, so a resume
7796/// failure kills the child and fails the spawn rather than returning a
7797/// `SupervisedChild` that can never run.
7798///
7799/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7800/// it did before this existed, whereas refusing to start one would be a new
7801/// outage. It is logged at warn because it means a helper process could leak.
7802#[cfg(windows)]
7803fn contain_spawned_child(
7804    child: &Child,
7805    spec: &ModuleSpec,
7806) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7807    let module_id = spec.module_id.as_str();
7808    let Some(pid) = child.id() else {
7809        // The child exited between spawn and here. Its tree, if it made one,
7810        // needs no containment: nothing is left to contain.
7811        warn!(
7812            module_id,
7813            "spawned child had already exited before containment; no job object attached"
7814        );
7815        return Ok(None);
7816    };
7817
7818    let job = match subc_jobobject::JobObject::new() {
7819        Ok(job) => job,
7820        Err(source) => {
7821            warn!(
7822                module_id,
7823                error = %source,
7824                "could not create a job object; this module's helper processes will not be \
7825                 reaped on teardown"
7826            );
7827            // Resume regardless: leaving the child suspended would turn a
7828            // containment gap into a hung module.
7829            resume_suspended_child(pid, spec)?;
7830            return Ok(None);
7831        }
7832    };
7833
7834    if let Err(source) = job.assign(child) {
7835        warn!(
7836            module_id,
7837            error = %source,
7838            "could not assign the child to its job object; this module's helper processes \
7839             will not be reaped on teardown"
7840        );
7841        resume_suspended_child(pid, spec)?;
7842        return Ok(None);
7843    }
7844
7845    resume_suspended_child(pid, spec)?;
7846    Ok(Some(job))
7847}
7848
7849/// Resume a suspended child, killing it if it cannot be started.
7850///
7851/// A suspended process holds a pid and does nothing, so there is no useful
7852/// state to return: the caller gets an error and the spawn fails.
7853#[cfg(windows)]
7854fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7855    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7856        // Kill it here rather than leaving a suspended process for the caller
7857        // to notice; `kill_on_drop` would eventually do this, but the module
7858        // would have been reported as running in between.
7859        let _ = std::process::Command::new("taskkill.exe")
7860            .args(["/PID", &pid.to_string(), "/T", "/F"])
7861            .stdin(Stdio::null())
7862            .stdout(Stdio::null())
7863            .stderr(Stdio::null())
7864            .status();
7865        return Err(SuperviseError::Spawn {
7866            program: spec.program.clone(),
7867            source,
7868            cgroup_path: None,
7869        });
7870    }
7871    Ok(())
7872}
7873
7874#[cfg(target_os = "linux")]
7875fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7876    match placement.remove_module(module_id) {
7877        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7878        Err(error) => warn!(
7879            module_id,
7880            error = %error,
7881            "could not remove module cgroup after process exit; continuing teardown"
7882        ),
7883    }
7884}
7885
7886#[cfg(target_os = "linux")]
7887async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7888    // Reaping the direct child is not proof its descendants exited. End the
7889    // residual tree and wait for the kernel's population fact before rmdir;
7890    // otherwise a successful parent wait leaks a directory on each restart.
7891    if matches!(
7892        subc_cgroup::kill_module(Some(placement), module_id),
7893        subc_cgroup::KillOutcome::Killed
7894    ) {
7895        if let Ok(path) = placement.module_path(module_id) {
7896            while std::fs::read_to_string(path.join("cgroup.events"))
7897                .ok()
7898                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7899            {
7900                sleep(Duration::from_millis(1)).await;
7901            }
7902        }
7903    }
7904    remove_module_cgroup(placement, module_id);
7905}
7906
7907#[cfg(target_os = "linux")]
7908fn apply_cgroup_placement(
7909    command: &mut Command,
7910    spec: &ModuleSpec,
7911    path: &std::path::Path,
7912) -> Result<(), SuperviseError> {
7913    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7914        module_id: spec.module_id.clone(),
7915        source,
7916    })
7917}
7918
7919fn capture_retention(spec: &ModuleSpec) -> Retention {
7920    let defaults = Retention::default();
7921    let value = |name: &str| {
7922        spec.env
7923            .iter()
7924            .rev()
7925            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7926    };
7927    Retention {
7928        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7929            .and_then(|value| value.parse().ok())
7930            .unwrap_or(defaults.max_file_mb),
7931        keep: value(CAPTURE_KEEP_ENV)
7932            .and_then(|value| value.parse().ok())
7933            .unwrap_or(defaults.keep),
7934        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7935            .and_then(|value| value.parse().ok())
7936            .unwrap_or(defaults.max_age_days),
7937    }
7938}
7939
7940/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
7941/// module's registration to the exact process the supervisor spawned.
7942fn generate_launch_nonce() -> Result<String, SuperviseError> {
7943    let mut bytes = [0u8; 32];
7944    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7945        reason: source.to_string(),
7946    })?;
7947    let mut hex = String::with_capacity(64);
7948    for b in bytes {
7949        use std::fmt::Write;
7950        let _ = write!(hex, "{b:02x}");
7951    }
7952    Ok(hex)
7953}
7954
7955/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
7956/// signal about how many leading bytes matched.
7957fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7958    if a.len() != b.len() {
7959        return false;
7960    }
7961    let mut diff = 0u8;
7962    for (x, y) in a.iter().zip(b.iter()) {
7963        diff |= x ^ y;
7964    }
7965    diff == 0
7966}
7967
7968/// The kernel's image after an acknowledged exec, shared by ordinary launches
7969/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
7970/// not identities inferred from a configured pathname.
7971fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
7972    subc_os::Process::open(pid)
7973        .ok()
7974        .flatten()
7975        .and_then(|process| process.observe())
7976}
7977
7978fn spawn_and_mark_running(
7979    spec: &ModuleSpec,
7980    runtime: &SupervisorRuntimeConfig,
7981    snapshot: &SharedSnapshot,
7982) -> Result<SupervisedChild, SuperviseError> {
7983    let child = spawn_child(
7984        spec,
7985        runtime.connection_file_path.as_deref(),
7986        runtime.supervisor_handle.as_ref(),
7987        &runtime.stderr_ring,
7988        runtime.capture_logs_dir.as_deref(),
7989        &runtime.child_roster,
7990        #[cfg(target_os = "linux")]
7991        runtime.cgroup_placement.as_ref(),
7992    )?;
7993    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
7994    Ok(child)
7995}
7996
7997enum RegistrationWaitOutcome {
7998    Registered,
7999    Exited(ExitReport),
8000    TimedOut,
8001}
8002
8003struct ReloadRegistrationFailure {
8004    exit_report: ExitReport,
8005    reason: String,
8006}
8007
8008#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8009enum BusyGaugeObservation {
8010    Quiescent,
8011    Busy,
8012    Omitted,
8013}
8014
8015fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8016    let Some(metrics) = metrics.and_then(Value::as_object) else {
8017        return BusyGaugeObservation::Omitted;
8018    };
8019    let mut sum = 0u128;
8020    for gauge in gauges {
8021        let Some(value) = metrics.get(gauge) else {
8022            return BusyGaugeObservation::Omitted;
8023        };
8024        let Some(value) = value.as_u64() else {
8025            return BusyGaugeObservation::Busy;
8026        };
8027        sum = sum.saturating_add(u128::from(value));
8028    }
8029    if sum == 0 {
8030        BusyGaugeObservation::Quiescent
8031    } else {
8032        BusyGaugeObservation::Busy
8033    }
8034}
8035
8036fn declared_busy_gauges(
8037    registry: &Registry,
8038    module_id: &str,
8039) -> Result<Vec<String>, SuperviseError> {
8040    busy_gauges_of(
8041        registry
8042            .get_module(module_id)
8043            .map_err(SuperviseError::Registry)?,
8044    )
8045}
8046
8047/// [`declared_busy_gauges`] for the registration a connection holds, in any
8048/// slot: after cutover the incumbent is no longer the id's active
8049/// registration, and its own manifest is the one that names its gauges.
8050fn declared_busy_gauges_for_connection(
8051    registry: &Registry,
8052    connection_id: ConnectionId,
8053) -> Result<Vec<String>, SuperviseError> {
8054    busy_gauges_of(
8055        registry
8056            .get_module_by_connection(connection_id)
8057            .map_err(SuperviseError::Registry)?,
8058    )
8059}
8060
8061fn busy_gauges_of(
8062    registration: Option<crate::registry::ModuleRegistration>,
8063) -> Result<Vec<String>, SuperviseError> {
8064    let Some(registration) = registration else {
8065        return Ok(Vec::new());
8066    };
8067    let Some(self_signals) = registration.manifest.self_signals else {
8068        return Ok(Vec::new());
8069    };
8070
8071    let mut gauges = Vec::new();
8072    for declaration in self_signals {
8073        if declaration.kind != SelfSignalKind::Busy {
8074            continue;
8075        }
8076        match declaration.anchored_to {
8077            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8078                gauges.extend(declared)
8079            }
8080            _ => {
8081                // An invalid Busy anchor is fail-safe: the empty name cannot be
8082                // present in a conforming health report, so this drain stays busy.
8083                gauges.push(String::new());
8084            }
8085        }
8086    }
8087    Ok(gauges)
8088}
8089
8090/// Wait for `endpoint` to have nothing in flight and, when the module declares
8091/// busy gauges, for a health probe to report them quiet. The probe is addressed
8092/// by `scope`: a swap's superseded incumbent must be asked about its own
8093/// gauges, and by module id the probe would reach the promoted candidate.
8094async fn wait_for_forwarding_quiescence(
8095    forwarding: &ForwardingTable,
8096    module_id: &str,
8097    runtime: &SupervisorRuntimeConfig,
8098    endpoint: crate::ModuleEndpointId,
8099    deadline: Instant,
8100    busy_gauges: &[String],
8101    scope: DrainScope,
8102) -> Result<bool, SuperviseError> {
8103    let mut gauges_quiescent = busy_gauges.is_empty();
8104    let mut next_probe_at = Instant::now();
8105    let mut omission_counted = false;
8106
8107    loop {
8108        let now = Instant::now();
8109        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8110            let report = match scope {
8111                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8112                DrainScope::Endpoint(endpoint) => {
8113                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8114                }
8115            };
8116            gauges_quiescent = match report {
8117                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8118                    BusyGaugeObservation::Quiescent => true,
8119                    BusyGaugeObservation::Busy => false,
8120                    BusyGaugeObservation::Omitted => {
8121                        if !omission_counted {
8122                            forwarding
8123                                .counters()
8124                                .increment_drains_with_undeclared_gauge();
8125                            omission_counted = true;
8126                        }
8127                        false
8128                    }
8129                },
8130                Err(err) => {
8131                    warn!(
8132                        module_id,
8133                        error = %err,
8134                        "drain health.check did not produce declared busy gauges; treating module as busy"
8135                    );
8136                    false
8137                }
8138            };
8139            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8140        }
8141
8142        let in_flight = forwarding
8143            .endpoint_in_flight_count(endpoint)
8144            .map_err(SuperviseError::Forwarding)?;
8145        if in_flight == 0 && gauges_quiescent {
8146            return Ok(true);
8147        }
8148
8149        let now = Instant::now();
8150        if now >= deadline {
8151            return Ok(false);
8152        }
8153        let mut wait = deadline
8154            .saturating_duration_since(now)
8155            .min(REGISTRY_RELEASE_POLL);
8156        if !busy_gauges.is_empty() {
8157            wait = wait.min(next_probe_at.saturating_duration_since(now));
8158        }
8159        sleep(wait).await;
8160    }
8161}
8162
8163/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8164///
8165/// `Ok` is always honest and passed straight through -- the wait actually measured
8166/// in-flight state. `Err` means the wait produced no measurement at all (the
8167/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8168/// constant: the drain did not complete. Never recomputed from route state, never a
8169/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8170fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8171    match wait_result {
8172        Ok(drained) => *drained,
8173        Err(_) => false,
8174    }
8175}
8176
8177fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8178    for released in released_routes {
8179        let frame = match Frame::build_with_version(
8180            released.negotiated_ver,
8181            FrameType::Goodbye,
8182            control_flags(),
8183            released.channel,
8184            released.epoch,
8185            0,
8186            Vec::new(),
8187        ) {
8188            Ok(frame) => frame,
8189            Err(err) => {
8190                warn!(
8191                    route_channel = released.channel,
8192                    error = %err,
8193                    "failed to build supervisor drain route GOODBYE frame"
8194                );
8195                continue;
8196            }
8197        };
8198        if !released.close_on_delivery_failure() {
8199            crate::forwarding::send_module_route_goodbye(
8200                &forwarding.counters(),
8201                &released.sink,
8202                frame,
8203                released.module_id.as_deref(),
8204                "supervisor drain",
8205            );
8206            continue;
8207        }
8208        if let Err(err) = released.sink.try_send(frame) {
8209            warn!(
8210                target_connection_id = released.connection_id.get(),
8211                route_channel = released.channel,
8212                error = %err,
8213                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8214            );
8215            let _ = forwarding.escalate_client_delivery_failure(
8216                released.connection_id,
8217                released.channel,
8218                released.epoch,
8219                CloseReason::new(
8220                    "route_goodbye_delivery_failed",
8221                    format!(
8222                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8223                        released.channel
8224                    ),
8225                ),
8226                crate::forwarding::UndeliveredFrame {
8227                    module_id: released.module_id.as_deref(),
8228                    sink: &released.sink,
8229                },
8230            );
8231        }
8232    }
8233}
8234
8235fn send_module_draining(
8236    module_id: &str,
8237    reason: RouteCloseReason,
8238    deadline_ms: u64,
8239    target: &ModuleDrainTarget,
8240) {
8241    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8242        reason,
8243        deadline_ms,
8244    }) {
8245        Ok(body) => body,
8246        Err(err) => {
8247            warn!(
8248                module_id,
8249                error = %err,
8250                "failed to encode module draining command"
8251            );
8252            return;
8253        }
8254    };
8255    let frame = match Frame::build_with_version(
8256        target.negotiated_ver,
8257        FrameType::Push,
8258        control_flags(),
8259        0,
8260        0,
8261        0,
8262        body,
8263    ) {
8264        Ok(frame) => frame,
8265        Err(err) => {
8266            warn!(
8267                module_id,
8268                error = %err,
8269                "failed to build module draining command frame"
8270            );
8271            return;
8272        }
8273    };
8274    if let Err(err) = target.sink.try_send(frame) {
8275        warn!(
8276            module_id,
8277            target_connection_id = target.endpoint.connection_id.get(),
8278            error = %err,
8279            "module draining command was not delivered to peer"
8280        );
8281    }
8282}
8283
8284/// The channel-0 GOODBYE that tells a module its stop is planned.
8285fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8286    match Frame::build_with_version(
8287        negotiated_ver,
8288        FrameType::Goodbye,
8289        control_flags(),
8290        0,
8291        0,
8292        0,
8293        Vec::new(),
8294    ) {
8295        Ok(frame) => Some(frame),
8296        Err(err) => {
8297            warn!(
8298                module_id,
8299                error = %err,
8300                "failed to build module GOODBYE frame"
8301            );
8302            None
8303        }
8304    }
8305}
8306
8307/// Send every registered module connection its module GOODBYE at daemon
8308/// shutdown, then request that connection's close.
8309///
8310/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8311/// before EOF, so the GOODBYE must reach the socket before the close. A close
8312/// request does not wait for the connection's queued frames: its writer gets a
8313/// bounded grace after the close, is aborted if it overruns it, and the daemon
8314/// process may exit before that grace ends. So with `wait_for_flush`, each
8315/// connection is closed only after its writer has acknowledged writing the
8316/// GOODBYE, or once a short shared budget runs out, so one module that is not
8317/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8318/// are only queued, for a shutdown the operator has told to stop waiting.
8319/// A connection that is already gone is skipped.
8320#[cfg(unix)]
8321async fn send_module_goodbyes_for_daemon_shutdown(
8322    forwarding: &Arc<ForwardingTable>,
8323    reason: &CloseReason,
8324    wait_for_flush: bool,
8325) {
8326    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8327    let targets = match forwarding.module_connections() {
8328        Ok(targets) => targets,
8329        Err(err) => {
8330            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8331            return;
8332        }
8333    };
8334    let deadline = Instant::now() + GOODBYE_BUDGET;
8335    let mut sends = tokio::task::JoinSet::new();
8336    for target in targets {
8337        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8338            continue;
8339        };
8340        if !wait_for_flush {
8341            if let Err(err) = target.sink.try_send(frame) {
8342                debug!(
8343                    module_id = %target.module_id,
8344                    error = %err,
8345                    "shutdown module GOODBYE was not queued"
8346                );
8347            }
8348            continue;
8349        }
8350        let forwarding = Arc::clone(forwarding);
8351        let reason = reason.clone();
8352        sends.spawn(async move {
8353            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8354                Ok(Ok(())) => {}
8355                Ok(Err(err)) => debug!(
8356                    module_id = %target.module_id,
8357                    error = %err,
8358                    "module connection closed before its shutdown GOODBYE was written"
8359                ),
8360                Err(_) => warn!(
8361                    module_id = %target.module_id,
8362                    budget = ?GOODBYE_BUDGET,
8363                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8364                ),
8365            }
8366            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8367        });
8368    }
8369    // Every task ends by the shared deadline, so this wait is bounded too.
8370    while sends.join_next().await.is_some() {}
8371}
8372
8373fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8374    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8375        return;
8376    };
8377    if let Err(err) = target.sink.try_send(frame) {
8378        warn!(
8379            module_id,
8380            target_connection_id = target.endpoint.connection_id.get(),
8381            error = %err,
8382            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8383        );
8384        forwarding.request_connection_close(
8385            target.endpoint.connection_id,
8386            CloseReason::new(
8387                "module_goodbye_delivery_failed",
8388                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8389            ),
8390        );
8391    }
8392}
8393
8394#[derive(Clone, Copy)]
8395struct ForwardingDrainContext<'a> {
8396    spec: &'a ModuleSpec,
8397    runtime: &'a SupervisorRuntimeConfig,
8398    registry: &'a Registry,
8399    scope: DrainScope,
8400}
8401
8402/// Which process a forwarding drain addresses.
8403#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8404enum DrainScope {
8405    /// Whatever endpoint is active for the module id: every plain stop,
8406    /// restart and reload. Also moves the module's state to `Draining`.
8407    Active,
8408    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8409    /// module id would resolve to the promoted candidate and leave neither
8410    /// process routable. The module's state is left alone, since the promoted
8411    /// candidate is what it describes and that process is running.
8412    Endpoint(crate::ModuleEndpointId),
8413}
8414
8415/// Whether a child being drained has already been asked to stop by the time
8416/// its drain wait starts.
8417///
8418/// The drain wait is the same budget whatever this says. What it decides is
8419/// whether the supervisor must ask by signal before that wait begins: a child
8420/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8421/// healthy or not.
8422#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8423enum StopNotice {
8424    /// The module was sent `module.draining` and a module GOODBYE over its own
8425    /// registered connection, and stops itself.
8426    SentOverConnection,
8427    /// The forwarding drain found no registered connection for the module: a
8428    /// subc child spawned moments ago that has not sent HELLO yet, or a
8429    /// `protocol: "none"` child, which never registers.
8430    NoConnection,
8431    /// This path sends nothing over the module's connection: the supervisor has
8432    /// no forwarding table, or the caller stops the child without a forwarding
8433    /// drain.
8434    NotSent,
8435}
8436
8437async fn begin_forwarding_drain(
8438    spec: &ModuleSpec,
8439    runtime: &SupervisorRuntimeConfig,
8440    registry: &Registry,
8441    snapshot: &SharedSnapshot,
8442    enabled: Option<bool>,
8443    reason: RouteCloseReason,
8444) -> Result<StopNotice, SuperviseError> {
8445    let Some(forwarding) = runtime.forwarding.as_ref() else {
8446        return Err(SuperviseError::ReloadUnavailable {
8447            module_id: spec.module_id.clone(),
8448            reason: "supervisor was not configured with a forwarding table".to_string(),
8449        });
8450    };
8451
8452    begin_forwarding_drain_with(
8453        forwarding,
8454        ForwardingDrainContext {
8455            spec,
8456            runtime,
8457            registry,
8458            scope: DrainScope::Active,
8459        },
8460        snapshot,
8461        enabled,
8462        reason,
8463        runtime.drain_timeout,
8464    )
8465    .await
8466}
8467
8468async fn begin_forwarding_drain_if_configured(
8469    spec: &ModuleSpec,
8470    runtime: &SupervisorRuntimeConfig,
8471    registry: &Registry,
8472    snapshot: &SharedSnapshot,
8473    enabled: Option<bool>,
8474    reason: RouteCloseReason,
8475) -> Result<StopNotice, SuperviseError> {
8476    begin_forwarding_drain_with_timeout(
8477        spec,
8478        runtime,
8479        registry,
8480        snapshot,
8481        enabled,
8482        reason,
8483        runtime.drain_timeout,
8484    )
8485    .await
8486}
8487
8488/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8489/// budget, for paths where the operator overrides the module's configured one
8490/// (`supervisor.restart{drain_timeout_ms}`).
8491async fn begin_forwarding_drain_with_timeout(
8492    spec: &ModuleSpec,
8493    runtime: &SupervisorRuntimeConfig,
8494    registry: &Registry,
8495    snapshot: &SharedSnapshot,
8496    enabled: Option<bool>,
8497    reason: RouteCloseReason,
8498    drain_timeout: Duration,
8499) -> Result<StopNotice, SuperviseError> {
8500    let Some(forwarding) = runtime.forwarding.as_ref() else {
8501        return Ok(StopNotice::NotSent);
8502    };
8503
8504    begin_forwarding_drain_with(
8505        forwarding,
8506        ForwardingDrainContext {
8507            spec,
8508            runtime,
8509            registry,
8510            scope: DrainScope::Active,
8511        },
8512        snapshot,
8513        enabled,
8514        reason,
8515        drain_timeout,
8516    )
8517    .await
8518}
8519
8520async fn begin_forwarding_drain_with(
8521    forwarding: &ForwardingTable,
8522    context: ForwardingDrainContext<'_>,
8523    snapshot: &SharedSnapshot,
8524    enabled: Option<bool>,
8525    reason: RouteCloseReason,
8526    drain_timeout: Duration,
8527) -> Result<StopNotice, SuperviseError> {
8528    let ForwardingDrainContext {
8529        spec,
8530        runtime,
8531        registry,
8532        scope,
8533    } = context;
8534    debug_assert_ne!(reason, RouteCloseReason::Crash);
8535    let terminal = matches!(reason, RouteCloseReason::Disable);
8536    let drain_started_at = Instant::now();
8537    let drain_deadline = drain_started_at + drain_timeout;
8538    let deadline_ms =
8539        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8540    let busy_gauges = match scope {
8541        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8542        DrainScope::Endpoint(endpoint) => {
8543            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8544        }
8545    };
8546
8547    // Admission gate first: route.open/commit and route REQUEST admission are closed
8548    // before the first quiescence check, so the outstanding count can only fall.
8549    let gate_started = Instant::now();
8550    let drain_target = match scope {
8551        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8552        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8553    }
8554    .map_err(SuperviseError::Forwarding)?;
8555    // The instant admission closed, and how long taking the forwarding write
8556    // lock to close it took. The timeout line reports only the quiescence
8557    // wait, so without this a drain that started late looked like one that
8558    // started on time.
8559    info!(
8560        module_id = %spec.module_id,
8561        ?reason,
8562        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8563        connected = drain_target.is_some(),
8564        "module drain began; route admission closed"
8565    );
8566    if scope == DrainScope::Active {
8567        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8568            state.state = ModuleState::Draining;
8569            state.draining_to_replace =
8570                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8571            if let Some(enabled) = enabled {
8572                state.enabled = enabled;
8573            }
8574        })?;
8575    }
8576
8577    let Some(target) = drain_target.as_ref() else {
8578        // Nothing was sent: the module has no registered connection to carry
8579        // `module.draining` or a GOODBYE. The caller must not assume the child
8580        // was asked to stop.
8581        return Ok(StopNotice::NoConnection);
8582    };
8583    {
8584        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8585        let routes = forwarding
8586            .endpoint_routes(target.endpoint)
8587            .map_err(SuperviseError::Forwarding)?;
8588        let routes_notified = routes.len();
8589        crate::control::send_route_control_pushes(
8590            forwarding,
8591            routes.clone(),
8592            ClientControlPush::RouteClosing {
8593                module_id: spec.module_id.clone(),
8594                channels: Vec::new(),
8595                reason,
8596            },
8597        );
8598        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8599
8600        // `route.closing` was just sent above: from here on every return path,
8601        // including an early one, MUST send `route.closed` before propagating
8602        // anything else. A client holds `closing` as a promise that a verdict is
8603        // coming; leaving early without `closed` strands it waiting forever, since
8604        // `closing` carries no timeout of its own.
8605        let wait_result = wait_for_forwarding_quiescence(
8606            forwarding,
8607            &spec.module_id,
8608            runtime,
8609            target.endpoint,
8610            drain_deadline,
8611            &busy_gauges,
8612            scope,
8613        )
8614        .await;
8615        let drained = drained_after_quiescence_wait(&wait_result);
8616        if let Err(err) = &wait_result {
8617            error!(
8618                module_id = %spec.module_id,
8619                ?reason,
8620                error = %err,
8621                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8622            );
8623        } else if !drained {
8624            // Name what the drain waited on. Without it the line says only that
8625            // something did not settle, and "one wedged call" and "every
8626            // session's held stream" read the same; the first is a module bug,
8627            // the second is a module that should end its streams on
8628            // module.draining. Read before teardown releases the routes.
8629            let holdouts = forwarding
8630                .endpoint_drain_holdouts(target.endpoint)
8631                .unwrap_or_default();
8632            warn!(
8633                module_id = %spec.module_id,
8634                waited = ?drain_timeout,
8635                ?reason,
8636                held_requests = holdouts.requests,
8637                held_routes = holdouts.routes,
8638                total_routes = holdouts.total_routes,
8639                top_connections = ?holdouts.top_connections,
8640                // `module_channel:corr`, so the module can find each held request
8641                // in its own log; capped, so `held_requests` is the full count.
8642                held = %holdouts
8643                    .held
8644                    .iter()
8645                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8646                    .collect::<Vec<_>>()
8647                    .join(","),
8648                "route drain timed out before request quiescence; forcing teardown"
8649            );
8650        }
8651        crate::control::send_route_control_pushes(
8652            forwarding,
8653            routes,
8654            ClientControlPush::RouteClosed {
8655                module_id: spec.module_id.clone(),
8656                channels: Vec::new(),
8657                reason,
8658                drained,
8659                abandoned: target.abandoned_bindings.len() as u32,
8660                excluded_subscriptions: target.excluded_subscriptions,
8661                terminal: Some(terminal),
8662            },
8663        );
8664        wait_result?;
8665
8666        // `route.closed` has now been sent unconditionally above. From here the
8667        // remaining steps are cleanup (route + module GOODBYE) rather than a
8668        // promise the client is waiting on, but a lock-poisoned
8669        // `release_module_endpoint_routes` would otherwise skip the module
8670        // GOODBYE silently too -- send it before propagating the error.
8671        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8672            Ok(routes) => routes,
8673            Err(err) => {
8674                warn!(
8675                    module_id = %spec.module_id,
8676                    ?reason,
8677                    error = %err,
8678                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8679                );
8680                send_module_goodbye(&spec.module_id, forwarding, target);
8681                return Err(SuperviseError::Forwarding(err));
8682            }
8683        };
8684        let route_goodbye_count = released_routes.len();
8685        send_route_goodbyes(forwarding, released_routes);
8686        send_module_goodbye(&spec.module_id, forwarding, target);
8687
8688        // The drain's happy path was previously silent: every emission above is
8689        // best-effort with only its failure arm logged, so "were consumers told"
8690        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8691        // hang where the open question was exactly whether teardown notice went
8692        // out). One summary line makes that class decidable in one grep.
8693        info!(
8694            module_id = %spec.module_id,
8695            ?reason,
8696            routes_notified,
8697            route_goodbyes = route_goodbye_count,
8698            abandoned_reservations = target.abandoned_bindings.len(),
8699            excluded_subscriptions = target.excluded_subscriptions,
8700            drained,
8701            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8702        );
8703    }
8704
8705    Ok(StopNotice::SentOverConnection)
8706}
8707
8708/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8709/// the only slot a plain (non-swap) spawn can register into.
8710async fn wait_for_registration_after_reload(
8711    registry: &Registry,
8712    module_id: &str,
8713    snapshot: &SharedSnapshot,
8714    child: &mut SupervisedChild,
8715    wait: Duration,
8716) -> Result<RegistrationWaitOutcome, SuperviseError> {
8717    wait_for_slot_registration(
8718        registry,
8719        crate::registry::RegistrationSlot::Active(module_id),
8720        module_id,
8721        snapshot,
8722        child,
8723        wait,
8724    )
8725    .await
8726}
8727
8728/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8729///
8730/// Keyed on the slot rather than the bare module id because during a swap the
8731/// id's active slot is already held by the incumbent: an id-keyed wait would
8732/// report the incumbent's registration as the candidate's and a candidate that
8733/// never registers would look registered. A swap candidate waits on
8734/// `crate::registry::RegistrationSlot::Candidate`.
8735async fn wait_for_slot_registration(
8736    registry: &Registry,
8737    slot: crate::registry::RegistrationSlot<'_>,
8738    module_id: &str,
8739    snapshot: &SharedSnapshot,
8740    child: &mut SupervisedChild,
8741    wait: Duration,
8742) -> Result<RegistrationWaitOutcome, SuperviseError> {
8743    let deadline = Instant::now() + wait;
8744    loop {
8745        if registry
8746            .registration(slot)
8747            .map_err(SuperviseError::Registry)?
8748            .is_some()
8749        {
8750            return Ok(RegistrationWaitOutcome::Registered);
8751        }
8752
8753        let now = Instant::now();
8754        if now >= deadline {
8755            return Ok(RegistrationWaitOutcome::TimedOut);
8756        }
8757        let remaining = deadline.saturating_duration_since(now);
8758        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8759
8760        tokio::select! {
8761            wait_result = child.wait() => {
8762                let status = wait_result.map_err(|source| SuperviseError::Wait {
8763                    module_id: module_id.to_string(),
8764                    source,
8765                })?;
8766                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8767                    snapshot,
8768                    child,
8769                    &status,
8770                )));
8771            }
8772            _ = sleep(poll) => {}
8773        }
8774    }
8775}
8776
8777fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8778    // A replacement process that exits before HELLO did not provide service, even
8779    // if it used status 0. Count it against the restart cap as a new-binary failure.
8780    if exit_report.kind != ExitKind::DeliberateSeverance {
8781        exit_report.kind = ExitKind::Crash;
8782    }
8783    exit_report
8784}
8785
8786async fn handle_reload_child_registration_failure(
8787    spec: &ModuleSpec,
8788    runtime: &SupervisorRuntimeConfig,
8789    registry: &Registry,
8790    process_liveness: &SupervisorProcessLiveness,
8791    snapshot: &SharedSnapshot,
8792    _child: &mut Option<SupervisedChild>,
8793    failure: ReloadRegistrationFailure,
8794) -> Result<(), SuperviseError> {
8795    let ReloadRegistrationFailure {
8796        exit_report,
8797        reason,
8798    } = failure;
8799    match on_child_exit(
8800        spec,
8801        runtime.restart_policy,
8802        registry,
8803        snapshot,
8804        &runtime.terminal_ring,
8805        &runtime.spawn_events,
8806        &runtime.child_roster,
8807        exit_report,
8808    )
8809    .await
8810    {
8811        NextAction::Stop {
8812            registration_released,
8813        } => {
8814            if registration_released {
8815                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8816            }
8817        }
8818        NextAction::Restart { schedule } => {
8819            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8820                schedule.delay
8821            });
8822            if let Some(schedule) = schedule {
8823                log_crash_respawn(&spec.module_id, schedule);
8824            }
8825            schedule_respawn(
8826                runtime,
8827                snapshot,
8828                &spec.module_id,
8829                delay,
8830                RespawnKind::Spawn,
8831            )?;
8832        }
8833    }
8834    Err(SuperviseError::ReloadFailed {
8835        module_id: spec.module_id.clone(),
8836        reason,
8837    })
8838}
8839
8840async fn handle_reload_spawn_failure(
8841    spec: &ModuleSpec,
8842    runtime: &SupervisorRuntimeConfig,
8843    process_liveness: &SupervisorProcessLiveness,
8844    snapshot: &SharedSnapshot,
8845    _child: &mut Option<SupervisedChild>,
8846    reason: String,
8847) -> Result<(), SuperviseError> {
8848    let now = Instant::now();
8849    let mut schedule = None;
8850    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8851        clear_current_process_facts(state);
8852        if state.enabled {
8853            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8854            state.state = if schedule.is_some() {
8855                ModuleState::Restarting
8856            } else {
8857                ModuleState::Failed
8858            };
8859        } else {
8860            state.state = ModuleState::Disabled;
8861        }
8862    })?;
8863    if let Some(schedule) = schedule {
8864        schedule_respawn(
8865            runtime,
8866            snapshot,
8867            &spec.module_id,
8868            schedule.delay,
8869            RespawnKind::Spawn,
8870        )?;
8871    } else {
8872        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8873    }
8874    Err(SuperviseError::ReloadFailed {
8875        module_id: spec.module_id.clone(),
8876        reason,
8877    })
8878}
8879
8880fn control_flags() -> Flags {
8881    Flags::new(false, Priority::Passive, false)
8882}
8883
8884#[allow(clippy::too_many_arguments)]
8885async fn drain_optional_child(
8886    module_id: &str,
8887    protocol: ModuleProtocol,
8888    stop_notice: StopNotice,
8889    registry: &Registry,
8890    forwarding: Option<&ForwardingTable>,
8891    snapshot: &SharedSnapshot,
8892    terminal_ring: &Arc<Mutex<TerminalRing>>,
8893    spawn_events: &SpawnEventFeed,
8894    child: &mut Option<SupervisedChild>,
8895    drain_timeout: Duration,
8896    final_state: ModuleState,
8897    enabled: Option<bool>,
8898) -> Result<(), SuperviseError> {
8899    if let Some(child) = child.take() {
8900        drain_child_to_state(
8901            module_id,
8902            protocol,
8903            stop_notice,
8904            registry,
8905            forwarding,
8906            snapshot,
8907            terminal_ring,
8908            spawn_events,
8909            child,
8910            drain_timeout,
8911            final_state,
8912            enabled,
8913        )
8914        .await
8915    } else {
8916        update_snapshot(snapshot, Some(module_id), |state| {
8917            state.state = final_state;
8918            if let Some(enabled) = enabled {
8919                state.enabled = enabled;
8920            }
8921            clear_current_process_facts(state);
8922        })?;
8923        release_dead_registration(registry, forwarding, snapshot, module_id).await
8924    }
8925}
8926
8927#[allow(clippy::too_many_arguments)]
8928async fn drain_child_to_state(
8929    module_id: &str,
8930    _protocol: ModuleProtocol,
8931    stop_notice: StopNotice,
8932    registry: &Registry,
8933    forwarding: Option<&ForwardingTable>,
8934    snapshot: &SharedSnapshot,
8935    terminal_ring: &Arc<Mutex<TerminalRing>>,
8936    spawn_events: &SpawnEventFeed,
8937    mut child: SupervisedChild,
8938    drain_timeout: Duration,
8939    final_state: ModuleState,
8940    enabled: Option<bool>,
8941) -> Result<(), SuperviseError> {
8942    let protocol = child.protocol;
8943    update_snapshot(snapshot, Some(module_id), |state| {
8944        state.state = ModuleState::Draining;
8945        state.draining_to_replace = final_state == ModuleState::Restarting;
8946        if let Some(enabled) = enabled {
8947            state.enabled = enabled;
8948        }
8949    })?;
8950
8951    // The wait below is the same budget in every case; what differs is
8952    // whether anything has ASKED the child to stop before it starts. Only a
8953    // forwarding drain that reached the module's registered connection has
8954    // (`module.draining`, then a module GOODBYE). Every other child was told
8955    // nothing: a `protocol: "none"` module, which never registers; a subc
8956    // module spawned moments ago that has not sent HELLO yet; or a stop that
8957    // runs no forwarding drain. Without a signal the budget is only a delay
8958    // in front of SIGKILL -- and the not-yet-registered child is the worst
8959    // case, because it registers into a module that is already draining,
8960    // is never told, and is killed while healthy.
8961    if stop_notice != StopNotice::SentOverConnection {
8962        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8963            info!(
8964                module_id,
8965                pid = child.pid,
8966                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8967                "module has no connection yet; requesting stop by signal"
8968            );
8969        }
8970        request_graceful_stop(module_id, &child);
8971    }
8972
8973    let exit_report = match timeout(drain_timeout, child.wait()).await {
8974        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
8975        Ok(Err(source)) => {
8976            fail_snapshot(snapshot, Some(module_id), None);
8977            return Err(SuperviseError::Wait {
8978                module_id: module_id.to_string(),
8979                source,
8980            });
8981        }
8982        Err(_) => {
8983            // Mirror the sibling arm above: state is already `Draining`, and an
8984            // error propagated from here would strand it there -- a state
8985            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
8986            // `Failed | Stopped`), leaving an operator Restart as the only exit.
8987            // `Failed` before `?` keeps the module operator-visible and
8988            // revivable. Trigger is an ESRCH race (process exits between the
8989            // drain timeout firing and the kill) or a post-kill wait failure
8990            // (issue #34).
8991            //
8992            // Logged because the kill is otherwise visible only as signal 9 in
8993            // the terminal ring, and the budget it follows can be long enough
8994            // that consumers see a stretch of refusals with no stated cause.
8995            warn!(
8996                module_id,
8997                pid = child.pid,
8998                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8999                reason = ?final_state,
9000                ?stop_notice,
9001                "drain budget expired before the module exited; killing it"
9002            );
9003            child.start_kill().map_err(|source| {
9004                fail_snapshot(snapshot, Some(module_id), None);
9005                SuperviseError::Kill {
9006                    module_id: module_id.to_string(),
9007                    source,
9008                }
9009            })?;
9010            let status = child.wait().await.map_err(|source| {
9011                fail_snapshot(snapshot, Some(module_id), None);
9012                SuperviseError::Wait {
9013                    module_id: module_id.to_string(),
9014                    source,
9015                }
9016            })?;
9017            classify_reaped_child_exit(snapshot, &child, &status)
9018        }
9019    };
9020
9021    update_snapshot(snapshot, Some(module_id), |state| {
9022        state.state = final_state;
9023        if let Some(enabled) = enabled {
9024            state.enabled = enabled;
9025        }
9026        clear_current_process_facts(state);
9027        state.last_exit = Some(exit_report.clone());
9028        if exit_report.kind == ExitKind::DeliberateSeverance {
9029            state.lifetime_restarts += 1;
9030        }
9031    })?;
9032    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9033    record_terminal_with_detail(
9034        module_id,
9035        terminal_ring,
9036        spawn_events,
9037        &exit_report,
9038        terminal_disposition(final_state),
9039        detail,
9040    );
9041    child.drain_stderr(module_id).await;
9042
9043    release_dead_registration(registry, forwarding, snapshot, module_id).await
9044}
9045
9046/// Ask a child that nothing else has asked to stop, by signal.
9047///
9048/// A registered subc module is asked over its own connection: the drain sends
9049/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
9050/// module GOODBYE, and the module stops itself. A module that speaks no subc
9051/// wire receives none of that, and neither does a subc module that has not
9052/// registered yet, so for them the drain budget would be pure delay in front of
9053/// a SIGKILL -- and for a process with a store to flush (JetStream is the
9054/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
9055/// into a recovery on the next start.
9056///
9057/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
9058/// rule rather than an optimisation: that module's graceful stop is already
9059/// running by the time its child is drained, and a signal would race it.
9060///
9061/// Best-effort by construction. A child that has already exited is the ordinary
9062/// case rather than an error (the kill lands on a reaped or exiting pid), so a
9063/// failure is logged at debug and the wait-then-kill below still decides the
9064/// outcome.
9065#[cfg(unix)]
9066fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9067    let Some(pid) = child
9068        .id()
9069        .and_then(|pid| i32::try_from(pid).ok())
9070        .and_then(rustix::process::Pid::from_raw)
9071    else {
9072        debug!(
9073            module_id,
9074            "no pid to signal for teardown; falling through to the drain wait"
9075        );
9076        return;
9077    };
9078    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9079        Ok(()) => debug!(
9080            module_id,
9081            "sent SIGTERM to a module nothing else asked to stop"
9082        ),
9083        Err(err) => debug!(
9084            module_id,
9085            error = %err,
9086            "SIGTERM to module failed; the drain wait and kill still apply"
9087        ),
9088    }
9089}
9090
9091/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9092/// Windows does offer need cooperation this supervisor cannot assume: a console
9093/// control event requires sharing a console with the child, and `WM_CLOSE`
9094/// requires the child to pump a message loop. A supervised server process does
9095/// neither, so there is nothing to send and teardown is the wait followed by the
9096/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9097/// the thing `protocol: "none"` exists to avoid.
9098#[cfg(not(unix))]
9099fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9100    debug!(
9101        module_id,
9102        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9103    );
9104}
9105
9106fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9107    match final_state {
9108        ModuleState::Stopped => TerminalDisposition::Stopped,
9109        ModuleState::Disabled => TerminalDisposition::Disabled,
9110        ModuleState::Restarting => TerminalDisposition::Restarting,
9111        ModuleState::Failed => TerminalDisposition::Failed,
9112        ModuleState::Starting
9113        | ModuleState::Running
9114        | ModuleState::Unresponsive
9115        | ModuleState::Draining => {
9116            unreachable!("terminal exits only finish in terminal or restarting states")
9117        }
9118    }
9119}
9120
9121/// Release a reaped child's registration before allowing another spawn.
9122///
9123/// EOF is not a process-lifetime signal: an inherited socket can stay open
9124/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9125/// reading EOF. After the normal release grace, request connection close (which
9126/// cancels both reads and dispatch), then allow one more release grace for the
9127/// connection guard's forwarding cleanup. Never evict a different connection.
9128async fn release_dead_registration(
9129    registry: &Registry,
9130    forwarding: Option<&ForwardingTable>,
9131    snapshot: &SharedSnapshot,
9132    module_id: &str,
9133) -> Result<(), SuperviseError> {
9134    let result = async {
9135        let registration = registry
9136            .get_module(module_id)
9137            .map_err(SuperviseError::Registry)?;
9138        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9139            Ok(()) => return Ok(()),
9140            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9141            Err(err) => return Err(err),
9142        }
9143        let pid = lock_snapshot(snapshot)?.reaped_pid;
9144        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9145            warn!(
9146                module_id,
9147                pid,
9148                connection_id = registration.connection_id.get(),
9149                "reaped module registration outlived release grace; closing dead connection"
9150            );
9151            forwarding.request_connection_close(
9152                registration.connection_id,
9153                CloseReason::new(
9154                    "supervised_process_reaped",
9155                    format!("module '{module_id}' pid {pid} exited"),
9156                ),
9157            );
9158            wait_for_slot_registration_release(
9159                registry,
9160                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9161                REGISTRY_RELEASE_TIMEOUT,
9162            )
9163            .await?;
9164        }
9165        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9166    }
9167    .await;
9168    if let Err(err) = &result {
9169        fail_snapshot(snapshot, Some(module_id), None);
9170        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9171    }
9172    result
9173}
9174
9175/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9176/// plain stop or restart waits for before it spawns a replacement.
9177async fn wait_for_registration_release(
9178    registry: &Registry,
9179    module_id: &str,
9180    wait: Duration,
9181) -> Result<(), SuperviseError> {
9182    wait_for_slot_registration_release(
9183        registry,
9184        crate::registry::RegistrationSlot::Active(module_id),
9185        wait,
9186    )
9187    .await
9188}
9189
9190/// Wait for the registration in `slot` to go away.
9191///
9192/// Keyed on the slot rather than the bare module id because a successful swap
9193/// never empties the id's active slot (the promoted candidate is in it), so an
9194/// id-keyed wait for the incumbent's release would always time out. Draining a
9195/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9196/// incumbent's connection instead.
9197async fn wait_for_slot_registration_release(
9198    registry: &Registry,
9199    slot: crate::registry::RegistrationSlot<'_>,
9200    wait: Duration,
9201) -> Result<(), SuperviseError> {
9202    let deadline = Instant::now() + wait;
9203    let mut release_events = registration_release_events().subscribe();
9204    let still_active = |registration: &crate::registry::ModuleRegistration| {
9205        SuperviseError::RegistrationStillActive {
9206            module_id: registration.manifest.module_id.clone(),
9207            waited: wait,
9208        }
9209    };
9210    loop {
9211        let _observed_generation = *release_events.borrow_and_update();
9212        let Some(registration) = registry
9213            .registration(slot)
9214            .map_err(SuperviseError::Registry)?
9215        else {
9216            return Ok(());
9217        };
9218
9219        let now = Instant::now();
9220        if now >= deadline {
9221            return Err(still_active(&registration));
9222        }
9223
9224        let remaining = deadline.saturating_duration_since(now);
9225        match timeout(remaining, release_events.changed()).await {
9226            Ok(Ok(())) | Ok(Err(_)) => {}
9227            Err(_) => return Err(still_active(&registration)),
9228        }
9229    }
9230}
9231
9232#[cfg(test)]
9233mod slot_registration_wait_tests {
9234    use super::*;
9235    use crate::registry::{ConnectionId, RegistrationSlot};
9236    use subc_protocol::manifest::ModuleManifest;
9237
9238    #[tokio::test]
9239    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9240        let registry = Arc::new(Registry::default());
9241        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9242        let runtime = supervisor.runtime_config();
9243        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9244        let spec = ModuleSpec {
9245            module_id: "enable-stale-registration".to_string(),
9246            program: PathBuf::from("/missing/enable-retry-test"),
9247            args: Vec::new(),
9248            env: Vec::new(),
9249            reserved: false,
9250            reserved_prefixes: Vec::new(),
9251            protocol: ModuleProtocol::Subc,
9252            overlap: Default::default(),
9253        };
9254        let connection = ConnectionId::new(90);
9255        registry
9256            .register_with_control_ops(
9257                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9258                1,
9259                connection,
9260                Vec::new(),
9261            )
9262            .unwrap();
9263        let mut child = None;
9264        let err = set_child_enabled(
9265            &spec,
9266            &runtime,
9267            &registry,
9268            &supervisor.process_liveness,
9269            &snapshot,
9270            &mut child,
9271            true,
9272        )
9273        .await
9274        .unwrap_err();
9275        assert!(matches!(
9276            err,
9277            SuperviseError::RegistrationStillActive { .. }
9278        ));
9279        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9280        assert!(child.is_none());
9281        registry.deregister_connection(connection).unwrap();
9282        let err = set_child_enabled(
9283            &spec,
9284            &runtime,
9285            &registry,
9286            &supervisor.process_liveness,
9287            &snapshot,
9288            &mut child,
9289            true,
9290        )
9291        .await
9292        .unwrap_err();
9293        assert!(
9294            matches!(err, SuperviseError::Spawn { .. }),
9295            "second enable must attempt a spawn: {err}"
9296        );
9297        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9298    }
9299
9300    const INCUMBENT: u64 = 1;
9301    const CANDIDATE: u64 = 2;
9302
9303    fn swapped_registry() -> Arc<Registry> {
9304        let registry = Arc::new(Registry::default());
9305        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9306        registry
9307            .register_with_control_ops(
9308                manifest.clone(),
9309                1,
9310                ConnectionId::new(INCUMBENT),
9311                Vec::new(),
9312            )
9313            .unwrap();
9314        registry
9315            .register_candidate_with_control_ops(
9316                manifest,
9317                1,
9318                ConnectionId::new(CANDIDATE),
9319                Vec::new(),
9320            )
9321            .unwrap();
9322        registry
9323    }
9324
9325    /// After a promotion the id's active slot is held by the new process, so an
9326    /// id-keyed wait for the incumbent's release can never succeed; the
9327    /// connection-keyed wait completes as soon as the incumbent deregisters.
9328    #[tokio::test]
9329    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9330        let registry = swapped_registry();
9331        registry.promote_candidate("m").unwrap().unwrap();
9332
9333        assert!(matches!(
9334            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9335            Err(SuperviseError::RegistrationStillActive { .. })
9336        ));
9337
9338        // Still held while the incumbent's connection has not deregistered.
9339        assert!(matches!(
9340            wait_for_slot_registration_release(
9341                &registry,
9342                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9343                Duration::from_millis(50),
9344            )
9345            .await,
9346            Err(SuperviseError::RegistrationStillActive { .. })
9347        ));
9348
9349        let releaser = Arc::clone(&registry);
9350        let release = tokio::spawn(async move {
9351            sleep(Duration::from_millis(20)).await;
9352            releaser
9353                .deregister_connection(ConnectionId::new(INCUMBENT))
9354                .unwrap();
9355            notify_registration_release();
9356        });
9357        wait_for_slot_registration_release(
9358            &registry,
9359            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9360            Duration::from_secs(5),
9361        )
9362        .await
9363        .expect("the incumbent's own registration is released");
9364        release.await.unwrap();
9365        assert!(registry.get_module("m").unwrap().is_some());
9366    }
9367
9368    /// The candidate slot is waited on separately from the active slot: the
9369    /// incumbent's registration neither holds up nor stands in for it.
9370    #[tokio::test]
9371    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9372        let registry = swapped_registry();
9373        assert!(matches!(
9374            wait_for_slot_registration_release(
9375                &registry,
9376                RegistrationSlot::Candidate("m"),
9377                Duration::from_millis(50),
9378            )
9379            .await,
9380            Err(SuperviseError::RegistrationStillActive { .. })
9381        ));
9382        registry
9383            .deregister_connection(ConnectionId::new(CANDIDATE))
9384            .unwrap();
9385        wait_for_slot_registration_release(
9386            &registry,
9387            RegistrationSlot::Candidate("m"),
9388            Duration::from_millis(50),
9389        )
9390        .await
9391        .expect("a candidate slot with no candidate is released");
9392        assert!(registry
9393            .registration(RegistrationSlot::Active("m"))
9394            .unwrap()
9395            .is_some());
9396    }
9397}
9398
9399fn classify_exit(status: &ExitStatus) -> ExitReport {
9400    ExitReport {
9401        kind: if status.success() {
9402            ExitKind::Clean
9403        } else {
9404            ExitKind::Crash
9405        },
9406        code: status.code(),
9407        signal: exit_signal(status),
9408        at_ms: unix_ms_now(),
9409    }
9410}
9411
9412/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9413/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9414/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9415/// disposition still must be `Failed` so the terminal ring is not silently missing
9416/// an entry, matching what `fail_snapshot` records for this same arm.
9417fn wait_error_exit_report() -> ExitReport {
9418    ExitReport {
9419        kind: ExitKind::Crash,
9420        code: None,
9421        signal: None,
9422        at_ms: unix_ms_now(),
9423    }
9424}
9425
9426#[cfg(unix)]
9427fn exit_signal(status: &ExitStatus) -> Option<i32> {
9428    use std::os::unix::process::ExitStatusExt;
9429
9430    status.signal()
9431}
9432
9433#[cfg(not(unix))]
9434fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9435    None
9436}
9437
9438/// Give an operator-touched module its full crash budget back.
9439///
9440/// Named for the counter it used to zero; it now empties the in-window ring,
9441/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9442/// ledger of what happened survives every operator action.
9443fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9444    update_snapshot(snapshot, Some(module_id), |state| {
9445        state.clear_crash_restarts();
9446    })
9447}
9448
9449fn set_running(
9450    snapshot: &SharedSnapshot,
9451    child: &SupervisedChild,
9452    module_id: &str,
9453    spawn_events: &SpawnEventFeed,
9454) -> Result<(), SuperviseError> {
9455    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9456        module_id: Some(module_id.to_string()),
9457    })?;
9458    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9459    if std::mem::take(&mut state.coalesced_restart_pending) {
9460        let generation = state.spawn_generation;
9461        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9462    }
9463    state.drain_disposition_detail = None;
9464    state.spawn_failure = None;
9465    // Every caller of this is a plain spawn, which always uses the primary key;
9466    // a promoted swap candidate sets the flag itself after this returns.
9467    state.in_alternate_slot = false;
9468    state.configuration_updated_since_spawn = false;
9469    state.spawned_protocol = Some(child.protocol);
9470    state.state = ModuleState::Running;
9471    state.enabled = true;
9472    state.process_alive = true;
9473    state.pid = child.id();
9474    #[cfg(target_os = "macos")]
9475    {
9476        state.report_ready = Some(Arc::clone(&child.report_ready));
9477    }
9478    state.spawned_at_ms = Some(child.spawned_at_ms);
9479    state.spawned_from = Some(child.spawned_from.clone());
9480    state.spawned_file_identity = child.spawned_file_identity;
9481    state.process_start_time = child.process_start_time;
9482    Ok(())
9483}
9484
9485fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9486    state.process_alive = false;
9487    state.spawned_protocol = None;
9488    state.pid = None;
9489    #[cfg(target_os = "macos")]
9490    {
9491        state.report_ready = None;
9492    }
9493    state.spawned_at_ms = None;
9494    state.spawned_from = None;
9495    state.spawned_file_identity = None;
9496    state.process_start_time = None;
9497    state.deliberate_severance = None;
9498}
9499
9500#[cfg(test)]
9501fn record_deliberate_severance(
9502    snapshot: &SharedSnapshot,
9503    identity: ProcessIdentity,
9504) -> Result<(), SuperviseError> {
9505    update_snapshot(snapshot, None, |state| {
9506        state.deliberate_severance = Some(identity);
9507    })
9508}
9509
9510fn apply_deliberate_severance_marker(
9511    snapshot: &SharedSnapshot,
9512    exited_identity: Option<ProcessIdentity>,
9513    mut exit_report: ExitReport,
9514) -> ExitReport {
9515    let marker = lock_snapshot(snapshot)
9516        .ok()
9517        .and_then(|mut state| state.deliberate_severance.take());
9518    if marker.is_some() && marker == exited_identity {
9519        exit_report.kind = ExitKind::DeliberateSeverance;
9520    }
9521    exit_report
9522}
9523
9524fn classify_reaped_child_exit(
9525    snapshot: &SharedSnapshot,
9526    child: &SupervisedChild,
9527    status: &ExitStatus,
9528) -> ExitReport {
9529    let _ = update_snapshot(snapshot, None, |state| {
9530        state.reaped_pid = Some(child.pid);
9531        state.spawn_failure = child.spawn_failure.clone();
9532    });
9533    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9534}
9535
9536fn fail_snapshot(
9537    snapshot: &SharedSnapshot,
9538    module_id: Option<&str>,
9539    last_exit: Option<ExitReport>,
9540) {
9541    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9542        state.state = ModuleState::Failed;
9543        clear_current_process_facts(state);
9544        if let Some(last_exit) = last_exit {
9545            state.last_exit = Some(last_exit);
9546        }
9547    }) {
9548        error!(error = %err, "failed to mark supervisor state failed");
9549    }
9550}
9551
9552fn update_snapshot(
9553    snapshot: &SharedSnapshot,
9554    module_id: Option<&str>,
9555    update: impl FnOnce(&mut SupervisorSnapshot),
9556) -> Result<(), SuperviseError> {
9557    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9558        module_id: module_id.map(ToOwned::to_owned),
9559    })?;
9560    update(&mut state);
9561    Ok(())
9562}
9563
9564const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9565
9566fn lock_snapshot_for_control<'a>(
9567    snapshot: &'a SharedSnapshot,
9568    module_id: &str,
9569    caller: &'static str,
9570) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9571    let started_at = Instant::now();
9572    let guard = lock_snapshot(snapshot)?;
9573    let waited = started_at.elapsed();
9574    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9575        warn!(
9576            module_id = %module_id,
9577            waited_ms = waited.as_millis() as u64,
9578            caller = %caller,
9579            "slow snapshot lock"
9580        );
9581    }
9582    Ok(guard)
9583}
9584
9585fn lock_snapshot(
9586    snapshot: &SharedSnapshot,
9587) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9588    snapshot
9589        .lock()
9590        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9591}
9592
9593#[cfg(test)]
9594mod terminal_history_tests {
9595    use std::{
9596        path::PathBuf,
9597        sync::Arc,
9598        time::{Duration, Instant},
9599    };
9600
9601    use tokio::time::sleep;
9602
9603    use super::{
9604        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9605        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9606        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9607        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9608        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9609        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
9610        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
9611    };
9612    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9613    // use for their own wall-clock deadlines: crash-restart instants must be on
9614    // the same clock the production code stamps them with, which is tokio's (and
9615    // is what `start_paused` tests can move).
9616    use super::Instant as ClockInstant;
9617    use crate::{
9618        registry::Registry,
9619        terminal_ring::{TerminalRing, TerminalRingConfig},
9620    };
9621    use std::sync::Mutex;
9622    use subc_control::TerminalDisposition;
9623
9624    /// See the twin in `control.rs` for why this derives the path from
9625    /// `current_exe()` and why the existence check is here: `--lib` alone does
9626    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9627    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9628    pub(super) fn fake_aft_stub_path() -> PathBuf {
9629        let mut path = std::env::current_exe().expect("current_exe available in tests");
9630        path.pop();
9631        path.pop();
9632        path.push(if cfg!(windows) {
9633            "fake-aft-stub.exe"
9634        } else {
9635            "fake-aft-stub"
9636        });
9637        assert!(
9638            path.exists(),
9639            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9640             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9641            path.display()
9642        );
9643        path
9644    }
9645
9646    #[test]
9647    fn reserved_never_spawned_refuses_every_hello() {
9648        // The canary hole: a reserved id whose module has never spawned had NO
9649        // gate entry and admitted anyone -- the reservation protected the nonce
9650        // holder, not the NAME. Now the entry is present with no legitimate
9651        // holder and refuses all comers.
9652        let supervisor = SupervisorHandle::default();
9653        supervisor.apply_identity_configuration(&ModuleSpec {
9654            module_id: "never-spawned".to_string(),
9655            program: PathBuf::from("/usr/bin/false"),
9656            args: Vec::new(),
9657            env: Vec::new(),
9658            reserved: true,
9659            reserved_prefixes: Vec::new(),
9660            protocol: ModuleProtocol::Subc,
9661            overlap: Default::default(),
9662        });
9663        assert!(
9664            supervisor
9665                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9666                .is_some(),
9667            "forged nonce must refuse on a reserved never-spawned id"
9668        );
9669        assert!(
9670            supervisor
9671                .reserved_hello_rejection("never-spawned", None)
9672                .is_some(),
9673            "absent nonce must refuse on a reserved never-spawned id"
9674        );
9675        // And a real spawn nonce minted later admits exactly that nonce.
9676        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9677        supervisor.apply_identity_configuration(&ModuleSpec {
9678            module_id: "never-spawned".to_string(),
9679            program: PathBuf::from("/usr/bin/false"),
9680            args: Vec::new(),
9681            env: Vec::new(),
9682            reserved: true,
9683            reserved_prefixes: Vec::new(),
9684            protocol: ModuleProtocol::Subc,
9685            overlap: Default::default(),
9686        });
9687        assert!(supervisor
9688            .reserved_hello_rejection("never-spawned", Some("minted"))
9689            .is_none());
9690        assert!(supervisor
9691            .reserved_hello_rejection("never-spawned", Some("forged"))
9692            .is_some());
9693    }
9694
9695    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9696    /// happened, which is what "spent budget" looks like to every reader.
9697    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9698        let now = ClockInstant::now();
9699        for _ in 0..count {
9700            state.crash_restarts.push_back(now);
9701        }
9702    }
9703
9704    /// Age the oldest recorded restart out of `window`, standing in for the hours
9705    /// that would otherwise have to pass. Injecting the instant is the point: a
9706    /// test that slept a real window would take ten minutes and still prove less.
9707    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9708        let aged = state
9709            .crash_restarts
9710            .front()
9711            .expect("a crash restart must be recorded before it can be aged")
9712            .checked_sub(window + Duration::from_secs(1))
9713            .expect("the test clock is far enough from its origin to age an instant");
9714        state.crash_restarts[0] = aged;
9715    }
9716
9717    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9718        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9719        seed_crash_restarts(&mut state, count);
9720        state
9721    }
9722
9723    #[test]
9724    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9725        let policy = RestartPolicy::new(3, Duration::ZERO);
9726        let now = ClockInstant::now();
9727        assert!(daemon_will_restart(
9728            &mut snapshot_with_restarts(true, 2),
9729            &policy,
9730            now
9731        ));
9732        assert!(!daemon_will_restart(
9733            &mut snapshot_with_restarts(true, 3),
9734            &policy,
9735            now
9736        ));
9737        assert!(!daemon_will_restart(
9738            &mut snapshot_with_restarts(false, 0),
9739            &policy,
9740            now
9741        ));
9742    }
9743
9744    #[test]
9745    fn crash_restart_backoff_escalates_with_in_window_count() {
9746        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9747            .with_max_backoff(Duration::from_secs(30));
9748        let now = ClockInstant::now();
9749        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9750        let schedules = (0..4)
9751            .map(|_| {
9752                state
9753                    .next_crash_restart(&policy, now)
9754                    .expect("the test policy allows four crash restarts")
9755            })
9756            .collect::<Vec<_>>();
9757
9758        assert_eq!(
9759            schedules
9760                .iter()
9761                .map(|schedule| schedule.restart_in_window)
9762                .collect::<Vec<_>>(),
9763            vec![0, 1, 2, 3]
9764        );
9765        assert_eq!(
9766            schedules
9767                .iter()
9768                .map(|schedule| schedule.delay)
9769                .collect::<Vec<_>>(),
9770            vec![
9771                Duration::from_millis(100),
9772                Duration::from_secs(1),
9773                Duration::from_secs(10),
9774                Duration::from_secs(30),
9775            ]
9776        );
9777    }
9778
9779    #[test]
9780    fn crash_restart_backoff_resets_after_ring_clear() {
9781        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9782        let now = ClockInstant::now();
9783        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9784        assert_eq!(
9785            state.next_crash_restart(&policy, now).unwrap().delay,
9786            Duration::from_millis(100)
9787        );
9788        assert_eq!(
9789            state.next_crash_restart(&policy, now).unwrap().delay,
9790            Duration::from_secs(1)
9791        );
9792
9793        state.clear_crash_restarts();
9794        let schedule = state
9795            .next_crash_restart(&policy, now)
9796            .expect("a cleared ring must allow another restart");
9797        assert_eq!(schedule.restart_in_window, 0);
9798        assert_eq!(schedule.delay, Duration::from_millis(100));
9799    }
9800
9801    #[test]
9802    fn crash_restart_backoff_ignores_aged_restarts() {
9803        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9804        let now = ClockInstant::now();
9805        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9806        state
9807            .next_crash_restart(&policy, now)
9808            .expect("the first restart is allowed");
9809        state
9810            .next_crash_restart(&policy, now)
9811            .expect("the second restart is allowed");
9812        state.crash_restarts[0] = now
9813            .checked_sub(policy.window + Duration::from_secs(1))
9814            .expect("the fake clock can age a restart past the window");
9815
9816        let schedule = state
9817            .next_crash_restart(&policy, now)
9818            .expect("an aged restart must release its slot");
9819        assert_eq!(schedule.restart_in_window, 1);
9820        assert_eq!(schedule.delay, Duration::from_secs(1));
9821        assert_eq!(state.crash_restarts.len(), 2);
9822    }
9823
9824    /// The budget is a rate: the same three spent restarts refuse a respawn
9825    /// while they are recent and allow one once they have aged past the window.
9826    /// Nothing about the module changed in between, which is the whole point.
9827    #[test]
9828    fn a_budget_spent_before_the_window_no_longer_refuses() {
9829        let policy = RestartPolicy::new(3, Duration::ZERO);
9830        let mut state = snapshot_with_restarts(true, 3);
9831        let now = ClockInstant::now();
9832        assert!(!daemon_will_restart(&mut state, &policy, now));
9833
9834        assert!(daemon_will_restart(
9835            &mut state,
9836            &policy,
9837            now + policy.window + Duration::from_secs(1)
9838        ));
9839        assert!(
9840            state.crash_restarts.is_empty(),
9841            "reading the budget must drop the instants that left the window"
9842        );
9843    }
9844
9845    fn module_with_recovery_snapshot(
9846        state: ModuleState,
9847        enabled: bool,
9848        restart_count: u32,
9849    ) -> SupervisedModule {
9850        let registry = Arc::new(Registry::default());
9851        let supervisor =
9852            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9853        let module = supervisor
9854            .spawn(ModuleSpec {
9855                module_id: "recovery-snapshot".to_string(),
9856                program: fake_aft_stub_path(),
9857                args: Vec::new(),
9858                env: Vec::new(),
9859                reserved: false,
9860                reserved_prefixes: Vec::new(),
9861                protocol: ModuleProtocol::Subc,
9862                overlap: Default::default(),
9863            })
9864            .unwrap();
9865        update_snapshot(
9866            &module.inner.snapshot,
9867            Some("recovery-snapshot"),
9868            |snapshot| {
9869                snapshot.state = state;
9870                snapshot.enabled = enabled;
9871                seed_crash_restarts(snapshot, restart_count);
9872            },
9873        )
9874        .unwrap();
9875        module
9876    }
9877
9878    #[cfg(target_os = "linux")]
9879    #[tokio::test]
9880    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9881        let supervisor =
9882            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
9883                .with_cgroup_placement(None);
9884        let result = supervisor.spawn(ModuleSpec {
9885            module_id: "no-cgroup-placement".to_string(),
9886            program: fake_aft_stub_path(),
9887            args: Vec::new(),
9888            env: Vec::new(),
9889            reserved: false,
9890            reserved_prefixes: Vec::new(),
9891            protocol: ModuleProtocol::Subc,
9892            overlap: Default::default(),
9893        });
9894
9895        assert!(
9896            result.is_ok(),
9897            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9898        );
9899    }
9900
9901    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9902    async fn undecided_snapshot_uses_shared_restart_predicate() {
9903        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9904            .will_recover_after_connection_loss()
9905            .unwrap());
9906        assert!(
9907            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9908                .will_recover_after_connection_loss()
9909                .unwrap()
9910        );
9911    }
9912
9913    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9914    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9915        assert!(
9916            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9917                .will_recover_after_connection_loss()
9918                .unwrap()
9919        );
9920    }
9921
9922    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9923    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9924        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9925            .will_recover_after_connection_loss()
9926            .unwrap());
9927        assert!(
9928            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9929                .will_recover_after_connection_loss()
9930                .unwrap()
9931        );
9932    }
9933
9934    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9935    async fn warming_snapshot_is_limited_to_startup_phases() {
9936        for state in [
9937            ModuleState::Starting,
9938            ModuleState::Running,
9939            ModuleState::Restarting,
9940        ] {
9941            assert!(
9942                module_with_recovery_snapshot(state, true, 0)
9943                    .is_warming()
9944                    .unwrap(),
9945                "{state:?} should be warming"
9946            );
9947        }
9948        for state in [
9949            ModuleState::Unresponsive,
9950            ModuleState::Draining,
9951            ModuleState::Stopped,
9952            ModuleState::Failed,
9953            ModuleState::Disabled,
9954        ] {
9955            assert!(
9956                !module_with_recovery_snapshot(state, true, 0)
9957                    .is_warming()
9958                    .unwrap(),
9959                "{state:?} should not be warming"
9960            );
9961        }
9962    }
9963
9964    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9965    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
9966        let registry = Arc::new(Registry::default());
9967        let supervisor =
9968            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
9969        let module = supervisor
9970            .spawn(ModuleSpec {
9971                module_id: "terminal-history".to_string(),
9972                program: fake_aft_stub_path(),
9973                args: Vec::new(),
9974                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9975                reserved: false,
9976                reserved_prefixes: Vec::new(),
9977                protocol: ModuleProtocol::Subc,
9978                overlap: Default::default(),
9979            })
9980            .unwrap();
9981
9982        let deadline = Instant::now() + Duration::from_secs(5);
9983        loop {
9984            let history = module.terminal_history();
9985            if history.entries.len() == 2 {
9986                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
9987                assert_eq!(history.dropped, 0);
9988                assert_eq!(
9989                    history
9990                        .entries
9991                        .iter()
9992                        .map(|entry| entry.exit_code)
9993                        .collect::<Vec<_>>(),
9994                    vec![Some(23), Some(23)]
9995                );
9996                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
9997                return;
9998            }
9999            assert!(
10000                Instant::now() < deadline,
10001                "module did not retain two terminal exits: {history:?}"
10002            );
10003            sleep(Duration::from_millis(10)).await;
10004        }
10005    }
10006
10007    /// A disable issued while a crash respawn is still backing off must preempt
10008    /// that respawn: the operator's stop wins, the disable must not queue behind
10009    /// the backoff, and the module must never come back up afterwards.
10010    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10011    async fn disable_during_crash_backoff_cancels_pending_respawn() {
10012        let backoff = Duration::from_secs(2);
10013        let supervisor = Supervisor::new_for_test(
10014            Arc::new(Registry::default()),
10015            RestartPolicy::new(10, backoff),
10016        );
10017        let module = supervisor
10018            .spawn(ModuleSpec {
10019                module_id: "disable-during-backoff".to_string(),
10020                program: fake_aft_stub_path(),
10021                args: Vec::new(),
10022                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10023                reserved: false,
10024                reserved_prefixes: Vec::new(),
10025                protocol: ModuleProtocol::Subc,
10026                overlap: Default::default(),
10027            })
10028            .unwrap();
10029
10030        // Wait for the first crash to put the module into its backoff window.
10031        let deadline = Instant::now() + Duration::from_secs(5);
10032        loop {
10033            if module.status().unwrap().state == ModuleState::Restarting {
10034                break;
10035            }
10036            assert!(
10037                Instant::now() < deadline,
10038                "module never entered the crash backoff"
10039            );
10040            sleep(Duration::from_millis(10)).await;
10041        }
10042
10043        let started = Instant::now();
10044        module.set_enabled(false).await.unwrap();
10045        let waited = started.elapsed();
10046
10047        assert!(
10048            waited < backoff / 2,
10049            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10050        );
10051        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10052
10053        // Outlast the backoff: the respawn it was counting down to must never run.
10054        sleep(backoff + Duration::from_millis(500)).await;
10055        let status = module.status().unwrap();
10056        assert_eq!(status.state, ModuleState::Disabled);
10057        assert_eq!(
10058            status.spawn_generation, 1,
10059            "module respawned after the operator disabled it"
10060        );
10061    }
10062
10063    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
10064    /// the shape of nats-server, the program this rule exists for.
10065    #[cfg(unix)]
10066    fn protocol_none_sigterm_exits_clean_spec(
10067        module_id: &str,
10068        dir: &std::path::Path,
10069    ) -> (ModuleSpec, PathBuf, PathBuf) {
10070        let ready = dir.join("ready");
10071        let marker = dir.join("sigterm");
10072        let spec = ModuleSpec {
10073            module_id: module_id.to_string(),
10074            program: fake_aft_stub_path(),
10075            args: Vec::new(),
10076            env: vec![
10077                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10078                (
10079                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10080                    marker.display().to_string(),
10081                ),
10082                (
10083                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10084                    ready.display().to_string(),
10085                ),
10086            ],
10087            reserved: false,
10088            reserved_prefixes: Vec::new(),
10089            protocol: ModuleProtocol::None,
10090            overlap: Default::default(),
10091        };
10092        (spec, ready, marker)
10093    }
10094
10095    /// Wait for a file the child writes, so a signal is never sent before the
10096    /// child's SIGTERM handler is installed (the default disposition would
10097    /// kill it by signal and the exit would not be clean).
10098    #[cfg(unix)]
10099    async fn wait_for_file(path: &std::path::Path) {
10100        let deadline = Instant::now() + Duration::from_secs(10);
10101        while !path.exists() {
10102            assert!(
10103                Instant::now() < deadline,
10104                "{} never appeared",
10105                path.display()
10106            );
10107            sleep(Duration::from_millis(10)).await;
10108        }
10109    }
10110
10111    /// A protocol-none module that exits 0 because something OUTSIDE the
10112    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10113    /// the crash-path disposition rather than `stopped`.
10114    #[cfg(unix)]
10115    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10116    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10117        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10118        let (spec, ready, marker) =
10119            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10120        let supervisor = Supervisor::new_for_test(
10121            Arc::new(Registry::default()),
10122            RestartPolicy::new(3, Duration::ZERO),
10123        );
10124        let module = supervisor.spawn(spec).unwrap();
10125        wait_for_file(&ready).await;
10126        let first_pid = module
10127            .status()
10128            .unwrap()
10129            .pid
10130            .expect("a running module reports its pid");
10131
10132        rustix::process::kill_process(
10133            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10134            rustix::process::Signal::TERM,
10135        )
10136        .unwrap();
10137
10138        let deadline = Instant::now() + Duration::from_secs(10);
10139        let respawned = loop {
10140            let status = module.status().unwrap();
10141            if status.state == ModuleState::Running
10142                && status.pid.is_some_and(|pid| pid != first_pid)
10143            {
10144                break status;
10145            }
10146            assert!(
10147                Instant::now() < deadline,
10148                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10149            );
10150            sleep(Duration::from_millis(10)).await;
10151        };
10152        assert_eq!(respawned.spawn_generation, 2);
10153        assert!(
10154            marker.exists(),
10155            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10156        );
10157
10158        let history = module.terminal_history();
10159        assert_eq!(history.entries.len(), 1, "{history:?}");
10160        let entry = &history.entries[0];
10161        assert_eq!(entry.exit_code, Some(0));
10162        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10163        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10164
10165        module.stop().await.unwrap();
10166    }
10167
10168    /// Repeated unrequested clean exits of a protocol-none module spend the
10169    /// restart budget exactly as crashes do, and the module ends `failed` with
10170    /// the budget named.
10171    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10172    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10173        let supervisor = Supervisor::new_for_test(
10174            Arc::new(Registry::default()),
10175            RestartPolicy::new(1, Duration::ZERO),
10176        );
10177        let module = supervisor
10178            .spawn(ModuleSpec {
10179                module_id: "none-clean-exit-budget".to_string(),
10180                program: fake_aft_stub_path(),
10181                args: Vec::new(),
10182                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10183                reserved: false,
10184                reserved_prefixes: Vec::new(),
10185                protocol: ModuleProtocol::None,
10186                overlap: Default::default(),
10187            })
10188            .unwrap();
10189
10190        let deadline = Instant::now() + Duration::from_secs(10);
10191        loop {
10192            let status = module.status().unwrap();
10193            if status.state == ModuleState::Failed {
10194                break;
10195            }
10196            assert!(
10197                Instant::now() < deadline,
10198                "module never exhausted its budget: {status:?} {:?}",
10199                module.terminal_history()
10200            );
10201            sleep(Duration::from_millis(10)).await;
10202        }
10203        let history = module.terminal_history();
10204        assert_eq!(
10205            history
10206                .entries
10207                .iter()
10208                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10209                .collect::<Vec<_>>(),
10210            vec![
10211                (Some(0), TerminalDisposition::Restarting),
10212                (Some(0), TerminalDisposition::Failed),
10213            ]
10214        );
10215        let detail = history.entries[1]
10216            .disposition_detail
10217            .as_deref()
10218            .expect("a budget failure names the budget");
10219        assert!(detail.contains("max_restarts=1"), "{detail}");
10220        assert_eq!(module.status().unwrap().spawn_generation, 2);
10221    }
10222
10223    /// A stop the supervisor itself requests still stops a protocol-none
10224    /// module, even though the child answers the SIGTERM with exit 0.
10225    #[cfg(unix)]
10226    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10227    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10228        for disable in [false, true] {
10229            let label = if disable {
10230                "none-requested-disable"
10231            } else {
10232                "none-requested-stop"
10233            };
10234            let dir = subc_test_support::TestTempDir::new(label);
10235            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10236            let supervisor = Supervisor::new_for_test(
10237                Arc::new(Registry::default()),
10238                RestartPolicy::new(3, Duration::ZERO),
10239            );
10240            let module = supervisor.spawn(spec).unwrap();
10241            wait_for_file(&ready).await;
10242
10243            if disable {
10244                module.set_enabled(false).await.unwrap();
10245            } else {
10246                module.stop().await.unwrap();
10247            }
10248            assert!(
10249                marker.exists(),
10250                "{label}: the child must have left through its SIGTERM handler with exit 0"
10251            );
10252
10253            // Long enough for a zero-backoff respawn to have happened if the
10254            // exit had been treated as a crash.
10255            sleep(Duration::from_millis(500)).await;
10256            let status = module.status().unwrap();
10257            let expected = if disable {
10258                ModuleState::Disabled
10259            } else {
10260                ModuleState::Stopped
10261            };
10262            assert_eq!(status.state, expected, "{label}");
10263            assert_eq!(
10264                status.spawn_generation, 1,
10265                "{label}: respawned after a requested stop"
10266            );
10267            let history = module.terminal_history();
10268            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10269            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10270            assert_ne!(
10271                history.entries[0].disposition,
10272                TerminalDisposition::Restarting,
10273                "{label}"
10274            );
10275        }
10276    }
10277
10278    /// A subc-wire module that exits 0 on its own is still a stop: the
10279    /// protocol-none rule must not reach it.
10280    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10281    async fn subc_wire_clean_exit_is_still_a_stop() {
10282        let supervisor = Supervisor::new_for_test(
10283            Arc::new(Registry::default()),
10284            RestartPolicy::new(3, Duration::ZERO),
10285        );
10286        let module = supervisor
10287            .spawn(ModuleSpec {
10288                module_id: "wire-clean-exit".to_string(),
10289                program: fake_aft_stub_path(),
10290                args: Vec::new(),
10291                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10292                reserved: false,
10293                reserved_prefixes: Vec::new(),
10294                protocol: ModuleProtocol::Subc,
10295                overlap: Default::default(),
10296            })
10297            .unwrap();
10298
10299        let deadline = Instant::now() + Duration::from_secs(10);
10300        while module.terminal_history().entries.is_empty() {
10301            assert!(Instant::now() < deadline, "module never exited");
10302            sleep(Duration::from_millis(10)).await;
10303        }
10304        // Long enough for a zero-backoff respawn to have happened.
10305        sleep(Duration::from_millis(500)).await;
10306        let status = module.status().unwrap();
10307        assert_eq!(status.state, ModuleState::Stopped);
10308        assert_eq!(status.spawn_generation, 1);
10309        let history = module.terminal_history();
10310        assert_eq!(history.entries.len(), 1, "{history:?}");
10311        assert_eq!(history.entries[0].exit_code, Some(0));
10312        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10313    }
10314
10315    #[cfg(unix)]
10316    #[tokio::test]
10317    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10318        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10319        let record = dir.join("live-children.json");
10320        let supervisor = Supervisor::new_for_test(
10321            Arc::new(Registry::default()),
10322            RestartPolicy::new(0, Duration::ZERO),
10323        );
10324        let mut runtime = supervisor.runtime_config();
10325        runtime.child_roster.record_to(record.clone());
10326        let gate = Arc::new(super::ReloadExitRecordGate::default());
10327        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10328        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10329        let spec = ModuleSpec {
10330            module_id: "reload-exit-roster".into(),
10331            program: fake_aft_stub_path(),
10332            args: Vec::new(),
10333            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10334            reserved: false,
10335            reserved_prefixes: Vec::new(),
10336            protocol: ModuleProtocol::Subc,
10337            overlap: Default::default(),
10338        };
10339        let mut child = None;
10340        let reload = super::finish_reload_child(
10341            &spec,
10342            &runtime,
10343            &supervisor.registry,
10344            &supervisor.process_liveness,
10345            &snapshot,
10346            &mut child,
10347        );
10348        tokio::pin!(reload);
10349        tokio::select! {
10350            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10351            _ = gate.reached.notified() => {}
10352        }
10353        assert!(runtime
10354            .terminal_ring
10355            .lock()
10356            .unwrap()
10357            .snapshot()
10358            .entries
10359            .is_empty());
10360        assert_eq!(
10361            crate::live_children::read_record(&record).unwrap().len(),
10362            1,
10363            "shutdown must still wait for the reaped child until its terminal record exists"
10364        );
10365        runtime.child_roster.close();
10366        gate.resume.notify_one();
10367        assert!(reload.await.is_err());
10368        assert!(crate::live_children::read_record(&record)
10369            .unwrap()
10370            .is_empty());
10371        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10372        assert_eq!(history.entries.len(), 1);
10373        assert_eq!(
10374            history.entries[0].disposition,
10375            TerminalDisposition::DaemonShutdown
10376        );
10377    }
10378
10379    /// Each restart-producing arm has its own state transition. Keeping their
10380    /// lifetime count assertions adjacent prevents a later new arm from silently
10381    /// spending budget without recording the historical restart.
10382    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10383    async fn every_restart_increment_path_advances_lifetime_count() {
10384        let supervisor = Supervisor::new_for_test(
10385            Arc::new(Registry::default()),
10386            RestartPolicy::new(1, Duration::ZERO),
10387        );
10388        let runtime = supervisor.runtime_config();
10389        let spec = ModuleSpec {
10390            module_id: "lifetime-increment-path".to_string(),
10391            program: PathBuf::from("/unused/lifetime-increment-path"),
10392            args: Vec::new(),
10393            env: Vec::new(),
10394            reserved: false,
10395            reserved_prefixes: Vec::new(),
10396            protocol: ModuleProtocol::Subc,
10397            overlap: Default::default(),
10398        };
10399
10400        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10401        assert!(matches!(
10402            on_child_exit(
10403                &spec,
10404                runtime.restart_policy,
10405                &supervisor.registry,
10406                &crash_snapshot,
10407                &runtime.terminal_ring,
10408                &runtime.spawn_events,
10409                &runtime.child_roster,
10410                ExitReport {
10411                    kind: ExitKind::Crash,
10412                    code: Some(1),
10413                    signal: None,
10414                    at_ms: 1,
10415                },
10416            )
10417            .await,
10418            NextAction::Restart { schedule: _ }
10419        ));
10420        let (crash_restarts, crash_lifetime) = {
10421            let state = lock_snapshot(&crash_snapshot).unwrap();
10422            (state.crash_restarts.len(), state.lifetime_restarts)
10423        };
10424        assert_eq!(crash_restarts, 1);
10425        assert_eq!(crash_lifetime, 1);
10426
10427        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10428        let mut health_child = None;
10429        assert!(matches!(
10430            health_restart_child(
10431                &spec,
10432                &runtime,
10433                &supervisor.registry,
10434                &supervisor.process_liveness,
10435                &health_snapshot,
10436                &mut health_child,
10437                SupervisorHealthStatus::Failing,
10438                None,
10439                2,
10440            )
10441            .await,
10442            Ok(())
10443        ));
10444        assert!(health_child.is_none());
10445        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10446        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10447        let (health_restarts, health_lifetime) = {
10448            let state = lock_snapshot(&health_snapshot).unwrap();
10449            (state.crash_restarts.len(), state.lifetime_restarts)
10450        };
10451        assert_eq!(health_restarts, 1);
10452        assert_eq!(health_lifetime, 1);
10453
10454        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10455        let mut reload_child = None;
10456        assert!(matches!(
10457            handle_reload_spawn_failure(
10458                &spec,
10459                &runtime,
10460                &supervisor.process_liveness,
10461                &reload_snapshot,
10462                &mut reload_child,
10463                "forced reload spawn failure".to_string(),
10464            )
10465            .await,
10466            Err(SuperviseError::ReloadFailed { .. })
10467        ));
10468        let (reload_restarts, reload_lifetime) = {
10469            let state = lock_snapshot(&reload_snapshot).unwrap();
10470            (state.crash_restarts.len(), state.lifetime_restarts)
10471        };
10472        assert_eq!(reload_restarts, 1);
10473        assert_eq!(reload_lifetime, 1);
10474    }
10475
10476    #[tokio::test]
10477    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10478        let supervisor = Supervisor::new_for_test(
10479            Arc::new(Registry::default()),
10480            RestartPolicy::new(3, Duration::ZERO),
10481        );
10482        let runtime = supervisor.runtime_config();
10483        let spec = ModuleSpec {
10484            module_id: "deliberately-severed".to_string(),
10485            program: PathBuf::from("/unused/deliberately-severed"),
10486            args: Vec::new(),
10487            env: Vec::new(),
10488            reserved: false,
10489            reserved_prefixes: Vec::new(),
10490            protocol: ModuleProtocol::Subc,
10491            overlap: Default::default(),
10492        };
10493        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10494        let process = ProcessIdentity {
10495            pid: 41,
10496            start_time: 101,
10497        };
10498        record_deliberate_severance(&snapshot, process).unwrap();
10499        let exit_report = apply_deliberate_severance_marker(
10500            &snapshot,
10501            Some(process),
10502            ExitReport {
10503                kind: ExitKind::Crash,
10504                code: Some(1),
10505                signal: None,
10506                at_ms: 1,
10507            },
10508        );
10509        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10510
10511        assert!(matches!(
10512            on_child_exit(
10513                &spec,
10514                runtime.restart_policy,
10515                &supervisor.registry,
10516                &snapshot,
10517                &runtime.terminal_ring,
10518                &runtime.spawn_events,
10519                &runtime.child_roster,
10520                exit_report,
10521            )
10522            .await,
10523            NextAction::Restart { schedule: _ }
10524        ));
10525        let state = lock_snapshot(&snapshot).unwrap();
10526        assert_eq!(state.lifetime_restarts, 1);
10527        assert_eq!(state.crash_restarts.len(), 0);
10528    }
10529
10530    #[tokio::test]
10531    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10532        let supervisor = Supervisor::new_for_test(
10533            Arc::new(Registry::default()),
10534            RestartPolicy::new(3, Duration::ZERO),
10535        );
10536        let runtime = supervisor.runtime_config();
10537        let spec = ModuleSpec {
10538            module_id: "genuine-crash".to_string(),
10539            program: PathBuf::from("/unused/genuine-crash"),
10540            args: Vec::new(),
10541            env: Vec::new(),
10542            reserved: false,
10543            reserved_prefixes: Vec::new(),
10544            protocol: ModuleProtocol::Subc,
10545            overlap: Default::default(),
10546        };
10547        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10548
10549        assert!(matches!(
10550            on_child_exit(
10551                &spec,
10552                runtime.restart_policy,
10553                &supervisor.registry,
10554                &snapshot,
10555                &runtime.terminal_ring,
10556                &runtime.spawn_events,
10557                &runtime.child_roster,
10558                ExitReport {
10559                    kind: ExitKind::Crash,
10560                    code: Some(1),
10561                    signal: None,
10562                    at_ms: 1,
10563                },
10564            )
10565            .await,
10566            NextAction::Restart { schedule: _ }
10567        ));
10568        let state = lock_snapshot(&snapshot).unwrap();
10569        assert_eq!(state.lifetime_restarts, 1);
10570        assert_eq!(state.crash_restarts.len(), 1);
10571    }
10572
10573    fn crash_exit_report(at_ms: u64) -> ExitReport {
10574        ExitReport {
10575            kind: ExitKind::Crash,
10576            code: Some(1),
10577            signal: None,
10578            at_ms,
10579        }
10580    }
10581
10582    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10583        ModuleSpec {
10584            module_id: module_id.to_string(),
10585            program: PathBuf::from("/unused").join(module_id),
10586            args: Vec::new(),
10587            env: Vec::new(),
10588            reserved: false,
10589            reserved_prefixes: Vec::new(),
10590            protocol: ModuleProtocol::Subc,
10591            overlap: Default::default(),
10592        }
10593    }
10594
10595    /// A real crash loop still stops. Three crashes with nothing aging out spend
10596    /// a budget of two and the third respawn is refused, and both surfaces an
10597    /// operator has -- the log line and the retained terminal record -- name the
10598    /// window rather than only the cap, because `max_restarts=2` alone is what
10599    /// this budget used to mean.
10600    #[tokio::test]
10601    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10602        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10603        let supervisor = Supervisor::new_for_test(
10604            Arc::new(Registry::default()),
10605            RestartPolicy::new(2, Duration::ZERO),
10606        );
10607        let runtime = supervisor.runtime_config();
10608        let spec = windowed_crash_spec("crash-loop-in-window");
10609        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10610
10611        for attempt in 1..=2 {
10612            assert!(
10613                matches!(
10614                    on_child_exit(
10615                        &spec,
10616                        runtime.restart_policy,
10617                        &supervisor.registry,
10618                        &snapshot,
10619                        &runtime.terminal_ring,
10620                        &runtime.spawn_events,
10621                        &runtime.child_roster,
10622                        crash_exit_report(attempt),
10623                    )
10624                    .await,
10625                    NextAction::Restart { schedule: _ }
10626                ),
10627                "crash {attempt} is inside the budget and must respawn"
10628            );
10629        }
10630
10631        assert!(matches!(
10632            on_child_exit(
10633                &spec,
10634                runtime.restart_policy,
10635                &supervisor.registry,
10636                &snapshot,
10637                &runtime.terminal_ring,
10638                &runtime.spawn_events,
10639                &runtime.child_roster,
10640                crash_exit_report(3),
10641            )
10642            .await,
10643            NextAction::Stop { .. }
10644        ));
10645
10646        {
10647            let state = lock_snapshot(&snapshot).unwrap();
10648            assert_eq!(state.state, ModuleState::Failed);
10649            assert_eq!(state.crash_restarts.len(), 2);
10650            assert_eq!(state.lifetime_restarts, 2);
10651        }
10652
10653        let history = runtime
10654            .terminal_ring
10655            .lock()
10656            .expect("terminal ring is not poisoned")
10657            .snapshot();
10658        let last = history
10659            .entries
10660            .last()
10661            .expect("the refused crash is retained");
10662        assert_eq!(last.disposition, TerminalDisposition::Failed);
10663        assert_eq!(
10664            last.disposition_detail.as_deref(),
10665            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10666        );
10667
10668        let captured = crate::router::test_log::captured_logs(&logs);
10669        assert!(
10670            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10671            "the stop must be logged with its window: {captured}"
10672        );
10673    }
10674
10675    /// The rate, stated as a test: three crashes where the first has aged past
10676    /// the window are two crashes as far as the budget is concerned, so the
10677    /// third respawn is allowed and the ring holds only the two recent ones.
10678    ///
10679    /// This is the case a lifetime counter got wrong -- and the case the daemon
10680    /// now hits routinely, since a module exits non-zero every time its
10681    /// connection to the daemon drops.
10682    #[tokio::test]
10683    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10684        let supervisor = Supervisor::new_for_test(
10685            Arc::new(Registry::default()),
10686            RestartPolicy::new(2, Duration::ZERO),
10687        );
10688        let runtime = supervisor.runtime_config();
10689        let spec = windowed_crash_spec("crash-across-windows");
10690        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10691
10692        for attempt in 1..=2 {
10693            assert!(matches!(
10694                on_child_exit(
10695                    &spec,
10696                    runtime.restart_policy,
10697                    &supervisor.registry,
10698                    &snapshot,
10699                    &runtime.terminal_ring,
10700                    &runtime.spawn_events,
10701                    &runtime.child_roster,
10702                    crash_exit_report(attempt),
10703                )
10704                .await,
10705                NextAction::Restart { schedule: _ }
10706            ));
10707        }
10708
10709        // The oldest crash moves out of the window; nothing else about the
10710        // module changes.
10711        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10712            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10713        })
10714        .unwrap();
10715
10716        assert!(
10717            matches!(
10718                on_child_exit(
10719                    &spec,
10720                    runtime.restart_policy,
10721                    &supervisor.registry,
10722                    &snapshot,
10723                    &runtime.terminal_ring,
10724                    &runtime.spawn_events,
10725                    &runtime.child_roster,
10726                    crash_exit_report(3),
10727                )
10728                .await,
10729                NextAction::Restart { schedule: _ }
10730            ),
10731            "a crash older than the window must not hold a budget slot"
10732        );
10733
10734        let state = lock_snapshot(&snapshot).unwrap();
10735        assert_eq!(state.state, ModuleState::Restarting);
10736        assert_eq!(
10737            state.crash_restarts.len(),
10738            2,
10739            "the aged instant is dropped and the new one takes its place"
10740        );
10741        assert_eq!(
10742            state.lifetime_restarts, 3,
10743            "the ledger counts every restart, including the ones the window forgot"
10744        );
10745    }
10746
10747    /// An operator restart hands the budget back whole, and the ledger keeps
10748    /// counting. Those are different questions -- "how close is this module to
10749    /// being stopped" and "how many times has it been replaced" -- and the
10750    /// operator action answers only the first.
10751    #[tokio::test]
10752    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10753        let supervisor = Supervisor::new_for_test(
10754            Arc::new(Registry::default()),
10755            RestartPolicy::new(2, Duration::ZERO),
10756        );
10757        let runtime = supervisor.runtime_config();
10758        let spec = windowed_crash_spec("operator-cleared-budget");
10759        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10760
10761        for attempt in 1..=2 {
10762            assert!(matches!(
10763                on_child_exit(
10764                    &spec,
10765                    runtime.restart_policy,
10766                    &supervisor.registry,
10767                    &snapshot,
10768                    &runtime.terminal_ring,
10769                    &runtime.spawn_events,
10770                    &runtime.child_roster,
10771                    crash_exit_report(attempt),
10772                )
10773                .await,
10774                NextAction::Restart { schedule: _ }
10775            ));
10776        }
10777
10778        reset_restart_count(&snapshot, &spec.module_id).unwrap();
10779        {
10780            let state = lock_snapshot(&snapshot).unwrap();
10781            assert!(
10782                state.crash_restarts.is_empty(),
10783                "an operator restart returns the full budget"
10784            );
10785            assert_eq!(
10786                state.lifetime_restarts, 2,
10787                "clearing the budget must not unmake the crashes"
10788            );
10789        }
10790
10791        assert!(
10792            matches!(
10793                on_child_exit(
10794                    &spec,
10795                    runtime.restart_policy,
10796                    &supervisor.registry,
10797                    &snapshot,
10798                    &runtime.terminal_ring,
10799                    &runtime.spawn_events,
10800                    &runtime.child_roster,
10801                    crash_exit_report(3),
10802                )
10803                .await,
10804                NextAction::Restart { schedule: _ }
10805            ),
10806            "the cleared budget must be spendable again"
10807        );
10808        let state = lock_snapshot(&snapshot).unwrap();
10809        assert_eq!(state.crash_restarts.len(), 1);
10810        assert_eq!(state.lifetime_restarts, 3);
10811    }
10812
10813    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10814    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
10815        let severed = ProcessIdentity {
10816            pid: 41,
10817            start_time: 101,
10818        };
10819        let successor = ProcessIdentity {
10820            pid: 41,
10821            start_time: 202,
10822        };
10823        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
10824        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
10825            state.pid = Some(successor.pid);
10826            state.process_start_time = Some(successor.start_time);
10827        })
10828        .unwrap();
10829        assert!(!module.record_deliberate_severance(severed).unwrap());
10830
10831        let exit_report = apply_deliberate_severance_marker(
10832            &module.inner.snapshot,
10833            Some(successor),
10834            ExitReport {
10835                kind: ExitKind::Crash,
10836                code: Some(1),
10837                signal: None,
10838                at_ms: 1,
10839            },
10840        );
10841
10842        assert_eq!(exit_report.kind, ExitKind::Crash);
10843    }
10844
10845    #[tokio::test]
10846    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
10847        let registry = Registry::default();
10848        let supervisor = Supervisor::new_for_test(
10849            Arc::new(Registry::default()),
10850            RestartPolicy::new(3, Duration::ZERO),
10851        );
10852        let runtime = supervisor.runtime_config();
10853        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10854        let spec = ModuleSpec {
10855            module_id: "drain-deliberate-severance".to_string(),
10856            program: fake_aft_stub_path(),
10857            args: Vec::new(),
10858            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10859            reserved: false,
10860            reserved_prefixes: Vec::new(),
10861            protocol: ModuleProtocol::Subc,
10862            overlap: Default::default(),
10863        };
10864        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10865        let process = ProcessIdentity {
10866            pid: 41,
10867            start_time: 101,
10868        };
10869        child.process_identity = Some(process);
10870        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10871            state.pid = Some(process.pid);
10872            state.process_start_time = Some(process.start_time);
10873        })
10874        .unwrap();
10875        record_deliberate_severance(&snapshot, process).unwrap();
10876
10877        drain_child_to_state(
10878            &spec.module_id,
10879            spec.protocol,
10880            // The child exits on its own; no signal may change the exit this
10881            // test classifies.
10882            StopNotice::SentOverConnection,
10883            &registry,
10884            None,
10885            &snapshot,
10886            &runtime.terminal_ring,
10887            &runtime.spawn_events,
10888            child,
10889            Duration::from_secs(1),
10890            ModuleState::Stopped,
10891            Some(false),
10892        )
10893        .await
10894        .unwrap();
10895
10896        let state = lock_snapshot(&snapshot).unwrap();
10897        assert_eq!(
10898            state.last_exit.as_ref().map(|exit| exit.kind),
10899            Some(ExitKind::DeliberateSeverance)
10900        );
10901        assert_eq!(state.lifetime_restarts, 1);
10902        assert_eq!(state.crash_restarts.len(), 0);
10903        drop(state);
10904        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10905        assert_eq!(
10906            history.entries[0].exit_kind,
10907            subc_control::TerminalExitKind::DeliberateSeverance
10908        );
10909    }
10910
10911    #[tokio::test]
10912    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
10913        let registry = Registry::default();
10914        let supervisor = Supervisor::new_for_test(
10915            Arc::new(Registry::default()),
10916            RestartPolicy::new(3, Duration::ZERO),
10917        );
10918        let runtime = supervisor.runtime_config();
10919        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10920        let spec = ModuleSpec {
10921            module_id: "ordinary-drain".to_string(),
10922            program: fake_aft_stub_path(),
10923            args: Vec::new(),
10924            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10925            reserved: false,
10926            reserved_prefixes: Vec::new(),
10927            protocol: ModuleProtocol::Subc,
10928            overlap: Default::default(),
10929        };
10930        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10931
10932        drain_child_to_state(
10933            &spec.module_id,
10934            spec.protocol,
10935            // The child exits on its own; no signal may change the exit this
10936            // test classifies.
10937            StopNotice::SentOverConnection,
10938            &registry,
10939            None,
10940            &snapshot,
10941            &runtime.terminal_ring,
10942            &runtime.spawn_events,
10943            child,
10944            Duration::from_secs(1),
10945            ModuleState::Stopped,
10946            Some(false),
10947        )
10948        .await
10949        .unwrap();
10950
10951        let state = lock_snapshot(&snapshot).unwrap();
10952        assert_eq!(
10953            state.last_exit.as_ref().map(|exit| exit.kind),
10954            Some(ExitKind::Crash)
10955        );
10956        assert_eq!(state.lifetime_restarts, 0);
10957        assert_eq!(state.crash_restarts.len(), 0);
10958    }
10959
10960    #[test]
10961    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
10962        // The server's generic fatal-routing branch only knows that the
10963        // connection failed; it does not know that the daemon deliberately
10964        // initiated a process-killing severance. Keep this seam explicit so a
10965        // future connection error path cannot silently reintroduce the stale
10966        // exemption that mislabels a later genuine crash.
10967        assert!(!include_str!("server.rs")
10968            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
10969    }
10970
10971    /// The `route.closed` `drained` value must be the quiescence wait's own
10972    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
10973    /// measurement at all and `false` is the one honest constant. This is the exact
10974    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
10975    /// on every return path, including the one that used to return early via `?`
10976    /// with `route.closing` already sent and no `route.closed` ever following.
10977    #[test]
10978    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
10979        assert!(drained_after_quiescence_wait(&Ok(true)));
10980        assert!(!drained_after_quiescence_wait(&Ok(false)));
10981        assert!(!drained_after_quiescence_wait(&Err(
10982            SuperviseError::StatePoisoned { module_id: None }
10983        )));
10984    }
10985
10986    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
10987    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
10988    /// already reaped out-of-band) still leaves a terminal record rather than none
10989    /// at all. Triggering the real `wait()` I/O error from an integration test would
10990    /// need a genuine already-reaped-child race, which is OS-specific and not
10991    /// something this suite attempts elsewhere; this test instead verifies the
10992    /// record produced for that arm end-to-end through the real `TerminalRing`, and
10993    /// the call site itself is verified by inspection to sit in that exact arm.
10994    #[test]
10995    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
10996        let ring = Arc::new(Mutex::new(TerminalRing::new(
10997            TerminalRingConfig::default(),
10998            0,
10999        )));
11000        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
11001
11002        let snapshot = ring.lock().unwrap().snapshot();
11003        assert_eq!(snapshot.entries.len(), 1);
11004        let entry = &snapshot.entries[0];
11005        assert_eq!(entry.exit_code, None);
11006        assert_eq!(entry.exit_signal, None);
11007        assert_eq!(entry.disposition, TerminalDisposition::Failed);
11008    }
11009
11010    #[test]
11011    fn wait_error_exit_path_preserves_spawn_event_density() {
11012        let feed = super::SpawnEventFeed::default();
11013        feed.configure_incarnation("wait-error-density".to_string());
11014        feed.emit_spawned("wait-error", 41, 1);
11015        let ring = Arc::new(Mutex::new(TerminalRing::new(
11016            TerminalRingConfig::default(),
11017            0,
11018        )));
11019
11020        record_wait_error_terminal("wait-error", &ring, &feed);
11021        feed.emit_spawned("after-wait-error", 42, 2);
11022
11023        let state = feed.0.lock().unwrap();
11024        let sequences = state
11025            .events
11026            .iter()
11027            .map(|event| event.cursor.seq)
11028            .collect::<Vec<_>>();
11029        assert_eq!(sequences, vec![1, 2, 3]);
11030        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11031        assert_eq!(state.events[1].exit_code, None);
11032        assert_eq!(state.events[1].exit_signal, None);
11033    }
11034
11035    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
11036    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
11037    /// not a clean exit it never actually observed.
11038    #[test]
11039    fn wait_error_exit_report_is_classified_as_a_crash() {
11040        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11041    }
11042}
11043
11044#[cfg(test)]
11045mod health_evidence_tests {
11046    use super::{HealthProbeError, HealthProbeEvidence};
11047    use std::collections::HashSet;
11048
11049    /// The evidential asymmetry, asserted rather than described.
11050    ///
11051    /// Exactly ONE observation is proof a module cannot serve, and the one that
11052    /// fires under CPU starvation is not it. Before the split, all fifteen
11053    /// construction sites collapsed into a single String, so a timeout carried the
11054    /// same weight as a dead lane -- which is how a healthy module was restarted
11055    /// three times in one day.
11056    #[test]
11057    fn only_a_dead_lane_is_proof_of_death() {
11058        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11059        // Three non-proof classes, each for a different reason: silence is
11060        // consistent with health, a bad answer proves the module ALIVE, and a
11061        // daemon-side fault never reached the module at all.
11062        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11063        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11064        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11065    }
11066
11067    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
11068    ///
11069    /// A shared label renders two different observations identically in the line an
11070    /// operator reads after an unexplained restart -- the exact confusion this
11071    /// change removes.
11072    #[test]
11073    fn every_evidence_class_has_a_distinct_label() {
11074        let labels = [
11075            HealthProbeError::lane_dead("").label(),
11076            HealthProbeError::no_answer("").label(),
11077            HealthProbeError::bad_answer("").label(),
11078            HealthProbeError::misconfigured("").label(),
11079        ];
11080        let unique: HashSet<_> = labels.iter().collect();
11081        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11082    }
11083
11084    /// The class is additional information, not a replacement.
11085    ///
11086    /// An operator needs both "this was silence" and the specific text saying how
11087    /// long we waited; a classification that swallowed the message would trade one
11088    /// missing distinction for another.
11089    #[test]
11090    fn classification_preserves_the_original_message() {
11091        let err = HealthProbeError::no_answer("module did not answer within 5s");
11092        assert_eq!(err.to_string(), "module did not answer within 5s");
11093        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11094    }
11095}
11096
11097#[cfg(test)]
11098mod health_tombstone_tests {
11099    use std::{path::PathBuf, sync::Arc, time::Duration};
11100
11101    use subc_protocol::{
11102        manifest::Concurrency,
11103        session::{HealthStatus, ModuleControlResponse},
11104    };
11105    use tokio::sync::mpsc;
11106
11107    use super::{
11108        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11109        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11110    };
11111    use crate::{
11112        control::ControlHandler,
11113        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11114        registry::{ConnectionId, Registry},
11115        router::FrameSink,
11116    };
11117
11118    struct ProbeHarness {
11119        spec: ModuleSpec,
11120        runtime: SupervisorRuntimeConfig,
11121        forwarding: Arc<ForwardingTable>,
11122        module_connection: ConnectionId,
11123        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11124        handler: ControlHandler,
11125        module: super::SupervisedModule,
11126    }
11127
11128    fn probe_harness() -> ProbeHarness {
11129        let registry = Arc::new(Registry::default());
11130        let forwarding = Arc::new(ForwardingTable::default());
11131        let supervisor_handle = super::SupervisorHandle::new();
11132        let health = HealthConfig {
11133            http: None,
11134            cadence: Duration::from_secs(30),
11135            deadline: Duration::from_secs(5),
11136            failure_threshold: 3,
11137            on_degraded: HealthAction::Report,
11138            on_failing: HealthAction::Report,
11139            critical: false,
11140        };
11141        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11142            .with_forwarding(Arc::clone(&forwarding))
11143            .with_handle(supervisor_handle.clone())
11144            .with_health_config(health);
11145        let spec = ModuleSpec {
11146            module_id: "late-health-module".to_string(),
11147            program: PathBuf::from("disabled-module"),
11148            args: Vec::new(),
11149            env: Vec::new(),
11150            reserved: false,
11151            reserved_prefixes: Vec::new(),
11152            protocol: ModuleProtocol::Subc,
11153            overlap: Default::default(),
11154        };
11155        let module = supervisor
11156            .supervise_configured(spec.clone(), false)
11157            .unwrap();
11158        let runtime = supervisor.runtime_config();
11159        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11160            .with_supervisor(supervisor_handle);
11161        let module_connection = ConnectionId::new(700);
11162        let (module_tx, module_rx) = mpsc::channel(8);
11163        forwarding
11164            .register_module_connection(
11165                module_connection,
11166                spec.module_id.clone(),
11167                subc_protocol::PROTOCOL_VERSION,
11168                Concurrency::ModuleManaged,
11169                FrameSink::new(module_tx),
11170            )
11171            .unwrap();
11172
11173        ProbeHarness {
11174            spec,
11175            runtime,
11176            forwarding,
11177            module_connection,
11178            module_rx,
11179            handler,
11180            module,
11181        }
11182    }
11183
11184    async fn finish_after(
11185        harness: &mut ProbeHarness,
11186        stall: Duration,
11187    ) -> ModuleControlRpcCompletion {
11188        assert!(stall > harness.runtime.health.deadline);
11189        let deadline = harness.runtime.health.deadline;
11190        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11191        let answer = async {
11192            let frame = harness.module_rx.recv().await.expect("health.check frame");
11193            tokio::time::advance(deadline).await;
11194            tokio::task::yield_now().await;
11195            tokio::time::advance(stall - deadline).await;
11196            harness
11197                .forwarding
11198                .complete_module_control_rpc(
11199                    harness.module_connection,
11200                    frame.header.corr,
11201                    Some("health.check"),
11202                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11203                        status: HealthStatus::Ok,
11204                        detail: None,
11205                        metrics: None,
11206                    }),
11207                )
11208                .unwrap()
11209        };
11210        let (probe_result, completion) = tokio::join!(probe, answer);
11211        let err = probe_result.expect_err("probe must miss its deadline");
11212        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11213        completion
11214    }
11215
11216    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11217        let deadline = harness.runtime.health.deadline;
11218        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11219        let exhaust_deadline = async {
11220            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11221            tokio::time::advance(deadline).await;
11222            tokio::task::yield_now().await;
11223        };
11224        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11225        let err = probe_result.expect_err("probe must miss its deadline");
11226        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11227    }
11228
11229    async fn run_probe_cycle(harness: &mut ProbeHarness, answer: bool) {
11230        let registry = Arc::clone(&harness.module.inner.registry);
11231        let snapshot = Arc::clone(&harness.module.inner.snapshot);
11232        let process_liveness = super::SupervisorProcessLiveness::default();
11233        let mut child = None;
11234        let cycle = super::run_health_probe_cycle(
11235            &harness.spec,
11236            &harness.runtime,
11237            &registry,
11238            &process_liveness,
11239            &snapshot,
11240            &mut child,
11241        );
11242        let peer = async {
11243            let frame = harness.module_rx.recv().await.expect("health.check frame");
11244            if answer {
11245                harness
11246                    .forwarding
11247                    .complete_module_control_rpc(
11248                        harness.module_connection,
11249                        frame.header.corr,
11250                        Some("health.check"),
11251                        ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11252                            status: HealthStatus::Ok,
11253                            detail: None,
11254                            metrics: Some(serde_json::json!({"ready": true})),
11255                        }),
11256                    )
11257                    .unwrap();
11258            } else {
11259                tokio::time::advance(harness.runtime.health.deadline).await;
11260                tokio::task::yield_now().await;
11261            }
11262        };
11263        tokio::join!(cycle, peer);
11264    }
11265
11266    #[tokio::test(start_paused = true)]
11267    async fn unanswered_probe_is_unknown_until_threshold_and_ok_report_recovers() {
11268        let mut harness = probe_harness();
11269        // Drive the probe cycle directly with an in-memory wire peer. Stop the
11270        // disabled module's monitor so only this test owns lifecycle transitions;
11271        // no OS process is launched, and a restart is observed at scheduling.
11272        let monitor = harness.module.inner.monitor.lock().unwrap().take().unwrap();
11273        monitor.abort();
11274        let _ = monitor.await;
11275        super::update_snapshot(&harness.module.inner.snapshot, None, |state| {
11276            state.enabled = true;
11277            state.state = super::ModuleState::Running;
11278            state.process_alive = true;
11279        })
11280        .unwrap();
11281
11282        run_probe_cycle(&mut harness, true).await;
11283        assert_eq!(
11284            harness.module.status().unwrap().health.status,
11285            super::SupervisorHealthStatus::Ok
11286        );
11287
11288        for failures in 1..harness.runtime.health.failure_threshold {
11289            run_probe_cycle(&mut harness, false).await;
11290            let status = harness.module.status().unwrap();
11291            assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11292            assert_eq!(status.health.consecutive_failures, failures);
11293            assert!(status.health.last_probe_ms.is_some());
11294            assert!(status.health.detail.unwrap().starts_with("[no-answer]"));
11295            assert!(status.health.metrics.is_none());
11296            assert_eq!(status.state, super::ModuleState::Running);
11297            assert!(status.process_alive);
11298            assert_eq!(status.restart_count, 0);
11299            assert_eq!(status.lifetime_restarts, 0);
11300            assert!(status.health.last_action.is_none());
11301            assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11302        }
11303
11304        run_probe_cycle(&mut harness, true).await;
11305        let recovered = harness.module.status().unwrap();
11306        assert_eq!(recovered.health.status, super::SupervisorHealthStatus::Ok);
11307        assert_eq!(recovered.health.consecutive_failures, 0);
11308        assert!(recovered.health.detail.is_none());
11309        assert_eq!(
11310            recovered.health.metrics,
11311            Some(serde_json::json!({"ready": true}))
11312        );
11313        assert_eq!(recovered.lifetime_restarts, 0);
11314
11315        for failures in 1..=harness.runtime.health.failure_threshold {
11316            run_probe_cycle(&mut harness, false).await;
11317            let status = harness.module.status().unwrap();
11318            assert_eq!(status.health.consecutive_failures, failures);
11319            if failures < harness.runtime.health.failure_threshold {
11320                assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11321                assert_eq!(status.state, super::ModuleState::Running);
11322                assert_eq!(status.lifetime_restarts, 0);
11323                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11324            } else {
11325                assert_eq!(
11326                    status.health.status,
11327                    super::SupervisorHealthStatus::Unresponsive
11328                );
11329                assert_eq!(status.state, super::ModuleState::Restarting);
11330                assert_eq!(status.restart_count, 1);
11331                assert_eq!(status.lifetime_restarts, 1);
11332                assert_eq!(status.health.last_action.as_deref(), Some("restart"));
11333                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_some());
11334            }
11335        }
11336    }
11337
11338    #[tokio::test(start_paused = true)]
11339    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
11340        let mut harness = probe_harness();
11341
11342        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
11343        let first_latency = match &first {
11344            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11345            other => panic!("late answer was not retained: {other:?}"),
11346        };
11347        assert!(harness.handler.observe_module_control_completion(first));
11348
11349        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
11350        let second_latency = match &second {
11351            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11352            other => panic!("late answer was not retained: {other:?}"),
11353        };
11354        assert!(harness.handler.observe_module_control_completion(second));
11355
11356        assert_eq!(first_latency, Duration::from_secs(8));
11357        assert_eq!(
11358            second_latency - first_latency,
11359            Duration::from_secs(3),
11360            "latency must grow linearly with the additional stall"
11361        );
11362        let health = harness.module.status().unwrap().health;
11363        assert_eq!(health.late_answer_count, 2);
11364        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
11365    }
11366
11367    /// A module that answers every probe late must never march to the kill
11368    /// threshold: the late answer proves it is alive, so it must clear the miss
11369    /// streak the timeout recorded. Without the reset, a CPU-starved module
11370    /// that serves every probe seconds past the deadline accumulates
11371    /// `consecutive_failures` to the threshold and is killed — the exact
11372    /// sequence from the 2026-08-14 aft disable, where the daemon logged
11373    /// "proves the module is alive" five times while counting five misses.
11374    #[tokio::test(start_paused = true)]
11375    async fn late_answer_clears_the_consecutive_failure_streak() {
11376        let mut harness = probe_harness();
11377
11378        // Timeout recorded first: the probe path saw no answer in time.
11379        time_out_without_answer(&mut harness).await;
11380        harness
11381            .module
11382            .record_health_probe_failure_for_test("[no-answer] test miss")
11383            .unwrap();
11384        assert_eq!(
11385            harness.module.status().unwrap().health.consecutive_failures,
11386            1,
11387            "precondition: the miss must be on the streak before the late answer"
11388        );
11389
11390        // The stalled reply then lands: proof of life.
11391        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
11392        assert!(matches!(
11393            late,
11394            ModuleControlRpcCompletion::LateHealthAnswer { .. }
11395        ));
11396        assert!(harness.handler.observe_module_control_completion(late));
11397
11398        let health = harness.module.status().unwrap().health;
11399        assert_eq!(
11400            health.consecutive_failures, 0,
11401            "a late answer is an answer: the streak must reset"
11402        );
11403        assert_eq!(health.late_answer_count, 1);
11404    }
11405
11406    #[tokio::test(start_paused = true)]
11407    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
11408        let mut harness = probe_harness();
11409
11410        for _ in 0..20 {
11411            time_out_without_answer(&mut harness).await;
11412            assert_eq!(
11413                harness.forwarding.health_probe_tombstone_count().unwrap(),
11414                1
11415            );
11416        }
11417    }
11418}
11419
11420#[cfg(test)]
11421mod child_env_tests {
11422    use super::{
11423        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
11424        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
11425        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
11426    };
11427    use std::{ffi::OsStr, path::PathBuf};
11428    use tokio::process::Command;
11429
11430    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
11431        ModuleSpec {
11432            module_id: "env-plan".to_string(),
11433            program: PathBuf::from("/nonexistent"),
11434            args: Vec::new(),
11435            env,
11436            reserved: false,
11437            reserved_prefixes: Vec::new(),
11438            protocol: ModuleProtocol::Subc,
11439            overlap: Default::default(),
11440        }
11441    }
11442
11443    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
11444    /// one still gets its own.
11445    ///
11446    /// This is the narrow goal `env_clear()` was reached for, and the reason the
11447    /// fix is `env_remove` rather than deleting the line: an operator's ambient
11448    /// filter silently becoming an unconfigured module's log level is a real
11449    /// defect, just a much smaller one than clearing the environment.
11450    ///
11451    /// Asserted on the command plan rather than a spawned child because proving
11452    /// the ABSENCE of an inherited variable needs the parent's environment
11453    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
11454    /// removal as `(key, None)`, which is exactly the distinction wanted: not
11455    /// "absent because nobody set it" but "explicitly unset for the child".
11456    #[test]
11457    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
11458        let mut command = Command::new("/nonexistent");
11459        apply_child_env(&mut command, &spec(Vec::new()));
11460        let removed = command
11461            .as_std()
11462            .get_envs()
11463            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
11464        assert!(
11465            removed,
11466            "ambient CK_LOG must be explicitly removed for an unconfigured module"
11467        );
11468
11469        let mut configured = Command::new("/nonexistent");
11470        apply_child_env(
11471            &mut configured,
11472            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
11473        );
11474        let effective = configured
11475            .as_std()
11476            .get_envs()
11477            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
11478            .last()
11479            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11480        assert_eq!(
11481            effective,
11482            Some(Some("debug".to_string())),
11483            "a module's configured CK_LOG must survive the ambient removal"
11484        );
11485    }
11486
11487    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
11488    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
11489    /// the same reason as the CK_LOG test above.
11490    ///
11491    /// The argument is the load-bearing half: a stock binary exits on an
11492    /// unknown flag before it listens, so with `--subc` appended the mode
11493    /// could not supervise the one process it exists for. Found by the first
11494    /// conformance run (nats-server: `flag provided but not defined: -subc`).
11495    #[test]
11496    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
11497        let connection_file = std::path::Path::new("/run/subc-connection.json");
11498        let handle = SupervisorHandle::new();
11499
11500        let mut none_spec = spec(Vec::new());
11501        none_spec.protocol = ModuleProtocol::None;
11502        let mut none = Command::new("/nonexistent");
11503        let none_handoff =
11504            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
11505                .expect("protocol-none spawn args apply");
11506        assert!(
11507            none_handoff.is_none(),
11508            "protocol:none spawn must not receive a nonce descriptor"
11509        );
11510        assert!(
11511            !none.as_std().get_envs().any(|(key, value)| key
11512                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
11513                && value.is_some()),
11514            "protocol:none spawn must not name a nonce descriptor"
11515        );
11516        let none_args: Vec<String> = none
11517            .as_std()
11518            .get_args()
11519            .map(|a| a.to_string_lossy().into_owned())
11520            .collect();
11521        assert!(
11522            !none_args.iter().any(|a| a == SUBC_ARG),
11523            "protocol:none argv must not carry --subc; got {none_args:?}"
11524        );
11525        let none_has_nonce = none
11526            .as_std()
11527            .get_envs()
11528            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
11529        assert!(
11530            !none_has_nonce,
11531            "protocol:none spawn must not receive a launch nonce"
11532        );
11533        let none_has_module_id = none
11534            .as_std()
11535            .get_envs()
11536            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
11537        assert!(
11538            none_has_module_id,
11539            "SUBC_MODULE_ID is inert and stays on every path"
11540        );
11541        assert!(
11542            handle.spawn_nonce(&none_spec.module_id).is_none(),
11543            "no nonce record for a process that will never present one"
11544        );
11545
11546        // Control: the subc-wire path is unchanged by the branch above.
11547        let wire_spec = spec(Vec::new());
11548        let mut wire = Command::new("/nonexistent");
11549        let wire_handoff =
11550            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
11551                .expect("subc-wire spawn args apply");
11552        let wire_fd_env = wire
11553            .as_std()
11554            .get_envs()
11555            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
11556            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11557        #[cfg(unix)]
11558        assert_eq!(
11559            wire_fd_env,
11560            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11561            "a subc-wire spawn names the pipe it will receive at descriptor 3"
11562        );
11563        #[cfg(not(unix))]
11564        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11565        let wire_args: Vec<String> = wire
11566            .as_std()
11567            .get_args()
11568            .map(|a| a.to_string_lossy().into_owned())
11569            .collect();
11570        assert_eq!(
11571            wire_args,
11572            vec![
11573                SUBC_ARG.to_string(),
11574                connection_file.to_string_lossy().into_owned()
11575            ],
11576            "a subc-wire spawn still carries --subc <path>"
11577        );
11578        assert_eq!(
11579            wire.as_std()
11580                .get_envs()
11581                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11582            !cfg!(unix),
11583            "only Windows supplies the environment nonce"
11584        );
11585        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11586    }
11587
11588    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
11589    /// spec tries to set it; only a swap candidate carries it.
11590    ///
11591    /// "Set it only on candidates" is not enough, because spawn applies the
11592    /// spec's env verbatim and the daemon's own environment is inherited: either
11593    /// could hand a plain restart the swap role, and a module reading it would
11594    /// warm on its long swap budget while callers wait. Asserted as an explicit
11595    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
11596    /// test above gives.
11597    #[test]
11598    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11599        let role = |command: &Command| {
11600            command
11601                .as_std()
11602                .get_envs()
11603                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11604                .last()
11605                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11606        };
11607        let forged = spec(vec![(
11608            SUBC_SPAWN_ROLE_ENV.to_string(),
11609            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11610        )]);
11611
11612        let mut plain = Command::new("/nonexistent");
11613        apply_child_env(&mut plain, &forged);
11614        apply_spawn_role(&mut plain, SpawnRole::Plain);
11615        assert_eq!(
11616            role(&plain),
11617            Some(None),
11618            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11619        );
11620
11621        let mut candidate = Command::new("/nonexistent");
11622        apply_child_env(&mut candidate, &spec(Vec::new()));
11623        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11624        assert_eq!(
11625            role(&candidate),
11626            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11627        );
11628    }
11629
11630    /// Daemon-private capture retention keys never reach the child.
11631    ///
11632    /// cortexkit-log exposes retention as a Rust struct with no environment
11633    /// names, so these entries are supervisor metadata. Passing them through
11634    /// would invent a public child-process contract by accident.
11635    #[test]
11636    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11637        let mut command = Command::new("/nonexistent");
11638        apply_child_env(
11639            &mut command,
11640            &spec(vec![
11641                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11642                ("KEPT".to_string(), "yes".to_string()),
11643            ]),
11644        );
11645        let keys: Vec<String> = command
11646            .as_std()
11647            .get_envs()
11648            .filter(|(_, value)| value.is_some())
11649            .map(|(key, _)| key.to_string_lossy().into_owned())
11650            .collect();
11651        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11652        assert!(
11653            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11654            "daemon-private capture key leaked to the child: {keys:?}"
11655        );
11656    }
11657}
11658
11659#[cfg(test)]
11660mod jitter_tests {
11661    use super::jittered_health_delay;
11662    use std::{collections::HashSet, time::Duration};
11663
11664    /// Module ids drawn from a real fleet, so the dispersal claim is about names
11665    /// that actually occur rather than invented ones.
11666    ///
11667    /// This is a SAMPLE, not a registry: the property under test is that distinct
11668    /// ids disperse, which holds for any set of distinct strings. Several entries
11669    /// are already historical (modules get renamed), and that costs nothing here --
11670    /// but it means a reader must not mistake this for the live module set, and a
11671    /// rename sweep will match it without there being anything to change.
11672    const FLEET: [&str; 14] = [
11673        "aft",
11674        "alfonso-core",
11675        "magic-context",
11676        "broca",
11677        "thalamus",
11678        "quota",
11679        "engram",
11680        "plexus",
11681        "cerebellum",
11682        "astrocyte",
11683        "synapse",
11684        "subc-mcp",
11685        "cortexkit-credentials",
11686        "subc-federation",
11687    ];
11688
11689    /// Probes must not converge after a fleet-wide restart.
11690    ///
11691    /// This is the property the jitter exists for: every module reconnects at
11692    /// once, and without dispersal all fourteen would then probe on the same
11693    /// tick forever. Nothing failed visibly when this went untested -- a
11694    /// convergent fleet still probes correctly, just in a burst, so the symptom
11695    /// is a periodic load spike that looks like whatever else is running.
11696    #[test]
11697    fn probe_delays_disperse_across_the_fleet() {
11698        let cadence = Duration::from_secs(30);
11699        let delays: HashSet<Duration> = FLEET
11700            .iter()
11701            .map(|id| jittered_health_delay(id, 0, cadence))
11702            .collect();
11703        assert_eq!(
11704            delays.len(),
11705            FLEET.len(),
11706            "every supervised module must land on its own probe offset"
11707        );
11708    }
11709
11710    /// The offset may only ever DELAY a probe, never bring it forward.
11711    ///
11712    /// A delay below the cadence would probe a module more often than
11713    /// configured, which is the opposite of what an operator asked for and
11714    /// would tighten the failure budget without anyone changing it.
11715    #[test]
11716    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11717        let cadence = Duration::from_secs(30);
11718        let span = cadence / 10;
11719        for id in FLEET {
11720            for probe_index in 0..8 {
11721                let delay = jittered_health_delay(id, probe_index, cadence);
11722                assert!(
11723                    delay >= cadence,
11724                    "{id}#{probe_index}: jitter must not shorten the cadence"
11725                );
11726                assert!(
11727                    delay < cadence + span,
11728                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11729                );
11730            }
11731        }
11732    }
11733
11734    /// A module keeps its offset across daemon restarts.
11735    ///
11736    /// The delay is derived rather than randomised precisely so a restart does
11737    /// not re-roll every module into a fresh chance of collision. A random
11738    /// source would satisfy the dispersal test above and quietly lose this.
11739    #[test]
11740    fn a_module_offset_is_stable_across_restarts() {
11741        let cadence = Duration::from_secs(30);
11742        for id in FLEET {
11743            assert_eq!(
11744                jittered_health_delay(id, 0, cadence),
11745                jittered_health_delay(id, 0, cadence),
11746                "{id}: the same module and probe index must produce the same offset"
11747            );
11748        }
11749    }
11750
11751    /// A zero cadence disables probing rather than producing a busy loop.
11752    #[test]
11753    fn zero_cadence_yields_zero_delay() {
11754        assert_eq!(
11755            jittered_health_delay("aft", 0, Duration::ZERO),
11756            Duration::ZERO
11757        );
11758    }
11759}
11760
11761#[cfg(all(test, target_os = "linux"))]
11762mod cgroup_placement_tests {
11763    use super::{
11764        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11765        SupervisedChild,
11766    };
11767    use crate::stderr_tail::{StderrRing, StderrTailConfig};
11768    use std::{
11769        fs, io,
11770        path::{Path, PathBuf},
11771        sync::{Arc, Mutex},
11772    };
11773    use subc_test_support::TestTempDir;
11774    use tokio::process::Command;
11775
11776    #[tokio::test]
11777    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11778        use super::*;
11779        let dir = TestTempDir::new("unique-spawn-cgroups");
11780        let root = PathBuf::from(format!(
11781            "/sys/fs/cgroup/subc-unique-test-{}-{}",
11782            std::process::id(),
11783            unix_ms_now()
11784        ));
11785        if let Err(error) = fs::create_dir(&root) {
11786            assert!(
11787                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11788                "required cgroup test cannot execute: {error}"
11789            );
11790            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11791            return;
11792        }
11793        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11794        let group_count = || {
11795            fs::read_dir(root.join("subc-modules"))
11796                .unwrap()
11797                .map(|entry| entry.unwrap().file_type().unwrap())
11798                .filter(|kind| kind.is_dir())
11799                .count()
11800        };
11801        let supervisor =
11802            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
11803                .with_cgroup_placement(Some(placement.clone()));
11804        let runtime = supervisor.runtime_config();
11805        let mut spec = ModuleSpec {
11806            module_id: "unique-spawn".into(),
11807            program: PathBuf::from("/bin/sleep"),
11808            args: vec!["60".into()],
11809            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11810                .into_iter()
11811                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
11812                .collect(),
11813            reserved: false,
11814            reserved_prefixes: vec![],
11815            protocol: ModuleProtocol::None,
11816            overlap: Default::default(),
11817        };
11818        let spawn = |spec: &ModuleSpec| {
11819            spawn_child(
11820                spec,
11821                None,
11822                None,
11823                &runtime.stderr_ring,
11824                None,
11825                &runtime.child_roster,
11826                Some(&placement),
11827            )
11828            .unwrap()
11829        };
11830        let mut live = spawn(&spec);
11831        for _ in 0..3 {
11832            // A new process can enter the old slot while retirement is pending.
11833            let next = spawn(&spec);
11834            assert_ne!(live.module_id, next.module_id);
11835            live.start_kill().unwrap();
11836            live.wait().await.unwrap();
11837            live = next;
11838            assert!(
11839                live.child.try_wait().unwrap().is_none(),
11840                "retiring the old slot must not kill the replacement"
11841            );
11842            assert_eq!(
11843                group_count(),
11844                1,
11845                "only the live spawn's cgroup should remain"
11846            );
11847        }
11848        supervisor.begin_daemon_shutdown();
11849        let reap = tokio::spawn(async move {
11850            live.wait().await.unwrap();
11851        });
11852        supervisor
11853            .end_children_for_daemon_shutdown(false, std::future::pending())
11854            .await;
11855        reap.await.unwrap();
11856        assert_eq!(group_count(), 0);
11857        // A normal exit uses the same tree-cleanup path as a killed spawn.
11858        spec.program = PathBuf::from("/bin/true");
11859        spec.args.clear();
11860        let fresh_roster = ChildRoster::default();
11861        let mut short = spawn_child(
11862            &spec,
11863            None,
11864            None,
11865            &runtime.stderr_ring,
11866            None,
11867            &fresh_roster,
11868            Some(&placement),
11869        )
11870        .unwrap();
11871        short.wait().await.unwrap();
11872        assert_eq!(group_count(), 0);
11873        spec.module_id = "_".repeat(255);
11874        let mut long_id = spawn_child(
11875            &spec,
11876            None,
11877            None,
11878            &runtime.stderr_ring,
11879            None,
11880            &fresh_roster,
11881            Some(&placement),
11882        )
11883        .unwrap();
11884        long_id.wait().await.unwrap();
11885        assert_eq!(
11886            group_count(),
11887            0,
11888            "valid long module IDs must not exceed cgroup NAME_MAX"
11889        );
11890        fs::remove_dir(root.join("subc-modules")).unwrap();
11891        fs::remove_dir(root).unwrap();
11892        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
11893    }
11894
11895    #[test]
11896    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
11897        let path = Path::new("/definitely-missing-subc-cgroup");
11898        let mut command = Command::new("true");
11899        let error = apply_cgroup_placement(
11900            &mut command,
11901            &ModuleSpec {
11902                module_id: "broken-cgroup".to_string(),
11903                program: PathBuf::from("true"),
11904                args: Vec::new(),
11905                env: Vec::new(),
11906                reserved: false,
11907                reserved_prefixes: Vec::new(),
11908                protocol: ModuleProtocol::Subc,
11909                overlap: Default::default(),
11910            },
11911            path,
11912        )
11913        .expect_err("a parent cgroup open failure must reject the supervised spawn");
11914        let reason = error.to_string();
11915
11916        assert!(
11917            matches!(error, SuperviseError::Cgroup { .. }),
11918            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
11919        );
11920        assert!(
11921            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
11922            "parent cgroup open failure must name cgroup.procs: {reason}"
11923        );
11924    }
11925
11926    #[tokio::test]
11927    async fn reaping_a_child_removes_its_empty_module_cgroup() {
11928        let root = TestTempDir::new("supervisor-reap-cgroup");
11929        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11930        let placement = subc_cgroup::prepare_at(&root)
11931            .expect("prepare scratch cgroup root")
11932            .expect("scratch root has a cgroup.procs marker");
11933        let module_id = "reaped-module";
11934        let module = placement
11935            .module_path(module_id)
11936            .expect("create scratch module cgroup");
11937        let child = Command::new("true")
11938            .env("XDG_DATA_HOME", root.path())
11939            .env("XDG_RUNTIME_DIR", root.path())
11940            .env("XDG_CONFIG_HOME", root.path())
11941            .spawn()
11942            .expect("spawn short-lived child");
11943        let pid = child.id().expect("spawned child has pid");
11944        let mut child = SupervisedChild {
11945            child,
11946            protocol: ModuleProtocol::Subc,
11947            module_id: module_id.to_string(),
11948            cgroup_placement: Some(placement),
11949            stdout_pump: None,
11950            stderr_pump: None,
11951            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
11952            spawned_at_ms: 0,
11953            spawned_from: PathBuf::from("true"),
11954            spawned_file_identity: None,
11955            process_start_time: None,
11956            process_identity: None,
11957            pid,
11958            roster_guard: None,
11959            #[cfg(target_os = "macos")]
11960            privacy_exec: None,
11961            spawn_failure: None,
11962        };
11963
11964        child.wait().await.expect("reap short-lived child");
11965
11966        assert!(
11967            !module.exists(),
11968            "reaping the supervised child must remove its empty cgroup"
11969        );
11970    }
11971
11972    #[test]
11973    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
11974        let root = TestTempDir::new("supervisor-non-empty-cgroup");
11975        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11976        let placement = subc_cgroup::prepare_at(&root)
11977            .expect("prepare scratch cgroup root")
11978            .expect("scratch root has a cgroup.procs marker");
11979        let module = placement
11980            .module_path("surviving-module")
11981            .expect("create scratch module cgroup");
11982        fs::write(module.join("surviving-process"), b"still present")
11983            .expect("make scratch cgroup non-empty");
11984        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
11985
11986        remove_module_cgroup(&placement, "surviving-module");
11987
11988        let logs = crate::router::test_log::captured_logs(&logs);
11989        assert!(
11990            module.exists(),
11991            "failed removal must leave the cgroup intact"
11992        );
11993        assert!(
11994            logs.contains("could not remove module cgroup after process exit; continuing teardown")
11995                && logs.contains("surviving-module"),
11996            "best-effort removal must report the failure without returning it: {logs}"
11997        );
11998    }
11999
12000    #[test]
12001    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
12002        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
12003        let reason = SuperviseError::Spawn {
12004            program: PathBuf::from("/bin/true"),
12005            source: io::Error::from_raw_os_error(13),
12006            cgroup_path: Some(cgroup_path.clone()),
12007        }
12008        .to_string();
12009
12010        assert!(
12011            reason.contains(&cgroup_path.display().to_string()),
12012            "a pre_exec spawn failure must name the cgroup path: {reason}"
12013        );
12014    }
12015}
12016
12017#[cfg(test)]
12018mod spawn_subscriber_lag_tests {
12019    use super::*;
12020
12021    /// A subscriber whose connection stops draining is dropped once its frame
12022    /// channel fills. The client must learn that from a terminal Error frame
12023    /// after the frames already queued for it, not from a stream that simply
12024    /// goes quiet.
12025    #[tokio::test]
12026    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
12027        let feed = SpawnEventFeed::default();
12028        feed.configure_incarnation("lag-incarnation".to_string());
12029        // A one-slot connection queue that nobody reads until the emits are
12030        // done: the forwarder parks on it and the subscriber channel fills.
12031        let (tx, mut rx) = mpsc::channel(1);
12032        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
12033            .expect("subscribe");
12034        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
12035        for index in 0..emitted {
12036            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
12037            // Let the forwarder take what it can so the fill point is the
12038            // subscriber channel, not a scheduling accident.
12039            tokio::task::yield_now().await;
12040        }
12041        assert_eq!(
12042            feed.subscriber_count(),
12043            0,
12044            "the lagged subscriber must be removed"
12045        );
12046
12047        let mut data = Vec::new();
12048        let mut last = None;
12049        loop {
12050            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
12051                .await
12052                .expect("the forwarder must finish once the subscriber is dropped");
12053            let Some(outbound) = next else { break };
12054            let frame = outbound.frame;
12055            if frame.header.ty == FrameType::StreamData {
12056                assert!(last.is_none(), "no data may follow the terminal frame");
12057                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
12058                data.push(event.cursor.seq);
12059            } else {
12060                assert!(last.is_none(), "exactly one terminal frame");
12061                last = Some(frame);
12062            }
12063        }
12064        assert!(!data.is_empty(), "queued frames drain before the terminal");
12065        for pair in data.windows(2) {
12066            assert_eq!(
12067                pair[1],
12068                pair[0] + 1,
12069                "queued frames arrive dense and in order"
12070            );
12071        }
12072        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
12073        assert_eq!(terminal.header.ty, FrameType::Error);
12074        assert_eq!(terminal.header.corr, 7);
12075        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
12076        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
12077        let detail = body.detail.expect("lagged error carries detail");
12078        assert_eq!(
12079            detail["first_undelivered_cursor"]["seq"],
12080            data.last().unwrap() + 1,
12081            "the named cursor is the first event the subscriber did not receive"
12082        );
12083        assert_eq!(
12084            detail["first_undelivered_cursor"]["daemon_incarnation"],
12085            "lag-incarnation"
12086        );
12087    }
12088}
12089
12090#[cfg(test)]
12091mod terminal_history_read_concurrency_tests {
12092    use super::*;
12093    use crate::terminal_journal::read_pause;
12094    use std::sync::mpsc as std_mpsc;
12095    use subc_test_support::TestTempDir;
12096
12097    fn journaled_ring(
12098        journal: &Arc<crate::terminal_journal::TerminalJournal>,
12099    ) -> Arc<Mutex<TerminalRing>> {
12100        Arc::new(Mutex::new(
12101            TerminalRing::new(TerminalRingConfig::default(), 1)
12102                .with_journal(Some(Arc::clone(journal))),
12103        ))
12104    }
12105
12106    fn crash(at_ms: u64) -> ExitReport {
12107        ExitReport {
12108            kind: ExitKind::Crash,
12109            code: Some(1),
12110            signal: None,
12111            at_ms,
12112        }
12113    }
12114
12115    /// Record an exit on another thread and report whether it finished within
12116    /// `bound`. The recorder thread is left running if it did not.
12117    fn record_within(
12118        module_id: &'static str,
12119        ring: &Arc<Mutex<TerminalRing>>,
12120        at_ms: u64,
12121        bound: Duration,
12122    ) -> bool {
12123        let ring = Arc::clone(ring);
12124        let (done, done_rx) = std_mpsc::channel();
12125        std::thread::spawn(move || {
12126            record_terminal(
12127                module_id,
12128                &ring,
12129                &SpawnEventFeed::default(),
12130                &crash(at_ms),
12131                TerminalDisposition::Restarting,
12132            );
12133            let _ = done.send(());
12134        });
12135        done_rx.recv_timeout(bound).is_ok()
12136    }
12137
12138    /// A history read in progress must not hold the journal writer (which every
12139    /// module's exit recording needs) or the module's own ring. Exits recorded
12140    /// while the read is paused complete promptly; the paused read answers as of
12141    /// the moment it started, and the next read has each exit exactly once.
12142    #[test]
12143    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12144        let dir = TestTempDir::new("terminal-history-concurrent-read");
12145        let path = dir.join("terminals.jsonl");
12146        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12147            path.clone(),
12148            "daemon".into(),
12149        ));
12150        let reader_ring = journaled_ring(&journal);
12151        let other_ring = journaled_ring(&journal);
12152        assert!(record_within(
12153            "reader-module",
12154            &reader_ring,
12155            10,
12156            Duration::from_secs(5)
12157        ));
12158
12159        let (started, release) = read_pause::install(&path);
12160        let reading = {
12161            let ring = Arc::clone(&reader_ring);
12162            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12163        };
12164        started
12165            .recv_timeout(Duration::from_secs(5))
12166            .expect("the history read reached its pause");
12167
12168        let bound = Duration::from_secs(1);
12169        assert!(
12170            record_within("other-module", &other_ring, 20, bound),
12171            "another module's exit waited on a history read (journal writer held)"
12172        );
12173        assert!(
12174            record_within("reader-module", &reader_ring, 30, bound),
12175            "the read module's own exit waited on its history read (ring held)"
12176        );
12177
12178        drop(release);
12179        let paused = reading.join().unwrap();
12180        assert_eq!(
12181            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12182            vec![10],
12183            "an exit recorded after the read began lands in neither half of it"
12184        );
12185        assert_eq!(paused.journal_skipped_lines, 0);
12186        assert_eq!(paused.journal_read_errors, 0);
12187
12188        let after = durable_terminal_history_of(&reader_ring, "reader-module");
12189        assert_eq!(
12190            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12191            vec![10, 30],
12192            "the next read merges ring and journal with no duplicate"
12193        );
12194        assert_eq!(after.journal_skipped_lines, 0);
12195    }
12196}
12197
12198/// What a restart does with the exited process's stderr reader. These drive
12199/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12200/// holds, so a reader that has not been scheduled by the bound is a controlled
12201/// input rather than something only a loaded machine produces.
12202#[cfg(test)]
12203mod stderr_settle_tests {
12204    use std::{
12205        future::Future,
12206        io,
12207        pin::Pin,
12208        sync::{Arc, Mutex},
12209        task::{Context, Poll},
12210        time::Duration,
12211    };
12212
12213    use tokio::{
12214        io::{AsyncRead, ReadBuf},
12215        sync::oneshot,
12216        time::Instant,
12217    };
12218
12219    use super::{settle_stderr_pump, StderrPump};
12220    use crate::stderr_tail::{
12221        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12222    };
12223
12224    const BOUND: Duration = Duration::from_millis(250);
12225
12226    /// Yields `before`, then stays pending until the gate is released, then
12227    /// yields `after` and reaches EOF. The bytes after the gate were written
12228    /// by a process that has already exited; only the reader is behind.
12229    struct HeldReader {
12230        before: Option<Vec<u8>>,
12231        gate: Option<oneshot::Receiver<()>>,
12232        after: io::Cursor<Vec<u8>>,
12233    }
12234
12235    impl AsyncRead for HeldReader {
12236        fn poll_read(
12237            mut self: Pin<&mut Self>,
12238            cx: &mut Context<'_>,
12239            buf: &mut ReadBuf<'_>,
12240        ) -> Poll<io::Result<()>> {
12241            if let Some(bytes) = self.before.take() {
12242                buf.put_slice(&bytes);
12243                return Poll::Ready(Ok(()));
12244            }
12245            if let Some(gate) = self.gate.as_mut() {
12246                match Pin::new(gate).poll(cx) {
12247                    Poll::Pending => return Poll::Pending,
12248                    Poll::Ready(_) => self.gate = None,
12249                }
12250            }
12251            Pin::new(&mut self.after).poll_read(cx, buf)
12252        }
12253    }
12254
12255    struct DiscardSink;
12256
12257    impl OutputSink for DiscardSink {
12258        fn write_line(&mut self, _line: &[u8]) {}
12259    }
12260
12261    fn line(text: &str) -> TailEntry {
12262        TailEntry::Line {
12263            text: text.to_string(),
12264            truncated: false,
12265            at_ms: None,
12266        }
12267    }
12268
12269    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
12270        ring.lock().unwrap()
12271    }
12272
12273    /// Start a reader for a new process generation that delivers `before`
12274    /// immediately and `after` only once the returned sender fires (or is
12275    /// dropped).
12276    fn held_pump(
12277        ring: &Arc<Mutex<StderrRing>>,
12278        before: &str,
12279        after: &str,
12280    ) -> (StderrPump, oneshot::Sender<()>) {
12281        let generation = lock(ring).begin_process();
12282        let (release, gate) = oneshot::channel();
12283        let reader = HeldReader {
12284            before: Some(before.as_bytes().to_vec()),
12285            gate: Some(gate),
12286            after: io::Cursor::new(after.as_bytes().to_vec()),
12287        };
12288        let task = tokio::spawn(pump_stderr_to(
12289            reader,
12290            Arc::clone(ring),
12291            generation,
12292            DiscardSink,
12293        ));
12294        (StderrPump { task, generation }, release)
12295    }
12296
12297    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
12298        for _ in 0..1000 {
12299            if done(&lock(ring)) {
12300                return;
12301            }
12302            tokio::time::sleep(Duration::from_millis(1)).await;
12303        }
12304        panic!(
12305            "ring never reached the expected state: {:?}",
12306            lock(ring).snapshot(None, None)
12307        );
12308    }
12309
12310    #[tokio::test(start_paused = true)]
12311    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
12312        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12313        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
12314
12315        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
12316        let before_release = lock(&ring).snapshot(None, None);
12317        assert!(
12318            matches!(before_release.capture, CaptureState::Incomplete { .. }),
12319            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
12320        );
12321
12322        // The restart: the next process starts and writes before the old
12323        // reader catches up.
12324        let next = lock(&ring).begin_process();
12325        lock(&ring).push_line_from(next, "next process booting");
12326        release.send(()).unwrap();
12327        wait_until(&ring, |ring| {
12328            ring.snapshot(None, None).capture == CaptureState::Captured
12329        })
12330        .await;
12331
12332        assert_eq!(
12333            untimed(lock(&ring).snapshot(None, None).entries),
12334            vec![
12335                line("booting"),
12336                line("config error: missing storage"),
12337                TailEntry::ProcessStart,
12338                line("next process booting"),
12339            ],
12340            "the crash's last line must survive a slow reader and stay in the crashed process's section"
12341        );
12342    }
12343
12344    #[tokio::test(start_paused = true)]
12345    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
12346    ) {
12347        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12348        // `_held` is never fired: a descendant keeps the pipe open for the
12349        // whole test.
12350        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
12351
12352        let started = Instant::now();
12353        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
12354        assert_eq!(
12355            started.elapsed(),
12356            BOUND,
12357            "the restart must wait exactly the bound for a pipe that stays open, no longer"
12358        );
12359
12360        let next = lock(&ring).begin_process();
12361        lock(&ring).push_line_from(next, "next process booting");
12362        tokio::time::sleep(Duration::from_secs(60)).await;
12363
12364        let snapshot = lock(&ring).snapshot(None, None);
12365        match &snapshot.capture {
12366            CaptureState::Incomplete { reason } => assert!(
12367                reason.contains("had not reached EOF") && reason.contains("250ms"),
12368                "the reason must say what is missing and after how long: {reason}"
12369            ),
12370            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
12371        }
12372        assert_eq!(
12373            untimed(snapshot.entries),
12374            vec![
12375                line("parent exiting"),
12376                TailEntry::ProcessStart,
12377                line("next process booting"),
12378            ]
12379        );
12380    }
12381
12382    #[tokio::test(start_paused = true)]
12383    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
12384        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12385        let (pump, release) = held_pump(&ring, "one\n", "two\n");
12386        release.send(()).unwrap();
12387
12388        settle_stderr_pump("clean", &ring, pump, BOUND).await;
12389
12390        let snapshot = lock(&ring).snapshot(None, None);
12391        assert_eq!(snapshot.capture, CaptureState::Captured);
12392        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
12393    }
12394}
12395
12396/// Containment of a module's process tree (issue #109).
12397///
12398/// The behaviour these defend against is a module helper surviving its module:
12399/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
12400/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
12401/// compounds it.
12402///
12403/// They run against the SUPERVISOR rather than the job-object crate because the
12404/// claim is about teardown: a crate-level test proves a job can reap a tree, not
12405/// that the daemon's drain path reaches it.
12406///
12407/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
12408/// lane there is a separate containment path with its own tests.
12409#[cfg(all(test, windows))]
12410mod job_containment_tests {
12411    use super::*;
12412    use std::{
12413        path::{Path, PathBuf},
12414        sync::{Arc, Mutex},
12415        time::{Duration, Instant},
12416    };
12417    use subc_test_support::TestTempDir;
12418
12419    /// The stub, expected beside this test executable.
12420    ///
12421    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
12422    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
12423    /// failure then reads as a broken test rather than an unbuilt dependency.
12424    fn stub_path() -> PathBuf {
12425        let mut path = std::env::current_exe().expect("current_exe available in tests");
12426        path.pop();
12427        path.pop();
12428        path.push("fake-aft-stub.exe");
12429        assert!(
12430            path.exists(),
12431            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
12432             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
12433            path.display()
12434        );
12435        path
12436    }
12437
12438    /// Poll for the grandchild pid the stub records, and parse it.
12439    fn read_grandchild_pid(path: &Path) -> u32 {
12440        let deadline = Instant::now() + Duration::from_secs(10);
12441        loop {
12442            if let Ok(contents) = std::fs::read_to_string(path) {
12443                if let Ok(pid) = contents.trim().parse() {
12444                    return pid;
12445                }
12446            }
12447            assert!(
12448                Instant::now() < deadline,
12449                "the stub never recorded a grandchild pid at {}",
12450                path.display()
12451            );
12452            std::thread::sleep(Duration::from_millis(10));
12453        }
12454    }
12455
12456    /// Everything one fixture run needs, so the two tests below differ in exactly
12457    /// one place: whether the child is contained.
12458    struct Fixture {
12459        _dir: TestTempDir,
12460        module_id: String,
12461        grandchild: u32,
12462        child: Option<SupervisedChild>,
12463        registry: Arc<Registry>,
12464        snapshot: Arc<Mutex<SupervisorSnapshot>>,
12465        terminal_ring: Arc<Mutex<TerminalRing>>,
12466        spawn_events: SpawnEventFeed,
12467    }
12468
12469    fn fixture(label: &str, module_id: &str) -> Fixture {
12470        let dir = TestTempDir::new(label);
12471        let pid_file = dir.join("grandchild.pid");
12472        let supervisor = Supervisor::new_for_test(
12473            Arc::new(Registry::default()),
12474            RestartPolicy::new(3, Duration::ZERO),
12475        );
12476        let runtime = supervisor.runtime_config();
12477        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12478        let spec = ModuleSpec {
12479            module_id: module_id.to_string(),
12480            program: stub_path(),
12481            // Zero args deliberately: a `--subc` argument would make the stub dial
12482            // a daemon that is not there, and the failure would land in the same
12483            // stderr ring this fixture exists to keep quiet.
12484            args: Vec::new(),
12485            env: vec![
12486                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
12487                (
12488                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
12489                    pid_file.display().to_string(),
12490                ),
12491            ],
12492            reserved: false,
12493            reserved_prefixes: Vec::new(),
12494            protocol: ModuleProtocol::Subc,
12495            overlap: Default::default(),
12496        };
12497        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
12498            .expect("spawn the supervised fixture");
12499        let grandchild = read_grandchild_pid(&pid_file);
12500        Fixture {
12501            _dir: dir,
12502            module_id: module_id.to_string(),
12503            grandchild,
12504            child: Some(child),
12505            registry: Arc::new(Registry::default()),
12506            snapshot,
12507            terminal_ring: Arc::clone(&runtime.terminal_ring),
12508            spawn_events: SpawnEventFeed::default(),
12509        }
12510    }
12511
12512    impl Fixture {
12513        /// Drain through the supervisor's own teardown path.
12514        async fn drain(&mut self) {
12515            let child = self
12516                .child
12517                .take()
12518                .expect("the fixture child is still present");
12519            drain_child_to_state(
12520                &self.module_id,
12521                ModuleProtocol::Subc,
12522                // No forwarding table in this fixture, so nothing reaches the
12523                // child over a connection.
12524                StopNotice::NotSent,
12525                &self.registry,
12526                None,
12527                &self.snapshot,
12528                &self.terminal_ring,
12529                &self.spawn_events,
12530                child,
12531                Duration::from_millis(500),
12532                ModuleState::Stopped,
12533                Some(false),
12534            )
12535            .await
12536            .expect("drain the supervised fixture");
12537        }
12538    }
12539
12540    /// Teardown reaps the grandchild, not merely the direct child.
12541    ///
12542    /// This is the assertion the change exists for. Before containment the
12543    /// grandchild survived: it is a separate process, and `start_kill` is
12544    /// `TerminateProcess` scoped to one pid.
12545    #[tokio::test]
12546    async fn teardown_reaps_the_grandchild() {
12547        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
12548        let grandchild = fixture.grandchild;
12549
12550        assert!(
12551            subc_jobobject::process_exists(grandchild),
12552            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
12553        );
12554
12555        fixture.drain().await;
12556
12557        assert!(
12558            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12559            "grandchild {grandchild} outlived module teardown: the tree was not contained"
12560        );
12561    }
12562
12563    /// The mutation control: with containment withheld, the grandchild survives
12564    /// the same kill.
12565    ///
12566    /// This is the defect reproduction from #109 — a direct-child kill reaches
12567    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
12568    /// supervisor because `spawn_and_mark_running` now always contains on
12569    /// Windows, which is the point: there is no longer a path that spawns
12570    /// uncontained, so the control has to construct one.
12571    ///
12572    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
12573    /// grandchild ever dies here, that test is passing for a reason unrelated to
12574    /// the job object and the containment claim is unproven.
12575    #[test]
12576    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12577        let dir = TestTempDir::new("teardown-uncontained");
12578        let pid_file = dir.join("grandchild.pid");
12579        let mut child = std::process::Command::new(stub_path())
12580            .env("FAKE_AFT_NEVER_CONNECT", "1")
12581            .env(
12582                "FAKE_AFT_GRANDCHILD_PID_FILE",
12583                pid_file.display().to_string(),
12584            )
12585            .stdin(std::process::Stdio::null())
12586            .stdout(std::process::Stdio::null())
12587            .stderr(std::process::Stdio::null())
12588            .spawn()
12589            .expect("spawn the uncontained fixture");
12590        let grandchild = read_grandchild_pid(&pid_file);
12591
12592        // Exactly what the pre-fix teardown did: kill the direct child.
12593        child.kill().expect("kill the direct child");
12594        let _ = child.wait();
12595
12596        assert!(
12597            subc_jobobject::process_exists(grandchild),
12598            "grandchild {grandchild} died with the direct child, so this control no longer \
12599             distinguishes contained from uncontained teardown and the regression test is \
12600             passing vacuously"
12601        );
12602
12603        // The orphan this control demonstrates is the leak the fix prevents, so
12604        // the control must not leave one behind.
12605        kill_tree(grandchild);
12606    }
12607
12608    /// Crash durability: closing the containment handle reaps the tree with no
12609    /// teardown code running at all.
12610    ///
12611    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
12612    /// call anything — and it is why containment is a kernel property of the
12613    /// handle rather than a step in the drain. Discovered by getting the
12614    /// mutation control wrong: clearing `job` to "disable" containment instead
12615    /// killed the tree, which is the guarantee, not a mistake.
12616    #[tokio::test]
12617    async fn dropping_containment_reaps_the_grandchild() {
12618        let mut fixture = fixture("drop-containment", "tree-drop");
12619        let grandchild = fixture.grandchild;
12620
12621        assert!(subc_jobobject::process_exists(grandchild));
12622
12623        // No `drain` call, no kill: dropping the handle is the entire mechanism.
12624        fixture.child.as_mut().expect("child present").job = None;
12625
12626        assert!(
12627            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12628            "grandchild {grandchild} survived the containment handle closing, so a daemon \
12629             crash would leave the tree behind"
12630        );
12631    }
12632
12633    /// Kill a pid and its tree, then confirm it is gone.
12634    fn kill_tree(pid: u32) {
12635        let _ = std::process::Command::new("taskkill.exe")
12636            .args(["/PID", &pid.to_string(), "/T", "/F"])
12637            .stdin(std::process::Stdio::null())
12638            .stdout(std::process::Stdio::null())
12639            .stderr(std::process::Stdio::null())
12640            .status();
12641        assert!(
12642            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12643            "could not clean up grandchild {pid}"
12644        );
12645    }
12646}
12647
12648#[cfg(test)]
12649mod privacy_trampoline_configuration_tests {
12650    #[tokio::test]
12651    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12652    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
12653        #[cfg(target_os = "macos")]
12654        {
12655            let supervisor = super::Supervisor::new(
12656                std::sync::Arc::new(crate::Registry::default()),
12657                super::RestartPolicy::default(),
12658            );
12659            let error = supervisor.spawn(spec()).unwrap_err();
12660            assert!(
12661                error
12662                    .to_string()
12663                    .contains("no privacy trampoline configured"),
12664                "{error}"
12665            );
12666        }
12667    }
12668
12669    #[tokio::test]
12670    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12671    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
12672        #[cfg(target_os = "macos")]
12673        {
12674            let supervisor = super::Supervisor::new(
12675                std::sync::Arc::new(crate::Registry::default()),
12676                super::RestartPolicy::default(),
12677            )
12678            .with_privacy_trampoline(std::env::current_exe().unwrap());
12679            let error = supervisor.spawn(spec()).unwrap_err();
12680            assert!(
12681                error
12682                    .to_string()
12683                    .contains("binary does not implement the privacy trampoline protocol"),
12684                "{error}"
12685            );
12686        }
12687    }
12688
12689    #[cfg(target_os = "macos")]
12690    fn spec() -> super::ModuleSpec {
12691        super::ModuleSpec {
12692            module_id: "privacy-configuration".into(),
12693            program: "/bin/sleep".into(),
12694            args: vec!["30".into()],
12695            env: vec![],
12696            reserved: false,
12697            reserved_prefixes: vec![],
12698            protocol: subc_control::ModuleProtocol::None,
12699            overlap: super::ModuleOverlap::Exclusive,
12700        }
12701    }
12702}
12703
12704/// The daemon's real spawn path hands a subc-wire child its launch nonce on
12705/// descriptor 3, without an environment copy. The shell records the nonce
12706/// and its environment after exec so these tests observe the real handover.
12707#[cfg(all(test, unix))]
12708mod launch_nonce_descriptor_tests {
12709    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
12710    use crate::stderr_tail::{StderrRing, StderrTailConfig};
12711    use std::{
12712        path::PathBuf,
12713        sync::{Arc, Mutex},
12714        time::{Duration, Instant},
12715    };
12716    use subc_test_support::TestTempDir;
12717
12718    async fn probe(role: super::SpawnRole) {
12719        let scratch = TestTempDir::new("launch-nonce-descriptor");
12720        let fd_copy = scratch.join("from-descriptor");
12721        let env_copy = scratch.join("environment");
12722        let script = format!(
12723            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
12724            fd = fd_copy.display(), env = env_copy.display(),
12725        );
12726        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
12727        let spec = ModuleSpec {
12728            module_id: "nonce-descriptor-probe".to_string(),
12729            program: PathBuf::from("/bin/sh"),
12730            args: vec!["-c".to_string(), script],
12731            env: vec![
12732                xdg("XDG_DATA_HOME"),
12733                xdg("XDG_RUNTIME_DIR"),
12734                xdg("XDG_CONFIG_HOME"),
12735                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
12736            ],
12737            reserved: true,
12738            reserved_prefixes: Vec::new(),
12739            protocol: ModuleProtocol::Subc,
12740            overlap: Default::default(),
12741        };
12742        let handle = SupervisorHandle::new();
12743        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12744        let roster = ChildRoster::default();
12745        #[cfg(target_os = "macos")]
12746        {
12747            let path = super::test_privacy_trampoline();
12748            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
12749        }
12750        let child = super::spawn_child_in_slot(
12751            &spec,
12752            None,
12753            Some(&handle),
12754            &ring,
12755            None,
12756            &roster,
12757            #[cfg(target_os = "linux")]
12758            None,
12759            role,
12760            matches!(role, super::SpawnRole::SwapCandidate),
12761        )
12762        .expect("spawn probe");
12763        let deadline = Instant::now() + Duration::from_secs(10);
12764        while !(fd_copy.exists() && env_copy.exists()) {
12765            assert!(Instant::now() < deadline, "probe never wrote its copies");
12766            tokio::time::sleep(Duration::from_millis(20)).await;
12767        }
12768        let nonce = std::fs::read_to_string(fd_copy).unwrap();
12769        assert!(!nonce.is_empty());
12770        let environment = std::fs::read_to_string(env_copy).unwrap();
12771        assert!(environment
12772            .lines()
12773            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
12774        let copy = environment
12775            .lines()
12776            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
12777        assert_eq!(
12778            copy, None,
12779            "Unix children must never receive the environment nonce"
12780        );
12781        if matches!(role, super::SpawnRole::Plain) {
12782            assert_eq!(
12783                handle.spawn_nonce(&spec.module_id).as_deref(),
12784                Some(nonce.as_str())
12785            );
12786        }
12787        drop(child);
12788    }
12789
12790    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12791    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
12792        probe(super::SpawnRole::Plain).await;
12793    }
12794
12795    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12796    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
12797        probe(super::SpawnRole::SwapCandidate).await;
12798    }
12799}
12800
12801#[cfg(all(test, target_os = "linux"))]
12802mod cgroup_containment_tests {
12803    use super::*;
12804    use subc_test_support::TestTempDir;
12805
12806    fn running(pid: u32) -> bool {
12807        // An orphan can remain a zombie until the container init reaps it.
12808        std::fs::read_to_string(format!("/proc/{pid}/stat"))
12809            .ok()
12810            .and_then(|stat| {
12811                stat.rsplit_once(") ")
12812                    .map(|(_, rest)| rest.starts_with('Z'))
12813            })
12814            .is_some_and(|zombie| !zombie)
12815    }
12816
12817    #[tokio::test]
12818    async fn linux_teardown_reaps_the_grandchild() {
12819        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
12820    }
12821
12822    #[tokio::test]
12823    async fn linux_shutdown_straggler_reaps_the_grandchild() {
12824        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
12825    }
12826
12827    async fn teardown_tree(test_name: &str, shutdown: bool) {
12828        let dir = TestTempDir::new(test_name);
12829        let root = PathBuf::from(format!(
12830            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
12831            std::process::id(),
12832            unix_ms_now()
12833        ));
12834        if let Err(error) = std::fs::create_dir(&root) {
12835            assert!(
12836                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12837                "required cgroup test cannot execute: {error}"
12838            );
12839            eprintln!(
12840                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
12841                root.display()
12842            );
12843            return;
12844        }
12845        let placement = subc_cgroup::prepare_at(&root)
12846            .expect("prepare isolated kernel cgroup")
12847            .expect("isolated cgroup is delegated");
12848        let module_id = "tree-teardown";
12849        let module = placement
12850            .module_path(module_id)
12851            .expect("create isolated module cgroup");
12852        if !module.join("cgroup.kill").exists() {
12853            std::fs::remove_dir(&module).unwrap();
12854            std::fs::remove_dir(root.join("subc-modules")).unwrap();
12855            std::fs::remove_dir(&root).unwrap();
12856            assert!(
12857                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12858                "required cgroup.kill interface unavailable"
12859            );
12860            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
12861            return;
12862        }
12863        let supervisor = Supervisor::new_for_test(
12864            Arc::new(Registry::default()),
12865            RestartPolicy::new(3, Duration::ZERO),
12866        )
12867        .with_cgroup_placement(Some(placement));
12868        let mut runtime = supervisor.runtime_config();
12869        runtime.child_roster = runtime
12870            .child_roster
12871            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
12872        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12873        let pid_file = dir.join("grandchild.pid");
12874        let spec = ModuleSpec {
12875            module_id: module_id.to_string(),
12876            program: PathBuf::from("/bin/sh"),
12877            args: vec![
12878                "-c".into(),
12879                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
12880                "fixture".into(),
12881                pid_file.display().to_string(),
12882            ],
12883            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
12884                .into_iter()
12885                .map(|key| (key.to_string(), dir.display().to_string()))
12886                .collect(),
12887            reserved: false,
12888            reserved_prefixes: Vec::new(),
12889            protocol: ModuleProtocol::None,
12890            overlap: Default::default(),
12891        };
12892        let child =
12893            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
12894        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
12895        let grandchild: u32 = loop {
12896            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
12897                if let Ok(pid) = contents.trim().parse() {
12898                    break pid;
12899                }
12900            }
12901            assert!(
12902                tokio::time::Instant::now() < deadline,
12903                "grandchild pid was not recorded"
12904            );
12905            tokio::time::sleep(Duration::from_millis(10)).await;
12906        };
12907        assert!(
12908            running(grandchild),
12909            "grandchild must be alive before teardown"
12910        );
12911        if shutdown {
12912            let mut child = child;
12913            crate::child_roster::end_children_for_daemon_shutdown(
12914                &runtime.child_roster,
12915                false,
12916                std::future::pending(),
12917            )
12918            .await;
12919            child.wait().await.expect("reap shutdown straggler");
12920        } else {
12921            drain_child_to_state(
12922                module_id,
12923                ModuleProtocol::None,
12924                StopNotice::NotSent,
12925                &Registry::default(),
12926                None,
12927                &snapshot,
12928                &runtime.terminal_ring,
12929                &SpawnEventFeed::default(),
12930                child,
12931                Duration::from_millis(100),
12932                ModuleState::Stopped,
12933                Some(false),
12934            )
12935            .await
12936            .expect("real supervisor teardown");
12937        }
12938        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
12939        while running(grandchild) && tokio::time::Instant::now() < deadline {
12940            tokio::time::sleep(Duration::from_millis(10)).await;
12941        }
12942        let survived = running(grandchild);
12943        // Kill a surviving grandchild so a failed test does not leave it behind.
12944        if survived {
12945            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
12946            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
12947            tokio::time::sleep(Duration::from_millis(100)).await;
12948        }
12949        if module.exists() {
12950            std::fs::remove_dir(&module).expect("remove empty module cgroup");
12951        }
12952        std::fs::remove_dir(root.join("subc-modules")).unwrap();
12953        std::fs::remove_dir(&root).unwrap();
12954        assert!(
12955            !survived,
12956            "grandchild {grandchild} outlived module teardown"
12957        );
12958        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
12959    }
12960}