Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    let path = std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture");
134    // Without the fixture every macOS spawn is refused, and the tests that
135    // spawn fail later as a module in state Failed, which names the wrong
136    // cause. `cargo test -p subc-daemon --lib` alone does not build it.
137    assert!(
138        path.exists(),
139        "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140         --bins --features test-support` or `cargo test -p subc-daemon` first",
141        path.display()
142    );
143    path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148    use std::io::Read;
149    let mut probe = std::process::Command::new(path)
150        .args(["__disclaim-exec", "--probe"])
151        .stdin(Stdio::null())
152        .stdout(Stdio::piped())
153        .stderr(Stdio::piped())
154        .spawn()
155        .map_err(|error| {
156            format!(
157                "privacy trampoline probe failed for {}: {error}",
158                path.display()
159            )
160        })?;
161    let deadline = std::time::Instant::now() + Duration::from_secs(5);
162    let status = loop {
163        match probe.try_wait() {
164            Ok(Some(status)) => break status,
165            Ok(None) if std::time::Instant::now() < deadline => {
166                std::thread::sleep(Duration::from_millis(5))
167            }
168            result => {
169                let _ = probe.kill();
170                let _ = probe.wait();
171                return Err(format!(
172                    "privacy trampoline probe failed or timed out for {}: {result:?}",
173                    path.display()
174                ));
175            }
176        }
177    };
178    let mut answer = String::new();
179    if let Some(stdout) = probe.stdout.take() {
180        let _ = stdout.take(256).read_to_string(&mut answer);
181    }
182    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183        return Ok(());
184    }
185    let mut diagnostic = String::new();
186    if let Some(stderr) = probe.stderr.take() {
187        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188    }
189    let cause = diagnostic
190        .trim()
191        .strip_prefix("ck-subc: own privacy identity refused: ")
192        .unwrap_or("binary does not implement the privacy trampoline protocol");
193    Err(format!(
194        "{cause}: probe of {} exited {status}",
195        path.display()
196    ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201    spec: &ModuleSpec,
202    roster: &ChildRoster,
203) -> Result<
204    (
205        Command,
206        Option<PrivacyExec>,
207        subc_os::privacy_identity::ExecAcknowledgement,
208    ),
209    SuperviseError,
210> {
211    let failure = |cause: String| {
212        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213        SuperviseError::Spawn {
214            program: spec.program.clone(),
215            source: io::Error::other(cause),
216            cgroup_path: None,
217        }
218    };
219    let trampoline = roster.privacy_trampoline().map_err(failure)?;
220    // Resolve PATH with the same environment the Command will receive. For
221    // scripts retain the existing orphan-identity rule: the kernel chooses
222    // the interpreter, and its observed image is the one recorded. Do not
223    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
224    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225        let path = spec
226            .env
227            .iter()
228            .find(|(key, _)| key == "PATH")
229            .map(|(_, value)| std::ffi::OsString::from(value))
230            .or_else(|| std::env::var_os("PATH"))
231            .unwrap_or_else(|| "/usr/bin:/bin".into());
232        std::env::split_paths(&path)
233            .map(|dir| dir.join(&spec.program))
234            .find(|path| path.is_file())
235            .unwrap_or_else(|| spec.program.clone())
236    } else {
237        spec.program.clone()
238    };
239    let expected = subc_os::file_identity(&program);
240    let trampoline_image = subc_os::file_identity(&trampoline);
241    let script = {
242        use std::io::Read;
243        let mut prefix = [0u8; 2];
244        std::fs::File::open(&program)
245            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246    };
247    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248        return Err(failure(
249            "privacy identity module executable is missing or is the trampoline itself".to_string(),
250        ));
251    }
252    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253        .map_err(|error| failure(error.to_string()))?;
254    let reader =
255        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256    let mut command = Command::new(&trampoline);
257    command
258        .arg("__disclaim-exec")
259        .arg(ack.fd().to_string())
260        .arg(&program);
261    ack.install(command.as_std_mut());
262    Ok((
263        command,
264        Some(PrivacyExec {
265            reader,
266            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267            expected,
268            trampoline: trampoline_image,
269            script,
270            module_id: spec.module_id.clone(),
271        }),
272        ack,
273    ))
274}
275
276struct SupervisedChild {
277    child: Child,
278    #[cfg(target_os = "macos")]
279    privacy_exec: Option<PrivacyExec>,
280    /// Set once this launch's exec acknowledgement confirms the module image.
281    /// On macOS the pid first runs the `ck-subc` launch trampoline (see
282    /// `subc_os::privacy_identity`), which then replaces itself with the
283    /// module. The supervisor owns and can kill that pid from spawn, but
284    /// status readers report it only after this latch is set, so nothing
285    /// reports the trampoline's image as the module's.
286    #[cfg(target_os = "macos")]
287    report_ready: Arc<OnceLock<()>>,
288    /// Refusal before the module image was accepted, retained for terminal records.
289    spawn_failure: Option<String>,
290    /// The protocol this process was launched with. A reload can store a new
291    /// launch spec with a different protocol, but that takes effect only at the
292    /// next spawn, so this process keeps being handled by the protocol it
293    /// actually speaks.
294    protocol: ModuleProtocol,
295    /// This process's cgroup name: a bounded module/slot label followed by a
296    /// spawn suffix unique to this process (when cgroup placement is on). A
297    /// retired process in a slot may still be draining when a later one is
298    /// spawned into that slot, so the suffix keeps the later process out of
299    /// the retired one's cgroup, which is the domain a kill applies to.
300    #[cfg(target_os = "linux")]
301    module_id: String,
302    #[cfg(target_os = "linux")]
303    cgroup_placement: Option<subc_cgroup::Placement>,
304    /// The job that contains this child and every process it spawns (issue #109).
305    ///
306    /// Dropping this handle is what reaps a surviving tree when no supervisor
307    /// code runs — a daemon crash — because the job carries
308    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
309    ///
310    /// That limit is not crash-only, and the difference is worth knowing: a
311    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
312    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
313    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
314    /// module at once. Before this change they survived that, saw EOF on the
315    /// control socket, and ran their own teardown; Unix keeps that path
316    /// deliberately, so a module can seal a WAL or close a capture rather than
317    /// be killed mid-write. So this trades graceful teardown on every Windows
318    /// daemon stop for containment on a crash, which is the right way round
319    /// today: orphaned GPU workers are a reported, recurring problem, and the
320    /// modules that write most heavily do not run on Windows.
321    ///
322    /// The fix is a real Windows stop path — the daemon draining before it
323    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
324    /// reaches only what the drain left behind, which is what it should reach.
325    #[cfg(windows)]
326    job: Option<subc_jobobject::JobObject>,
327    stdout_pump: Option<JoinHandle<()>>,
328    stderr_pump: Option<StderrPump>,
329    stderr_ring: Arc<Mutex<StderrRing>>,
330    spawned_at_ms: u64,
331    spawned_from: PathBuf,
332    spawned_file_identity: Option<SpawnedFileIdentity>,
333    process_start_time: Option<u64>,
334    process_identity: Option<ProcessIdentity>,
335    pid: u32,
336    /// This process's entry in the daemon's child roster, released when the
337    /// process is reaped or this handle is dropped.
338    roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342    fn id(&self) -> Option<u32> {
343        Some(self.pid)
344    }
345
346    fn process_identity(&self) -> Option<ProcessIdentity> {
347        self.process_identity
348    }
349
350    async fn wait(&mut self) -> io::Result<ExitStatus> {
351        #[cfg(target_os = "macos")]
352        self.confirm_privacy_exec().await;
353        // The roster entry is NOT released here. A daemon shutdown waits for the
354        // roster to empty and then exits the process, so releasing at the reap
355        // let it exit before the exit handler wrote this child's terminal record
356        // (the stderr drain and snapshot update sit in between), and the
357        // shutdown's own `daemon_shutdown` record was intermittently lost. The
358        // caller releases it after recording the exit (`release_roster`), and
359        // dropping the handle releases it too.
360        let result = self.child.wait().await;
361        #[cfg(target_os = "linux")]
362        if result.is_ok() {
363            if let Some(placement) = self.cgroup_placement.as_ref() {
364                cleanup_reaped_cgroup(placement, &self.module_id).await;
365                // Keep ownership while awaiting kernel population changes: a
366                // drain timeout may cancel this wait and then escalate/reap.
367                self.cgroup_placement = None;
368            }
369        }
370        result
371    }
372
373    #[cfg(target_os = "macos")]
374    async fn confirm_privacy_exec(&mut self) {
375        let Some(pending) = &mut self.privacy_exec else {
376            return;
377        };
378        let result = tokio::time::timeout_at(pending.deadline, async {
379            let mut record = Vec::new();
380            loop {
381                let mut ready = pending.reader.readable().await?;
382                let read = ready.try_io(|reader| {
383                    use std::io::Read;
384                    let mut reader = reader.get_ref();
385                    let mut buffer = [0u8; 256];
386                    reader.read(&mut buffer).map(|count| (count, buffer))
387                });
388                match read {
389                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390                    Ok(Ok((count, buffer))) => {
391                        if record.len() + count > 1024 {
392                            return Err(io::Error::other(
393                                "privacy exec refusal record is too long",
394                            ));
395                        }
396                        record.extend_from_slice(&buffer[..count]);
397                    }
398                    Ok(Err(error)) => return Err(error),
399                    Err(_) => continue,
400                }
401            }
402        })
403        .await;
404        // Keep the reader in self across await: select cancellation must not
405        // discard the handshake or reset its original five-second deadline.
406        let pending = self.privacy_exec.as_ref().expect("pending exec");
407        let cause = match result {
408            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409            Ok(Err(error)) => Some(format!(
410                "privacy identity exec acknowledgement failed: {error}"
411            )),
412            Ok(Ok(record)) if !record.is_empty() => Some(
413                std::str::from_utf8(&record)
414                    .ok()
415                    .and_then(|record| {
416                        record
417                            .trim()
418                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419                    })
420                    .filter(|cause| !cause.is_empty())
421                    .unwrap_or("invalid privacy exec refusal record")
422                    .to_string(),
423            ),
424            Ok(Ok(_)) => match self.child.try_wait() {
425                // Empty EOF is the exec acknowledgement. A real module may exit
426                // immediately, including with a reserved trampoline status; no
427                // image is admitted, and its ordinary exit contract stays intact.
428                Ok(Some(_status)) => None,
429                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430                Ok(None) => {
431                    let image = observe_spawned_image(self.pid);
432                    if let Some(image) = image.filter(|image| {
433                        image.executable.is_some()
434                            && image.executable != pending.trampoline
435                            && (image.executable == pending.expected || pending.script)
436                    }) {
437                        if let Some(guard) = &self.roster_guard {
438                            guard.confirm_executable(image);
439                        }
440                        let _ = self.report_ready.set(());
441                        info!(module_id = %pending.module_id, pid = self.pid,
442                            "module spawned with own privacy identity (responsibility disclaimed)");
443                        None
444                    } else if image.is_none()
445                        || image.is_some_and(|image| image.executable.is_none())
446                    {
447                        // A process can exit between try_wait and the kernel
448                        // image read. Empty EOF already acknowledged exec, so
449                        // preserve that module's ordinary exit rather than
450                        // mislabel a disappearing image as trampoline refusal.
451                        // Keep pending in self across await so cancellation does
452                        // not discard validation or reset its original deadline.
453                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454                            Ok(Ok(_status)) => None,
455                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457                        }
458                    } else {
459                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460                    }
461                }
462            },
463        };
464        let pending = self.privacy_exec.take().expect("pending exec");
465        if let Some(cause) = cause {
466            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467            self.spawn_failure = Some(cause);
468            // No image is admitted on failure. Reach the entire fresh process
469            // group, including a module which spawned a helper before refusal.
470            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472            }
473            let _ = self.child.start_kill();
474        }
475    }
476
477    /// Releases this child's daemon-shutdown roster entry once its exit has
478    /// been recorded. The pid is already reaped and free for reuse, so the
479    /// entry must not outlive the record any longer than that.
480    fn release_roster(&mut self) {
481        self.roster_guard = None;
482    }
483
484    /// Kill the child and, where containment is available, its process tree.
485    ///
486    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
487    /// helper process leaked the helper — the Synapse embedding module's CUDA
488    /// worker holds the GPU allocation, so the leak cost VRAM until the next
489    /// restart of something else. Terminating the job reaches grandchildren that
490    /// a tree walk cannot, including one whose parent has already exited and
491    /// been reparented away.
492    ///
493    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
494    /// direct-child kill still decides the outcome, so containment can never
495    /// change whether a module is reported as stopped.
496    fn start_kill(&mut self) -> io::Result<()> {
497        #[cfg(windows)]
498        if let Some(job) = &self.job {
499            if let Err(error) = job.terminate() {
500                debug!(
501                    error = %error,
502                    "job termination failed; the direct-child kill still owns the outcome"
503                );
504            }
505        }
506        #[cfg(target_os = "linux")]
507        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508        self.child.start_kill()
509    }
510
511    async fn drain_stderr(&mut self, module_id: &str) {
512        if let Some(mut pump) = self.stdout_pump.take() {
513            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514                Ok(Ok(())) => {}
515                Ok(Err(error)) => {
516                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517                }
518                Err(_) => {
519                    pump.abort();
520                    warn!(
521                        module_id,
522                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523                        "stdout pump did not drain before restart; stopped it before the next process"
524                    );
525                }
526            }
527        }
528
529        let Some(pump) = self.stderr_pump.take() else {
530            return;
531        };
532        settle_stderr_pump(
533            module_id,
534            &self.stderr_ring,
535            pump,
536            STDERR_PUMP_DRAIN_TIMEOUT,
537        )
538        .await;
539    }
540}
541
542/// The reader task for one process's stderr, with the ring generation its
543/// lines are attributed to.
544struct StderrPump {
545    task: JoinHandle<()>,
546    generation: u64,
547}
548
549/// Retire an exited process's stderr reader and wait up to `bound` for it to
550/// reach EOF. A reader still running at the bound is detached, not stopped: it
551/// keeps filling the exited process's section of the ring until its pipe
552/// closes, and the tail reads `Incomplete` until then. See
553/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
554async fn settle_stderr_pump(
555    module_id: &str,
556    ring: &Arc<Mutex<StderrRing>>,
557    pump: StderrPump,
558    bound: Duration,
559) {
560    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561    let StderrPump {
562        mut task,
563        generation,
564    } = pump;
565    lock().retire_pump(generation);
566    match timeout(bound, &mut task).await {
567        Ok(Ok(())) => {}
568        Ok(Err(err)) => {
569            let mut ring = lock();
570            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571            ring.finish_pump(generation);
572            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573        }
574        Err(_) => {
575            // Dropping the handle detaches the task; it ends at EOF on its pipe.
576            drop(task);
577            lock().mark_pump_late(
578                generation,
579                format!(
580                    "stderr of the exited process had not reached EOF {bound:?} after it was \
581                     retired (a descendant may still hold the pipe open); lines it still \
582                     writes are kept in that process's section"
583                ),
584            );
585            warn!(
586                module_id,
587                waited = ?bound,
588                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589            );
590        }
591    }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596    EVENTS.get_or_init(|| {
597        let (sender, _receiver) = watch::channel(0);
598        sender
599    })
600}
601
602pub(crate) fn notify_registration_release() {
603    let events = registration_release_events();
604    let next_generation = (*events.borrow()).wrapping_add(1);
605    events.send_replace(next_generation);
606}
607
608/// How to launch one singleton module process.
609#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611    pub module_id: String,
612    pub program: PathBuf,
613    pub args: Vec<String>,
614    pub env: Vec<(String, String)>,
615    /// When true this is a reserved module: each spawn gets a fresh one-time launch
616    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
617    /// process can register this module_id (a security-boundary module like the
618    /// credential vault must not be impersonable while it is down/restarting).
619    pub reserved: bool,
620    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
621    /// Prefixes come from daemon config and must end in `:` before they reach the
622    /// supervisor; the owner module's current spawn nonce authorizes claims under
623    /// each prefix.
624    pub reserved_prefixes: Vec<String>,
625    /// The wire protocol this module speaks, as DECLARED in daemon config.
626    ///
627    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
628    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
629    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
630    /// and NO launch nonce, and a clean exit the daemon did not request is
631    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
632    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
633    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
634    /// because a process ignores an environment variable it does not read.
635    ///
636    /// The argument is the part that cannot be "harmless to a process that
637    /// ignores it": a stock binary exits on an unknown flag before it listens
638    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
639    /// first conformance run against this mode found it. The nonce is withheld
640    /// because a process that will never present it gains nothing from holding
641    /// it, and a secret in the environment of a process that does not need it is
642    /// a leak surface for no benefit.
643    pub protocol: ModuleProtocol,
644    /// Whether two processes of this module may run at once, which is what a
645    /// blue/green swap does for the length of its overlap. Declared in daemon
646    /// config because the daemon must be able to answer it while the module is
647    /// down, and so a module cannot talk itself into it after registering.
648    pub overlap: ModuleOverlap,
649}
650
651/// Whether a module tolerates a second process of itself running alongside.
652///
653/// Most modules are single-writer on their store (a WAL, a capture log, a
654/// resident index behind a writer barrier), and two processes on one store
655/// corrupt it. So a swap, which overlaps the old and new process by design,
656/// is refused unless the module's config opts in with `overlap: "safe"`.
657#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659    /// Never run two processes of this module at once. The default.
660    #[default]
661    Exclusive,
662    /// The module has said a second process of itself is harmless for the
663    /// length of a swap.
664    ///
665    /// Declare it only if a second instance can run for a few seconds without
666    /// touching ANY single-writer store: every database, WAL, index, projector
667    /// and scheduled job the module owns. A lease on part of that state is not
668    /// enough. broca's session lease guards WAL appends while its run index, its
669    /// store projector and its archive fold timer (which unlinks live WAL files)
670    /// stay single-writer, so broca is exclusive despite holding a lease. The
671    /// refusal only fires after this has been decided, so the decision is the
672    /// check.
673    Safe,
674}
675
676impl ModuleOverlap {
677    pub fn as_str(self) -> &'static str {
678        match self {
679            Self::Exclusive => "exclusive",
680            Self::Safe => "safe",
681        }
682    }
683}
684
685/// Environment variable telling a spawned module which case it was started
686/// for, before it sends HELLO. Only a swap candidate carries it, as
687/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
688///
689/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
690/// longer because nobody waits on it, while a plain restart must flip ready
691/// quickly because callers see `module_warming` until it does. Absence means
692/// plain restart, the safe reading. The daemon trusts nothing about it; the
693/// candidate is proven by its launch nonce at HELLO.
694pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
696pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697/// How long a swap waits for its candidate to register and declare itself
698/// ready when the operator does not say. A module warming as a swap candidate
699/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
700/// daemon allows that plus time to start the process and send HELLO.
701pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703/// Bounded restart policy for crash exits.
704///
705/// `max_restarts` is the number of replacement processes allowed after the
706/// initial spawn WITHIN `window`. After that many crash restarts inside one
707/// window the module enters [`ModuleState::Failed`] and the supervisor stops
708/// the crash loop.
709///
710/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
711/// and that only survived because crashes were rare: a module that crashed
712/// three times across a week was disabled forever by crashes that had nothing
713/// to do with each other. That stopped being survivable once modules began
714/// exiting non-zero whenever the daemon's connection to them drops, because
715/// then every daemon-side connection drop spends a unit of the same budget and
716/// one flappy hour permanently stops a healthy module. Restarts older than
717/// `window` release their slot, so a module that crashed twice yesterday has a
718/// full budget today, while a genuine crash loop -- which is fast by
719/// definition -- still reaches the cap and stops.
720#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722    pub max_restarts: u32,
723    /// Base delay before a crash replacement. The actual delay escalates with
724    /// the number of recent crash replacements and is capped by `max_backoff`.
725    pub backoff: Duration,
726    /// Maximum delay before a crash replacement.
727    pub max_backoff: Duration,
728    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
729    /// budget effectively infinite (nothing is ever in-window), which is why
730    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
731    pub window: Duration,
732}
733
734impl RestartPolicy {
735    /// A policy with the default crash window. Callers that care about the
736    /// window say so with [`Self::with_window`]; the ones that do not are
737    /// asking for the standard rate limit, not for no limit.
738    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739        Self {
740            max_restarts,
741            backoff,
742            max_backoff: DEFAULT_MAX_BACKOFF,
743            window: DEFAULT_RESTART_WINDOW,
744        }
745    }
746
747    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748        self.max_backoff = max_backoff;
749        self
750    }
751
752    pub fn with_window(mut self, window: Duration) -> Self {
753        self.window = window;
754        self
755    }
756
757    /// Calculate the capped exponential delay for the next crash replacement.
758    /// `restart_in_window` is zero for the first replacement after an operator
759    /// action (restart, reload, re-enable) cleared the crash ring, or after all
760    /// older crash replacements have aged out of the window.
761    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762        if self.backoff.is_zero() || self.max_backoff.is_zero() {
763            return Duration::ZERO;
764        }
765
766        let mut delay = self.backoff;
767        for _ in 0..restart_in_window {
768            if delay >= self.max_backoff {
769                return self.max_backoff;
770            }
771            delay = delay
772                .checked_mul(10)
773                .unwrap_or(self.max_backoff)
774                .min(self.max_backoff);
775        }
776        delay.min(self.max_backoff)
777    }
778
779    /// The one sentence that explains a budget-exhausted stop, used for both the
780    /// log line and the terminal record so the two cannot drift. It names the
781    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
782    /// exactly what this budget is not.
783    fn budget_exhausted_detail(&self) -> String {
784        format!(
785            "crash budget exhausted: max_restarts={} within window_secs={}",
786            self.max_restarts,
787            self.window.as_secs()
788        )
789    }
790}
791
792impl Default for RestartPolicy {
793    fn default() -> Self {
794        Self {
795            max_restarts: DEFAULT_MAX_RESTARTS,
796            backoff: DEFAULT_BACKOFF,
797            max_backoff: DEFAULT_MAX_BACKOFF,
798            window: DEFAULT_RESTART_WINDOW,
799        }
800    }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805    restart_in_window: u32,
806    delay: Duration,
807}
808
809/// Whether the daemon itself will bring this module back after the exit being
810/// handled: it is enabled AND its in-window crash restarts are below the cap.
811///
812/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
813/// the window are dropped here rather than by a timer, so the count is right
814/// the moment somebody asks and no bookkeeping runs for idle modules.
815fn daemon_will_restart(
816    state: &mut SupervisorSnapshot,
817    policy: &RestartPolicy,
818    now: Instant,
819) -> bool {
820    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830    Report,
831    Restart,
832    Alert,
833}
834
835impl fmt::Display for HealthAction {
836    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837        f.write_str(match self {
838            Self::Report => "report",
839            Self::Restart => "restart",
840            Self::Alert => "alert",
841        })
842    }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847    /// Optional loopback HTTP endpoint for a managed non-wire process.
848    /// Changing it applies live on rescan; the process protocol changes only
849    /// at its next spawn.
850    pub http: Option<String>,
851    pub cadence: Duration,
852    pub deadline: Duration,
853    pub failure_threshold: u32,
854    pub on_degraded: HealthAction,
855    pub on_failing: HealthAction,
856    pub critical: bool,
857}
858
859impl Default for HealthConfig {
860    fn default() -> Self {
861        Self {
862            http: None,
863            cadence: DEFAULT_HEALTH_CADENCE,
864            deadline: DEFAULT_HEALTH_DEADLINE,
865            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866            on_degraded: HealthAction::Report,
867            on_failing: HealthAction::Report,
868            critical: false,
869        }
870    }
871}
872
873/// The supervisor's view of one module's health, relayed to clients over
874/// channel-0 and rendered by `ck health`.
875///
876/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
877/// stated here rather than only at the wire type a consumer reads. A reader can
878/// look up what `None` means; only a writer can silently change it, and the
879/// writer has no reason to go looking at a downstream contract before editing.
880///
881/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
882/// back to `None` on re-registration precisely so a respawned module does not
883/// carry its predecessor's timestamp — so an old value and an absent one call for
884/// opposite readings, and anything that defaulted this to a number would make a
885/// never-probed module indistinguishable from one probed at the epoch.
886///
887/// `detail` and `metrics` are `None` when the module published none on this
888/// probe, which does not mean it reported nothing wrong — it is also the shape
889/// when the probe never reached it. `last_probe_ms` is what separates those.
890#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892    pub status: SupervisorHealthStatus,
893    pub last_probe_ms: Option<u64>,
894    pub detail: Option<String>,
895    pub metrics: Option<Value>,
896    pub consecutive_failures: u32,
897    /// Number of replies received after a recurring health probe's deadline.
898    /// Unlike a timeout, every increment proves the module was alive.
899    pub late_answer_count: u64,
900    /// End-to-end latency of the newest late reply, measured from probe start.
901    pub last_late_answer_latency_ms: Option<u64>,
902    pub last_action: Option<String>,
903    /// Set together with `last_action`; the pair moves as one, and both being
904    /// absent means no escalation has ever been taken rather than that the last
905    /// one succeeded.
906    pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910    fn default() -> Self {
911        Self {
912            status: SupervisorHealthStatus::Unknown,
913            last_probe_ms: None,
914            detail: None,
915            metrics: None,
916            consecutive_failures: 0,
917            late_answer_count: 0,
918            last_late_answer_latency_ms: None,
919            last_action: None,
920            last_action_ms: None,
921        }
922    }
923}
924
925/// Typed lifecycle state for a supervised module.
926#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928    Starting,
929    Running,
930    Unresponsive,
931    Restarting,
932    Draining,
933    Stopped,
934    Failed,
935    Disabled,
936}
937
938impl fmt::Display for ModuleState {
939    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940        f.write_str(match self {
941            Self::Starting => "starting",
942            Self::Running => "running",
943            Self::Unresponsive => "unresponsive",
944            Self::Restarting => "restarting",
945            Self::Draining => "draining",
946            Self::Stopped => "stopped",
947            Self::Failed => "failed",
948            Self::Disabled => "disabled",
949        })
950    }
951}
952
953/// Supervisor classification of a child-process exit.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956    Clean,
957    Crash,
958    DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962    fn from(kind: ExitKind) -> Self {
963        match kind {
964            ExitKind::Clean => Self::Clean,
965            ExitKind::Crash => Self::Crash,
966            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967        }
968    }
969}
970
971/// Exact process identity retained when a supervised module registers its
972/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
973#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975    pub(crate) pid: u32,
976    pub(crate) start_time: u64,
977}
978
979/// Last observed child exit, if any.
980#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982    pub kind: ExitKind,
983    pub code: Option<i32>,
984    pub signal: Option<i32>,
985    pub at_ms: u64,
986}
987
988/// Point-in-time module status answerable by subc without forwarding to the
989/// module process.
990#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992    pub module_id: String,
993    pub state: ModuleState,
994    pub enabled: bool,
995    pub process_alive: bool,
996    pub registration_active: bool,
997    /// The module's declared wire protocol, carried beside `live` because it is
998    /// what makes `live` readable: the two fields answer one question together.
999    /// While a process is alive this is its launch declaration, not a later
1000    /// pending-reload edit. When down it is the configured next launch protocol.
1001    pub protocol: ModuleProtocol,
1002    /// Whether the module is serving, under the strongest definition the daemon
1003    /// can assert for its protocol.
1004    ///
1005    /// A subc module must also be REGISTERED: its process being alive says
1006    /// nothing about whether it can take a request. A `protocol: "none"` module
1007    /// never registers, so that term is dropped and this falls back to "enabled,
1008    /// running, and the process the daemon launched is alive" -- which is all
1009    /// the daemon observes about a process that speaks no subc wire. It stays a
1010    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
1011    /// rather than printing it bare.
1012    pub live: bool,
1013    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
1014    /// restarts have already released their slot, so this count can go down
1015    /// without anybody touching the module.
1016    pub restart_count: u32,
1017    /// Replacement processes spawned over this module's entire supervisor lifetime;
1018    /// unlike `restart_count`, this value is never reset by an operator action
1019    /// and never falls out of a window.
1020    pub lifetime_restarts: u32,
1021    pub spawn_generation: u64,
1022    /// The budget `restart_count` is spent against. Carried alongside the count
1023    /// because the count alone does not say how close the module is to being
1024    /// disabled, and reporting one without the other is what makes an
1025    /// about-to-be-retired module look ordinary.
1026    pub max_restarts: u32,
1027    /// The span `restart_count` is counted over. Carried with the pair above for
1028    /// the same reason they are carried together: "2 of 3" means one thing for a
1029    /// ten-minute window and something else entirely for a lifetime.
1030    pub restart_window: Duration,
1031    /// Effective drain and restart timing policy used by this running module.
1032    /// These values are carried together with the restart budget so status
1033    /// readers can compare configured intent with what the supervisor applied.
1034    pub drain_timeout: Duration,
1035    pub restart_backoff: Duration,
1036    pub restart_max_backoff: Duration,
1037    /// The module's process. On macOS this stays absent while the `ck-subc`
1038    /// launch trampoline is still running in that pid, and appears once the
1039    /// exec acknowledgement confirms the module image has replaced it. Launch
1040    /// time and the supervisor's own hold on the process are unaffected.
1041    pub pid: Option<u32>,
1042    pub spawned_at_ms: Option<u64>,
1043    pub spawned_from: Option<PathBuf>,
1044    pub process_start_time: Option<u64>,
1045    pub last_exit: Option<ExitReport>,
1046    pub health: ModuleHealthStatus,
1047}
1048
1049#[derive(Debug, Clone, PartialEq)]
1050struct SupervisorSnapshot {
1051    state: ModuleState,
1052    enabled: bool,
1053    process_alive: bool,
1054    spawned_protocol: Option<ModuleProtocol>,
1055    spawn_failure: Option<String>,
1056    /// When each crash restart was spent, oldest first. This IS the crash
1057    /// budget: its in-window length is the count an operator sees and the count
1058    /// the restart decision is made against, so there is no second counter that
1059    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1060    /// operator actions that used to zero the old lifetime counter.
1061    crash_restarts: VecDeque<Instant>,
1062    lifetime_restarts: u32,
1063    /// Successful child spawns in this daemon incarnation.
1064    ///
1065    /// `lifetime_restarts` was considered and rejected: it starts at zero
1066    /// (line 640), successful initial/operator spawns in `set_running` do not
1067    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1068    /// increments before a successful replacement exists (lines 604, 3846,
1069    /// and 3921), so a failed spawn can consume it. This counter moves only
1070    /// when a live PID is accepted below.
1071    spawn_generation: u64,
1072    pid: Option<u32>,
1073    #[cfg(target_os = "macos")]
1074    report_ready: Option<Arc<OnceLock<()>>>,
1075    /// Last reaped child, retained after current process facts are cleared.
1076    reaped_pid: Option<u32>,
1077    /// Whether the command-serving supervision loop has a scheduled respawn.
1078    respawn_pending: bool,
1079    /// A second restart is waiting for the replacement already scheduled.
1080    coalesced_restart_pending: bool,
1081    spawned_at_ms: Option<u64>,
1082    spawned_from: Option<PathBuf>,
1083    spawned_file_identity: Option<SpawnedFileIdentity>,
1084    process_start_time: Option<u64>,
1085    deliberate_severance: Option<ProcessIdentity>,
1086    last_exit: Option<ExitReport>,
1087    /// Diagnostic attached to the next drain's terminal record, if any.
1088    drain_disposition_detail: Option<String>,
1089    health: ModuleHealthStatus,
1090    /// Whether the current process was started as a swap candidate and so
1091    /// lives in the module's alternate cgroup. The next swap's candidate takes
1092    /// the other one, so the two processes of a swap never share a cgroup. A
1093    /// plain spawn always uses the primary cgroup.
1094    in_alternate_slot: bool,
1095    /// Whether the current `Draining` state ends in a replacement process
1096    /// (restart, reload, health restart) rather than a stop. Only meaningful
1097    /// while `state` is `Draining`; every entry into that state rewrites it.
1098    /// It is what lets route.open answer the retryable `module_reloading` to a
1099    /// consumer that reaches a still-registered process mid-restart, instead of
1100    /// the `supervisor_not_live` a stop or disable deserves.
1101    draining_to_replace: bool,
1102    /// Whether a configuration update has been applied since the current
1103    /// process was spawned, so that process runs an older spec than the one
1104    /// the supervisor now holds. A queued restart is only coalesced into a
1105    /// fresher process when this is false: a restart requested to pick up a
1106    /// new configuration must not be satisfied by a process that predates it.
1107    configuration_updated_since_spawn: bool,
1108}
1109
1110impl SupervisorSnapshot {
1111    /// The pid that status, provenance and resource readings may report. While
1112    /// the launch trampoline still runs in the pid, reading its executable or
1113    /// resource use would describe `ck-subc`, not the module, so none is
1114    /// reported until the exec acknowledgement confirms the module image.
1115    fn reported_pid(&self) -> Option<u32> {
1116        #[cfg(target_os = "macos")]
1117        if self
1118            .report_ready
1119            .as_ref()
1120            .is_some_and(|ready| ready.get().is_none())
1121        {
1122            return None;
1123        }
1124        self.pid
1125    }
1126
1127    fn starting() -> Self {
1128        Self::new(ModuleState::Starting, true)
1129    }
1130
1131    fn disabled() -> Self {
1132        Self::new(ModuleState::Disabled, false)
1133    }
1134
1135    fn failed() -> Self {
1136        Self::new(ModuleState::Failed, true)
1137    }
1138
1139    /// Crash restarts still inside `window`, having dropped the ones that are
1140    /// not. Pruning on read is what makes the budget a rate: an instant older
1141    /// than the window stops holding a slot the moment anybody counts.
1142    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1143        while let Some(oldest) = self.crash_restarts.front() {
1144            if now.duration_since(*oldest) > window {
1145                self.crash_restarts.pop_front();
1146            } else {
1147                break;
1148            }
1149        }
1150        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1151    }
1152
1153    /// Spend one unit of the crash budget and record the restart in the ledger.
1154    ///
1155    /// The ring is bounded by the cap because more than `max_restarts` in-window
1156    /// instants can never be reached (the caller refuses the restart first), so
1157    /// anything beyond that is an unbounded queue waiting to happen.
1158    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1159        self.crash_restarts.push_back(now);
1160        while self.crash_restarts.len() > policy.max_restarts as usize {
1161            self.crash_restarts.pop_front();
1162        }
1163        self.lifetime_restarts += 1;
1164    }
1165
1166    /// Reserve one crash-restart slot and calculate the delay before respawning.
1167    /// The count is captured before recording this restart, so the first retry
1168    /// uses the base delay and each later in-window retry escalates once.
1169    fn next_crash_restart(
1170        &mut self,
1171        policy: &RestartPolicy,
1172        now: Instant,
1173    ) -> Option<CrashRestartSchedule> {
1174        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1175        if restart_in_window >= policy.max_restarts {
1176            return None;
1177        }
1178        self.record_crash_restart(policy, now);
1179        Some(CrashRestartSchedule {
1180            restart_in_window,
1181            delay: policy.delay_for_restart(restart_in_window),
1182        })
1183    }
1184
1185    /// Give the module its full budget back, as an operator restart, reload, or
1186    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1187    /// ledger of what actually happened, and an operator action does not unmake
1188    /// the crashes.
1189    fn clear_crash_restarts(&mut self) {
1190        self.crash_restarts.clear();
1191    }
1192
1193    fn new(state: ModuleState, enabled: bool) -> Self {
1194        Self {
1195            state,
1196            enabled,
1197            process_alive: false,
1198            spawned_protocol: None,
1199            spawn_failure: None,
1200            crash_restarts: VecDeque::new(),
1201            lifetime_restarts: 0,
1202            spawn_generation: 0,
1203            pid: None,
1204            #[cfg(target_os = "macos")]
1205            report_ready: None,
1206            reaped_pid: None,
1207            respawn_pending: false,
1208            coalesced_restart_pending: false,
1209            spawned_at_ms: None,
1210            spawned_from: None,
1211            spawned_file_identity: None,
1212            process_start_time: None,
1213            deliberate_severance: None,
1214            last_exit: None,
1215            drain_disposition_detail: None,
1216            health: ModuleHealthStatus::default(),
1217            in_alternate_slot: false,
1218            draining_to_replace: false,
1219            configuration_updated_since_spawn: false,
1220        }
1221    }
1222}
1223
1224type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1225
1226type SpawnSubscriberKey = (ConnectionId, u64);
1227
1228#[derive(Debug)]
1229struct SpawnSubscriber {
1230    version: u8,
1231    frames: mpsc::Sender<Frame>,
1232    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1233    /// from which event. The full frame channel cannot carry that news, so it
1234    /// travels beside it; see `SpawnEventFeed::subscribe`.
1235    lagged: Option<oneshot::Sender<SpawnCursor>>,
1236}
1237
1238#[derive(Debug)]
1239struct SpawnEventState {
1240    daemon_incarnation: String,
1241    seq: u64,
1242    capacity: usize,
1243    live: HashMap<String, LiveSpawn>,
1244    generations: HashMap<String, u64>,
1245    events: VecDeque<SpawnEvent>,
1246    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1247}
1248
1249impl Default for SpawnEventState {
1250    fn default() -> Self {
1251        Self {
1252            daemon_incarnation: "unconfigured".to_string(),
1253            seq: 0,
1254            capacity: SPAWN_EVENT_RING_CAPACITY,
1255            live: HashMap::new(),
1256            generations: HashMap::new(),
1257            events: VecDeque::new(),
1258            subscribers: HashMap::new(),
1259        }
1260    }
1261}
1262
1263#[derive(Debug, Clone, Default)]
1264struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1265
1266#[derive(Debug, Clone, PartialEq, Eq)]
1267pub(crate) enum SpawnSubscribeRefusal {
1268    ForeignIncarnation { current: String },
1269    TooOld { oldest: SpawnCursor },
1270    Frame(String),
1271}
1272
1273impl SpawnEventFeed {
1274    fn configure_incarnation(&self, daemon_incarnation: String) {
1275        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1276        state.daemon_incarnation = daemon_incarnation;
1277        state.seq = 0;
1278        state.live.clear();
1279        state.generations.clear();
1280        state.events.clear();
1281        state.subscribers.clear();
1282    }
1283
1284    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1285        SpawnCursor {
1286            daemon_incarnation: state.daemon_incarnation.clone(),
1287            seq: state.seq,
1288        }
1289    }
1290
1291    fn snapshot(&self) -> SpawnSnapshot {
1292        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1293        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1294        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1295        SpawnSnapshot {
1296            cursor: Self::cursor(&state),
1297            ring_bound: state.capacity as u64,
1298            live,
1299        }
1300    }
1301
1302    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1303        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1304        let generation = state
1305            .generations
1306            .get(module_id)
1307            .copied()
1308            .unwrap_or(0)
1309            .checked_add(1)
1310            .expect("spawn generation exhausted");
1311        state.generations.insert(module_id.to_string(), generation);
1312        let live = LiveSpawn {
1313            module_id: module_id.to_string(),
1314            spawn_generation: generation,
1315            pid,
1316            spawned_at_ms,
1317        };
1318        state.live.insert(module_id.to_string(), live);
1319        Self::emit_locked(
1320            &mut state,
1321            SpawnEventKind::Spawned,
1322            module_id.to_string(),
1323            generation,
1324            pid,
1325            None,
1326            None,
1327        );
1328        generation
1329    }
1330
1331    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1332        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1333        let Some(live) = state.live.remove(module_id) else {
1334            warn!(
1335                module_id,
1336                "terminal record had no live spawn event identity"
1337            );
1338            return;
1339        };
1340        Self::emit_locked(
1341            &mut state,
1342            SpawnEventKind::Exited,
1343            module_id.to_string(),
1344            live.spawn_generation,
1345            live.pid,
1346            exit_code,
1347            exit_signal,
1348        );
1349    }
1350
1351    /// Report the exit of a process that a swap has already replaced.
1352    ///
1353    /// `emit_exited` removes the module's live entry, which after a swap's
1354    /// cutover describes the promoted candidate, not the old process now
1355    /// exiting. This emits the old generation's exit and leaves the live entry
1356    /// alone unless it still names that generation.
1357    fn emit_superseded_exited(
1358        &self,
1359        module_id: &str,
1360        spawn_generation: u64,
1361        pid: u32,
1362        exit_code: Option<i32>,
1363        exit_signal: Option<i32>,
1364    ) {
1365        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1366        if state
1367            .live
1368            .get(module_id)
1369            .is_some_and(|live| live.spawn_generation == spawn_generation)
1370        {
1371            state.live.remove(module_id);
1372        }
1373        Self::emit_locked(
1374            &mut state,
1375            SpawnEventKind::Exited,
1376            module_id.to_string(),
1377            spawn_generation,
1378            pid,
1379            exit_code,
1380            exit_signal,
1381        );
1382    }
1383
1384    #[allow(clippy::too_many_arguments)]
1385    fn emit_locked(
1386        state: &mut SpawnEventState,
1387        kind: SpawnEventKind,
1388        module_id: String,
1389        spawn_generation: u64,
1390        pid: u32,
1391        exit_code: Option<i32>,
1392        exit_signal: Option<i32>,
1393    ) {
1394        state.seq = state
1395            .seq
1396            .checked_add(1)
1397            .expect("spawn event sequence exhausted");
1398        let event = SpawnEvent {
1399            cursor: Self::cursor(state),
1400            kind,
1401            module_id,
1402            spawn_generation,
1403            pid,
1404            exit_code,
1405            exit_signal,
1406        };
1407        state.events.push_back(event.clone());
1408        while state.events.len() > state.capacity {
1409            state.events.pop_front();
1410        }
1411        let body = match serde_json::to_vec(&event) {
1412            Ok(body) => body,
1413            Err(error) => {
1414                error!(%error, "failed to serialize supervisor spawn event");
1415                return;
1416            }
1417        };
1418        state.subscribers.retain(|(connection_id, corr), subscriber| {
1419            let frame = Frame::build_with_version(
1420                subscriber.version,
1421                FrameType::StreamData,
1422                control_flags(),
1423                0,
1424                0,
1425                *corr,
1426                body.clone(),
1427            );
1428            match frame {
1429                Ok(frame) => {
1430                    if subscriber.frames.try_send(frame).is_ok() {
1431                        true
1432                    } else {
1433                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1434                        if let Some(lagged) = subscriber.lagged.take() {
1435                            let _ = lagged.send(event.cursor.clone());
1436                        }
1437                        false
1438                    }
1439                }
1440                Err(error) => {
1441                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1442                    false
1443                }
1444            }
1445        });
1446    }
1447
1448    fn subscribe(
1449        &self,
1450        connection_id: ConnectionId,
1451        corr: u64,
1452        version: u8,
1453        since: Option<SpawnCursor>,
1454        sink: FrameSink,
1455    ) -> Result<(), SpawnSubscribeRefusal> {
1456        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1457        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1458        {
1459            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1460            let replay = if let Some(since) = since {
1461                if since.daemon_incarnation != state.daemon_incarnation {
1462                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1463                        current: state.daemon_incarnation.clone(),
1464                    });
1465                }
1466                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1467                    if since.seq < oldest.seq.saturating_sub(1) {
1468                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1469                    }
1470                }
1471                state
1472                    .events
1473                    .iter()
1474                    .filter(|event| event.cursor.seq > since.seq)
1475                    .cloned()
1476                    .collect::<Vec<_>>()
1477            } else {
1478                Vec::new()
1479            };
1480            for event in replay {
1481                let body = serde_json::to_vec(&event)
1482                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1483                let frame = Frame::build_with_version(
1484                    version,
1485                    FrameType::StreamData,
1486                    control_flags(),
1487                    0,
1488                    0,
1489                    corr,
1490                    body,
1491                )
1492                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1493                frames
1494                    .try_send(frame)
1495                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1496            }
1497            state.subscribers.insert(
1498                (connection_id, corr),
1499                SpawnSubscriber {
1500                    version,
1501                    frames,
1502                    lagged: Some(lagged),
1503                },
1504            );
1505        }
1506        // The lagged terminal is sent here, by the forwarder, rather than by
1507        // the emitter: at the moment of the drop the subscriber's own channel
1508        // is full, and writing to the connection sink directly from the emitter
1509        // would put the Error AHEAD of the events still queued in that channel
1510        // (and the emitter holds the feed lock, so it cannot await the sink).
1511        // Dropping the subscriber drops the only sender, so `recv` drains every
1512        // queued event and then returns `None`; only then is the Error sent, so
1513        // the client sees each event it can keep, then the reason it was cut.
1514        // Cancel and connection removal drop the oneshot unsent, so they end
1515        // the stream with no Error.
1516        tokio::spawn(async move {
1517            while let Some(frame) = receiver.recv().await {
1518                if sink.send(frame).await.is_err() {
1519                    return;
1520                }
1521            }
1522            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1523                return;
1524            };
1525            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1526                Ok(frame) => {
1527                    let _ = sink.send(frame).await;
1528                }
1529                Err(error) => {
1530                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1531                }
1532            }
1533        });
1534        Ok(())
1535    }
1536
1537    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1538        let Some(subscriber) = self
1539            .0
1540            .lock()
1541            .unwrap_or_else(|p| p.into_inner())
1542            .subscribers
1543            .remove(&(connection_id, corr))
1544        else {
1545            return false;
1546        };
1547        if let Ok(frame) = Frame::build_with_version(
1548            subscriber.version,
1549            FrameType::StreamEnd,
1550            control_flags(),
1551            0,
1552            0,
1553            corr,
1554            Vec::new(),
1555        ) {
1556            tokio::spawn(async move {
1557                let _ = subscriber.frames.send(frame).await;
1558            });
1559        }
1560        true
1561    }
1562
1563    fn remove_connection(&self, connection_id: ConnectionId) {
1564        self.0
1565            .lock()
1566            .unwrap_or_else(|p| p.into_inner())
1567            .subscribers
1568            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1569    }
1570
1571    #[cfg(any(test, feature = "test-support"))]
1572    fn set_capacity(&self, capacity: usize) {
1573        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1574    }
1575
1576    #[cfg(any(test, feature = "test-support"))]
1577    fn subscriber_count(&self) -> usize {
1578        self.0
1579            .lock()
1580            .unwrap_or_else(|p| p.into_inner())
1581            .subscribers
1582            .len()
1583    }
1584}
1585
1586/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1587/// The terminal Error a lagged spawn subscriber receives after its queued events.
1588fn spawn_subscriber_lagged_frame(
1589    version: u8,
1590    corr: u64,
1591    first_undelivered: SpawnCursor,
1592) -> Result<Frame, String> {
1593    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1594        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1595        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1596            .to_string(),
1597        detail: Some(serde_json::json!({
1598            "first_undelivered_cursor": first_undelivered
1599        })),
1600    })
1601    .map_err(|error| error.to_string())?;
1602    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1603        .map_err(|error| error.to_string())
1604}
1605
1606pub trait ModuleProcessLiveness: Send + Sync {
1607    fn process_live(&self, module_id: &str) -> Option<bool>;
1608
1609    /// Whether the supervisor is replacing this module's process right now: an
1610    /// operator restart or reload, a health restart, or a crash respawn whose
1611    /// backoff is running. A module in that state is not live, but a consumer
1612    /// refused now should retry shortly rather than treat the target as gone.
1613    /// Stopped, failed, and disabled modules are not replacing.
1614    fn process_replacing(&self, _module_id: &str) -> bool {
1615        false
1616    }
1617}
1618
1619/// Shared process-liveness registry keyed by supervised `module_id`.
1620#[derive(Debug, Clone, Default)]
1621pub struct SupervisorProcessLiveness {
1622    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1623}
1624
1625impl SupervisorProcessLiveness {
1626    pub fn new() -> Self {
1627        Self::default()
1628    }
1629
1630    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1631        let mut snapshots = self
1632            .snapshots
1633            .lock()
1634            .unwrap_or_else(|poisoned| poisoned.into_inner());
1635        snapshots.insert(module_id, snapshot);
1636    }
1637
1638    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1639        let mut snapshots = self
1640            .snapshots
1641            .lock()
1642            .unwrap_or_else(|poisoned| poisoned.into_inner());
1643        let is_current = snapshots
1644            .get(module_id)
1645            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1646            .unwrap_or(false);
1647        if is_current {
1648            snapshots.remove(module_id);
1649        }
1650    }
1651}
1652
1653impl ModuleProcessLiveness for SupervisorProcessLiveness {
1654    fn process_live(&self, module_id: &str) -> Option<bool> {
1655        let snapshot = {
1656            let snapshots = self
1657                .snapshots
1658                .lock()
1659                .unwrap_or_else(|poisoned| poisoned.into_inner());
1660            snapshots.get(module_id).cloned()
1661        }?;
1662        let snapshot = snapshot
1663            .lock()
1664            .unwrap_or_else(|poisoned| poisoned.into_inner());
1665        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1666    }
1667
1668    fn process_replacing(&self, module_id: &str) -> bool {
1669        let Some(snapshot) = self
1670            .snapshots
1671            .lock()
1672            .unwrap_or_else(|poisoned| poisoned.into_inner())
1673            .get(module_id)
1674            .cloned()
1675        else {
1676            return false;
1677        };
1678        let snapshot = snapshot
1679            .lock()
1680            .unwrap_or_else(|poisoned| poisoned.into_inner());
1681        snapshot.enabled
1682            && match snapshot.state {
1683                ModuleState::Restarting => true,
1684                ModuleState::Draining => snapshot.draining_to_replace,
1685                ModuleState::Starting
1686                | ModuleState::Running
1687                | ModuleState::Unresponsive
1688                | ModuleState::Stopped
1689                | ModuleState::Failed
1690                | ModuleState::Disabled => false,
1691            }
1692    }
1693}
1694
1695#[cfg(test)]
1696#[derive(Debug, Default)]
1697struct ReloadExitRecordGate {
1698    reached: tokio::sync::Notify,
1699    resume: tokio::sync::Notify,
1700}
1701
1702#[derive(Debug, Clone, Copy)]
1703enum RespawnKind {
1704    Spawn,
1705    Reload,
1706}
1707
1708#[derive(Debug, Clone, Copy)]
1709struct PendingRespawn {
1710    deadline: Instant,
1711    kind: RespawnKind,
1712}
1713
1714type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1715
1716#[derive(Debug, Clone)]
1717struct SupervisorRuntimeConfig {
1718    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1719    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1720    /// A reload acknowledges completion only after its replacement registers.
1721    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1722    restart_policy: RestartPolicy,
1723    /// This module's RESOLVED drain budget: per-module config when present,
1724    /// else `default_drain_timeout`.
1725    drain_timeout: Duration,
1726    /// Shared with the status handle so the attested value changes atomically
1727    /// when a rescan updates the running drain policy.
1728    effective_drain_timeout: Arc<Mutex<Duration>>,
1729    /// The supervisor-wide fallback, kept so a configuration update that
1730    /// REMOVES the per-module override can re-resolve to it.
1731    default_drain_timeout: Duration,
1732    health: HealthConfig,
1733    connection_file_path: Option<PathBuf>,
1734    capture_logs_dir: Option<PathBuf>,
1735    forwarding: Option<Arc<ForwardingTable>>,
1736    /// The shared handle, so every spawn path (initial, restart, reload) records the
1737    /// reserved-module launch nonce the HELLO verifier checks against.
1738    supervisor_handle: Option<SupervisorHandle>,
1739    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1740    /// status queries.
1741    ///
1742    /// One ring per module, held across every respawn. The lines explaining an exit
1743    /// are written BEFORE that exit, so a ring recreated per process would be empty
1744    /// exactly when it is asked for.
1745    stderr_ring: Arc<Mutex<StderrRing>>,
1746    terminal_ring: Arc<Mutex<TerminalRing>>,
1747    spawn_events: SpawnEventFeed,
1748    child_roster: ChildRoster,
1749    #[cfg(target_os = "linux")]
1750    cgroup_placement: Option<subc_cgroup::Placement>,
1751    #[cfg(test)]
1752    test_seed_stale_facts_before_enable_spawn: bool,
1753    #[cfg(test)]
1754    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1755}
1756
1757#[derive(Debug, Clone, PartialEq, Eq)]
1758struct SupervisedConfiguration {
1759    spec: ModuleSpec,
1760    health: HealthConfig,
1761}
1762
1763/// Shared daemon lookup table for supervised module handles.
1764///
1765/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1766/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1767/// launch nonces recorded at spawn are checked by the same daemon instance.
1768#[derive(Debug, Clone, Default)]
1769pub struct SupervisorHandle {
1770    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1771    /// Module ids the supervisor has taken on. An id is added BEFORE the
1772    /// module's first process is spawned and removed only when the module
1773    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1774    /// the keys of `modules`.
1775    ///
1776    /// `modules` cannot answer "is this module configured?" on its own: a
1777    /// [`SupervisedModule`] only exists once its process has been spawned, and
1778    /// a fast child can connect, register, sync its scopes and ask about them
1779    /// before the supervisor has inserted it. Answering "not configured" in that
1780    /// gap makes scope admission refuse with the terminal "will never sync"
1781    /// instead of the retryable "has not synced yet".
1782    configured_ids: Arc<Mutex<HashSet<String>>>,
1783    spawn_events: SpawnEventFeed,
1784    /// The current expected launch nonce for each reserved module_id. Set when the
1785    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1786    /// non-reserved module never has an entry here and is never nonce-checked.
1787    /// Reserved module ids and the nonce that authorizes their next HELLO.
1788    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1789    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1790    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1791    /// had NO entry and admitted anyone: the reservation protected the nonce
1792    /// holder, not the NAME (found live by CKCRED's canary probe registering
1793    /// against a reserved scratch id).
1794    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1795    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1796    ///
1797    /// This is deliberately in-memory only: subc is state-free across daemon
1798    /// restarts, and the tombstone only explains the hours-after-removal window
1799    /// while this executing daemon is still alive. Do not persist it in a store.
1800    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1801    /// The current launch nonce for every supervised spawn. This is separate from
1802    /// reserved_nonces because consumer route.open attestation applies to all spawned
1803    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1804    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1805    /// Reserved namespace prefixes mapped to the supervised owner module whose
1806    /// current spawn nonce authorizes HELLO claims below the prefix.
1807    ///
1808    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1809    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1810    /// accidental collisions and lower-trust processes from squatting protected
1811    /// namespaces.
1812    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1813    /// Blue/green swaps in progress, by module id. An entry exists from just
1814    /// before the candidate process is spawned until the swap has failed, or
1815    /// has cut over and the old process is gone. While it exists, HELLO for the
1816    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1817    /// consumer attestation accepts both processes' nonces.
1818    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1819    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1820    promotion_observer: PromotionObserverSlot,
1821    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1822    /// this daemon-wide ordering, a rescan could retire or update a module while a
1823    /// concurrent reload still held its old handle and launch specification.
1824    operation_lock: Arc<AsyncMutex<()>>,
1825}
1826
1827/// Told when a swap has promoted its candidate to be the module's active
1828/// registration.
1829///
1830/// An ordinary HELLO runs the control plane's registration side effects (the
1831/// capability cache, the deny census, the requirement recompute) as it
1832/// registers. A swap candidate's HELLO does not, because it is not routable;
1833/// promotion is when those must run instead, and promotion happens in the
1834/// supervisor, which has no other way into the control handler.
1835pub(crate) trait SwapPromotionObserver: Send + Sync {
1836    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1837}
1838
1839/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1840/// control handler) owns this handle, so a strong reference back would be a
1841/// cycle that keeps both alive.
1842#[derive(Clone, Default)]
1843struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1844
1845impl fmt::Debug for PromotionObserverSlot {
1846    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1847        f.write_str("PromotionObserverSlot")
1848    }
1849}
1850
1851/// The nonces of one open swap.
1852#[derive(Debug, Clone)]
1853struct OpenSwap {
1854    /// The launch nonce minted for the candidate process. It is the swap
1855    /// token: the only thing that admits a HELLO into the candidate slot.
1856    candidate_nonce: String,
1857    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1858    /// here because cutover moves the module's recorded spawn nonce to the
1859    /// candidate while the incumbent is still draining and its consumers are
1860    /// still attesting with this one.
1861    incumbent_nonce: Option<String>,
1862    /// Set once a HELLO has been admitted with the swap token, so the token
1863    /// admits one registration and cannot be replayed after cutover empties
1864    /// the candidate slot.
1865    candidate_admitted: bool,
1866}
1867
1868/// What the swap gate says about a HELLO. See
1869/// [`SupervisorHandle::swap_hello_admission`].
1870#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1871pub(crate) enum SwapHelloAdmission {
1872    /// No swap is open for the id (or the HELLO carries the incumbent's own
1873    /// nonce); the ordinary gates decide.
1874    NotSwapping,
1875    /// The HELLO carries the swap token: register it into the candidate slot.
1876    Candidate,
1877    /// A swap is open and the HELLO carries a nonce the supervisor did not
1878    /// mint for this id, no nonce, or a token already used.
1879    Refused,
1880}
1881
1882#[derive(Debug, Clone, PartialEq, Eq)]
1883pub(crate) enum ReservedHelloRejection {
1884    Exact {
1885        module_id: String,
1886    },
1887    Prefix {
1888        prefix: String,
1889        owner_module_id: String,
1890    },
1891}
1892
1893impl SupervisorHandle {
1894    pub fn new() -> Self {
1895        Self::default()
1896    }
1897
1898    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1899        self.spawn_events.snapshot()
1900    }
1901
1902    pub(crate) fn subscribe_spawns(
1903        &self,
1904        connection_id: ConnectionId,
1905        corr: u64,
1906        version: u8,
1907        since: Option<SpawnCursor>,
1908        sink: FrameSink,
1909    ) -> Result<(), SpawnSubscribeRefusal> {
1910        self.spawn_events
1911            .subscribe(connection_id, corr, version, since, sink)
1912    }
1913
1914    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1915        self.spawn_events.cancel(connection_id, corr)
1916    }
1917
1918    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1919        self.spawn_events.remove_connection(connection_id);
1920    }
1921
1922    #[cfg(any(test, feature = "test-support"))]
1923    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1924        assert!(capacity > 0, "spawn event capacity must be non-zero");
1925        self.spawn_events.set_capacity(capacity);
1926    }
1927
1928    #[cfg(any(test, feature = "test-support"))]
1929    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1930        self.spawn_events.subscriber_count()
1931    }
1932
1933    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1934    /// a respawn invalidates stale consumer identities.
1935    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1936        self.spawn_nonces
1937            .lock()
1938            .unwrap_or_else(|poisoned| poisoned.into_inner())
1939            .insert(module_id.to_string(), nonce);
1940    }
1941
1942    /// Record the launch nonce expected from the next HELLO for a reserved module,
1943    /// replacing any prior nonce (a respawn invalidates the previous one).
1944    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1945        self.reserved_nonces
1946            .lock()
1947            .unwrap_or_else(|poisoned| poisoned.into_inner())
1948            .insert(module_id.to_string(), Some(nonce));
1949    }
1950
1951    /// Record namespace prefixes owned by a supervised module.
1952    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1953        let mut owners = self
1954            .reserved_prefix_owners
1955            .lock()
1956            .unwrap_or_else(|poisoned| poisoned.into_inner());
1957        owners.retain(|_, owner| owner != owner_module_id);
1958        for prefix in prefixes {
1959            owners.insert(prefix.clone(), owner_module_id.to_string());
1960        }
1961    }
1962
1963    /// The launch nonce most recently minted for a module's spawn, if any.
1964    #[cfg(test)]
1965    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1966        self.spawn_nonces
1967            .lock()
1968            .unwrap_or_else(|poisoned| poisoned.into_inner())
1969            .get(module_id)
1970            .cloned()
1971    }
1972
1973    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1974        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1975        let spawn_nonce = self
1976            .spawn_nonces
1977            .lock()
1978            .unwrap_or_else(|poisoned| poisoned.into_inner())
1979            .get(&spec.module_id)
1980            .cloned();
1981        let mut reserved_nonces = self
1982            .reserved_nonces
1983            .lock()
1984            .unwrap_or_else(|poisoned| poisoned.into_inner());
1985        if spec.reserved {
1986            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1987            // reserved name whose module has never spawned has no legitimate
1988            // holder, and the entry's absence is what used to leave the name
1989            // open to the first claimant.
1990            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1991        }
1992        drop(reserved_nonces);
1993        // A later unreserved declaration must not silently unreserve an id that
1994        // was retained after its reserved configuration was removed. The explicit
1995        // release ceremony is the only operation that retires that gate.
1996        self.removal_tombstones
1997            .lock()
1998            .unwrap_or_else(|poisoned| poisoned.into_inner())
1999            .remove(&spec.module_id);
2000    }
2001
2002    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
2003    /// authorized only by its expected nonce; otherwise a matching reserved prefix
2004    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
2005    /// with no matching prefix are always authorized.
2006    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2007        self.reserved_hello_rejection(module_id, presented)
2008            .is_none()
2009    }
2010
2011    pub(crate) fn reserved_hello_rejection(
2012        &self,
2013        module_id: &str,
2014        presented: Option<&str>,
2015    ) -> Option<ReservedHelloRejection> {
2016        let nonces = self
2017            .reserved_nonces
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner());
2020        if let Some(expected) = nonces.get(module_id) {
2021            // `None` = reserved with no legitimate holder: refuse every
2022            // presentation, because no process can hold a nonce that was never
2023            // minted. Only a real minted nonce admits, in constant time.
2024            let authorized = match expected {
2025                Some(expected) => {
2026                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2027                }
2028                None => false,
2029            };
2030            if authorized {
2031                return None;
2032            }
2033            return Some(ReservedHelloRejection::Exact {
2034                module_id: module_id.to_string(),
2035            });
2036        }
2037        drop(nonces);
2038
2039        let matched_prefix = self
2040            .reserved_prefix_owners
2041            .lock()
2042            .unwrap_or_else(|poisoned| poisoned.into_inner())
2043            .iter()
2044            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2045            .max_by_key(|(prefix, _)| prefix.len())
2046            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2047        let (prefix, owner_module_id) = matched_prefix?;
2048
2049        let authorized = presented.is_some_and(|presented| {
2050            self.spawn_nonces
2051                .lock()
2052                .unwrap_or_else(|poisoned| poisoned.into_inner())
2053                .get(&owner_module_id)
2054                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2055                // While the owner is being swapped, children started by
2056                // either of its two processes hold that process's nonce.
2057                || self.swap_nonce_matches(&owner_module_id, presented)
2058        });
2059        if authorized {
2060            None
2061        } else {
2062            Some(ReservedHelloRejection::Prefix {
2063                prefix,
2064                owner_module_id,
2065            })
2066        }
2067    }
2068
2069    /// Whether a consumer connection proved it came from a daemon-spawned module.
2070    ///
2071    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2072    /// accepted only for module ids the supervisor has spawned.
2073    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2074        if presented.is_empty() {
2075            return false;
2076        }
2077        let nonces = self
2078            .spawn_nonces
2079            .lock()
2080            .unwrap_or_else(|poisoned| poisoned.into_inner());
2081        let current = nonces
2082            .get(module_id)
2083            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2084        drop(nonces);
2085        // During a swap two processes of the module are alive, and a consumer
2086        // started by either one presents that process's nonce. Accepting only
2087        // the recorded one would fail the incumbent's consumers for the whole
2088        // overlap once cutover moves the record to the candidate.
2089        current || self.swap_nonce_matches(module_id, presented)
2090    }
2091
2092    /// Whether `presented` is either nonce of an open swap for `module_id`.
2093    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2094        let swaps = self
2095            .swaps
2096            .lock()
2097            .unwrap_or_else(|poisoned| poisoned.into_inner());
2098        swaps.get(module_id).is_some_and(|swap| {
2099            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2100                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2101                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2102                })
2103        })
2104    }
2105
2106    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2107    /// Called before the candidate process exists.
2108    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2109        let incumbent_nonce = self
2110            .spawn_nonces
2111            .lock()
2112            .unwrap_or_else(|poisoned| poisoned.into_inner())
2113            .get(module_id)
2114            .cloned();
2115        self.swaps
2116            .lock()
2117            .unwrap_or_else(|poisoned| poisoned.into_inner())
2118            .insert(
2119                module_id.to_string(),
2120                OpenSwap {
2121                    candidate_nonce,
2122                    incumbent_nonce,
2123                    candidate_admitted: false,
2124                },
2125            );
2126    }
2127
2128    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2129    /// the module's recorded one.
2130    pub(crate) fn close_swap(&self, module_id: &str) {
2131        self.swaps
2132            .lock()
2133            .unwrap_or_else(|poisoned| poisoned.into_inner())
2134            .remove(module_id);
2135    }
2136
2137    /// Install the observer told about swap promotions, replacing any earlier
2138    /// one.
2139    pub(crate) fn set_swap_promotion_observer(
2140        &self,
2141        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2142    ) {
2143        *self
2144            .promotion_observer
2145            .0
2146            .lock()
2147            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2148    }
2149
2150    /// Tell the installed observer, if it is still alive, that a swap promoted
2151    /// `registration`.
2152    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2153        let observer = self
2154            .promotion_observer
2155            .0
2156            .lock()
2157            .unwrap_or_else(|poisoned| poisoned.into_inner())
2158            .as_ref()
2159            .and_then(std::sync::Weak::upgrade);
2160        if let Some(observer) = observer {
2161            observer.swap_promoted(registration);
2162        }
2163    }
2164
2165    /// Whether a swap is open for `module_id`.
2166    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2167        self.swaps
2168            .lock()
2169            .unwrap_or_else(|poisoned| poisoned.into_inner())
2170            .contains_key(module_id)
2171    }
2172
2173    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2174    /// respawn would, once cutover has made the candidate the module's process.
2175    /// The swap stays open so the incumbent's nonce keeps attesting until the
2176    /// incumbent has drained and exited.
2177    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2178        let candidate_nonce = self
2179            .swaps
2180            .lock()
2181            .unwrap_or_else(|poisoned| poisoned.into_inner())
2182            .get(module_id)
2183            .map(|swap| swap.candidate_nonce.clone());
2184        let Some(nonce) = candidate_nonce else {
2185            return;
2186        };
2187        self.set_spawn_nonce(module_id, nonce.clone());
2188        if reserved {
2189            self.set_reserved_nonce(module_id, nonce);
2190        }
2191    }
2192
2193    /// The swap gate for a HELLO claiming `module_id`.
2194    ///
2195    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2196    /// presents the candidate nonce, which the reserved gate (holding the
2197    /// incumbent's nonce) would refuse as `reserved_module` before swap
2198    /// admission was ever reached. And it applies to unreserved ids too: for an
2199    /// unreserved id the only thing that ever stopped a second process claiming
2200    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2201    /// refusal a swap lifts for its candidate.
2202    ///
2203    /// The incumbent's own nonce falls through to the ordinary gates, which
2204    /// treat it as they always have (a live incumbent is refused as a
2205    /// duplicate). Anything else while a swap is open is refused, including an
2206    /// absent nonce.
2207    pub(crate) fn swap_hello_admission(
2208        &self,
2209        module_id: &str,
2210        presented: Option<&str>,
2211    ) -> SwapHelloAdmission {
2212        let swaps = self
2213            .swaps
2214            .lock()
2215            .unwrap_or_else(|poisoned| poisoned.into_inner());
2216        let Some(swap) = swaps.get(module_id) else {
2217            return SwapHelloAdmission::NotSwapping;
2218        };
2219        let Some(presented) = presented else {
2220            return SwapHelloAdmission::Refused;
2221        };
2222        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2223            return if swap.candidate_admitted {
2224                SwapHelloAdmission::Refused
2225            } else {
2226                SwapHelloAdmission::Candidate
2227            };
2228        }
2229        if swap
2230            .incumbent_nonce
2231            .as_deref()
2232            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2233        {
2234            return SwapHelloAdmission::NotSwapping;
2235        }
2236        SwapHelloAdmission::Refused
2237    }
2238
2239    /// Record that the swap token has registered a candidate, so it admits no
2240    /// second HELLO.
2241    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2242        if let Some(swap) = self
2243            .swaps
2244            .lock()
2245            .unwrap_or_else(|poisoned| poisoned.into_inner())
2246            .get_mut(module_id)
2247        {
2248            swap.candidate_admitted = true;
2249        }
2250    }
2251
2252    /// Test/support lookup for the current launch nonce of a supervised spawn.
2253    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2254        self.spawn_nonces
2255            .lock()
2256            .unwrap_or_else(|poisoned| poisoned.into_inner())
2257            .get(module_id)
2258            .cloned()
2259    }
2260
2261    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2262    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2263        self.reserved_nonces
2264            .lock()
2265            .unwrap_or_else(|poisoned| poisoned.into_inner())
2266            .get(module_id)
2267            .cloned()
2268            .flatten()
2269    }
2270
2271    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2272        // Normally already marked before the process was spawned; marking here
2273        // too keeps `configured_ids` a superset of the roster for any caller
2274        // that inserts a module directly.
2275        self.mark_configured(module.module_id());
2276        let mut modules = self
2277            .modules
2278            .lock()
2279            .unwrap_or_else(|poisoned| poisoned.into_inner());
2280        modules.insert(module.module_id().to_string(), module)
2281    }
2282
2283    /// Record that the supervisor has taken on `module_id`. Called before the
2284    /// module's first process is spawned, so that by the time that process can
2285    /// register, [`Self::is_configured`] already answers true.
2286    fn mark_configured(&self, module_id: &str) {
2287        self.configured_ids
2288            .lock()
2289            .unwrap_or_else(|poisoned| poisoned.into_inner())
2290            .insert(module_id.to_string());
2291    }
2292
2293    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2294    /// before it was ever put on the roster. A module already on the roster
2295    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2296    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2297        let modules = self
2298            .modules
2299            .lock()
2300            .unwrap_or_else(|poisoned| poisoned.into_inner());
2301        if !modules.contains_key(module_id) {
2302            self.configured_ids
2303                .lock()
2304                .unwrap_or_else(|poisoned| poisoned.into_inner())
2305                .remove(module_id);
2306        }
2307    }
2308
2309    /// Whether `module_id` is a module this daemon supervises: on the roster,
2310    /// or about to be (its process is being spawned right now).
2311    ///
2312    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2313    /// for scopes: a supervised module's process can register and sync before
2314    /// [`Self::get`] can return it, and in that window it is still a module
2315    /// that will sync, not one that never will.
2316    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2317        self.configured_ids
2318            .lock()
2319            .unwrap_or_else(|poisoned| poisoned.into_inner())
2320            .contains(module_id)
2321    }
2322
2323    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2324        let modules = self
2325            .modules
2326            .lock()
2327            .unwrap_or_else(|poisoned| poisoned.into_inner());
2328        modules.get(module_id).cloned()
2329    }
2330
2331    pub(crate) fn record_late_health_answer(
2332        &self,
2333        module_id: &str,
2334        latency_ms: u64,
2335    ) -> Result<bool, SuperviseError> {
2336        let Some(module) = self.get(module_id) else {
2337            return Ok(false);
2338        };
2339        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2340            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2341            state.health.last_late_answer_latency_ms = Some(latency_ms);
2342            // A late answer is an answer: the module served the probe, just past
2343            // the deadline. Leaving the miss streak in place while logging
2344            // "proves the module is alive" is how a CPU-starved module that
2345            // answers every probe a few seconds late still marches to the
2346            // threshold and gets killed — the exact kill class `NoAnswer` is
2347            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2348            // is degradation, and degradation reports; it does not restart.
2349            state.health.consecutive_failures = 0;
2350        })?;
2351        Ok(true)
2352    }
2353
2354    /// Arm the one-shot marker for the module process that this caller
2355    /// deliberately initiated severance against. Generic connection teardown
2356    /// must not call this:
2357    /// a surviving process would otherwise retain an exemption for a later
2358    /// genuine crash.
2359    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2360        let Some(module) = self.get(module_id) else {
2361            return Ok(false);
2362        };
2363        let snapshot = lock_snapshot(&module.inner.snapshot)?;
2364        let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2365            return Ok(false);
2366        };
2367        drop(snapshot);
2368        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2369    }
2370
2371    pub fn list(&self) -> Vec<SupervisedModule> {
2372        let modules = self
2373            .modules
2374            .lock()
2375            .unwrap_or_else(|poisoned| poisoned.into_inner());
2376        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2377        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2378        modules
2379    }
2380
2381    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2382        self.spawn_nonces
2383            .lock()
2384            .unwrap_or_else(|poisoned| poisoned.into_inner())
2385            .remove(module_id);
2386        self.close_swap(module_id);
2387        let mut reserved_nonces = self
2388            .reserved_nonces
2389            .lock()
2390            .unwrap_or_else(|poisoned| poisoned.into_inner());
2391        if reserved_nonces.contains_key(module_id) {
2392            // The old nonce must die with the removed process, but the exact-id
2393            // gate remains until an operator explicitly releases it.
2394            reserved_nonces.insert(module_id.to_string(), None);
2395        }
2396        drop(reserved_nonces);
2397        self.reserved_prefix_owners
2398            .lock()
2399            .unwrap_or_else(|poisoned| poisoned.into_inner())
2400            .retain(|_, owner| owner != module_id);
2401        let removed = self
2402            .modules
2403            .lock()
2404            .unwrap_or_else(|poisoned| poisoned.into_inner())
2405            .remove(module_id);
2406        self.configured_ids
2407            .lock()
2408            .unwrap_or_else(|poisoned| poisoned.into_inner())
2409            .remove(module_id);
2410        removed
2411    }
2412
2413    /// Remember a module removed by a non-preview rescan so route.open can
2414    /// distinguish that intentional removal from an unknown id.
2415    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2416        self.removal_tombstones
2417            .lock()
2418            .unwrap_or_else(|poisoned| poisoned.into_inner())
2419            .insert(module_id.to_string(), unix_ms_now());
2420    }
2421
2422    /// Return how long ago a rescan removed this module in milliseconds.
2423    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2424        self.removal_tombstones
2425            .lock()
2426            .unwrap_or_else(|poisoned| poisoned.into_inner())
2427            .get(module_id)
2428            .copied()
2429            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2430    }
2431
2432    /// Retire a reserved-id gate only after its module has left supervision.
2433    ///
2434    /// A retained gate has no live nonce (`None`), so releasing any other entry
2435    /// would weaken a currently configured or otherwise active reservation.
2436    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2437        if self.get(module_id).is_some() {
2438            return false;
2439        }
2440        let mut reserved_nonces = self
2441            .reserved_nonces
2442            .lock()
2443            .unwrap_or_else(|poisoned| poisoned.into_inner());
2444        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2445            return false;
2446        }
2447        reserved_nonces.remove(module_id);
2448        true
2449    }
2450
2451    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2452        Arc::clone(&self.operation_lock)
2453    }
2454}
2455
2456/// Process supervisor for subc-owned singleton modules.
2457#[derive(Debug, Clone)]
2458pub struct Supervisor {
2459    registry: Arc<Registry>,
2460    restart_policy: RestartPolicy,
2461    drain_timeout: Duration,
2462    connection_file_path: Option<PathBuf>,
2463    capture_logs_dir: Option<PathBuf>,
2464    forwarding: Option<Arc<ForwardingTable>>,
2465    process_liveness: Arc<SupervisorProcessLiveness>,
2466    supervisor_handle: Option<SupervisorHandle>,
2467    health: HealthConfig,
2468    daemon_start_clock: crate::clock::StartClock,
2469    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2470    spawn_events: SpawnEventFeed,
2471    provenance_probe: ExecutableIdentityProbe,
2472    /// Every process spawned through this supervisor (and its clones) and not
2473    /// yet reaped, so daemon shutdown can end them.
2474    child_roster: ChildRoster,
2475    #[cfg(target_os = "linux")]
2476    cgroup_placement: Option<subc_cgroup::Placement>,
2477    #[cfg(test)]
2478    test_after_first_spawn: AfterFirstSpawnHook,
2479}
2480
2481/// Test-only hook run on the path that takes on a new module, right after its
2482/// first `spawn_child` returns (the process exists and could already be
2483/// registering) and before that process is handed to the module's supervise
2484/// loop and put on the roster. Lets a test observe what a fast child would see
2485/// in that window without racing a real one.
2486#[cfg(test)]
2487#[derive(Clone, Default)]
2488struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2489
2490#[cfg(test)]
2491type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2492
2493#[cfg(test)]
2494impl fmt::Debug for AfterFirstSpawnHook {
2495    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2496        f.write_str("AfterFirstSpawnHook")
2497    }
2498}
2499
2500#[cfg(test)]
2501impl AfterFirstSpawnHook {
2502    fn run(&self, module_id: &str) {
2503        if let Some(hook) = &self.0 {
2504            hook(module_id);
2505        }
2506    }
2507}
2508
2509impl Supervisor {
2510    #[cfg(test)]
2511    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2512        let supervisor = Self::new(registry, policy);
2513        #[cfg(target_os = "macos")]
2514        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2515        supervisor
2516    }
2517    /// Verify the trampoline once when configured. Missing private OS support
2518    /// refuses every macOS launch by name but does not stop the daemon's control
2519    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2520    /// entry point; the library must not exec an arbitrary hosting program.
2521    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2522        let path = path.into();
2523        #[cfg(target_os = "macos")]
2524        {
2525            let result = probe_privacy_trampoline(&path).map(|()| path);
2526            if let Err(cause) = &result {
2527                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2528            }
2529            self.child_roster.set_privacy_trampoline(result);
2530        }
2531        #[cfg(not(target_os = "macos"))]
2532        let _ = path;
2533        self
2534    }
2535    /// The first step of an announced daemon shutdown, before the notice and
2536    /// before any connection is closed.
2537    ///
2538    /// Sets the daemon-shutdown flag first: from here on no module is
2539    /// respawned (crash restart, operator restart, or swap), and every child
2540    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2541    /// the module exits on the EOF this shutdown gives it or is signalled by a
2542    /// service manager that kills the whole cgroup. Then writes the journal's
2543    /// shutdown marker, which records the instant and closes this daemon
2544    /// incarnation's stretch of the journal.
2545    #[cfg(unix)]
2546    pub(crate) fn begin_daemon_shutdown(&self) {
2547        self.child_roster.close();
2548        if let Some(journal) = &self.terminal_journal {
2549            journal.stamp_shutdown();
2550        }
2551    }
2552
2553    /// Announce a cut while established connections can still carry replies.
2554    /// These budgets promise notice and a bounded wait, not child completion;
2555    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2556    #[cfg(unix)]
2557    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2558        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2559        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2560        let Some(forwarding) = &self.forwarding else {
2561            return Ok(());
2562        };
2563        let module_ids = forwarding
2564            .begin_daemon_drain()
2565            .map_err(SuperviseError::Forwarding)?;
2566        let deadline_ms =
2567            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2568        let mut notices = tokio::task::JoinSet::new();
2569        let mut drains = Vec::new();
2570        for module_id in module_ids {
2571            let Some(target) = forwarding
2572                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2573                .map_err(SuperviseError::Forwarding)?
2574            else {
2575                continue;
2576            };
2577            let routes = forwarding
2578                .endpoint_routes(target.endpoint)
2579                .map_err(SuperviseError::Forwarding)?;
2580            // Restart allows deployed consumers to reopen after the new daemon
2581            // appears. The wire reason stays `restart`; what tells a daemon cut
2582            // apart from a module restart afterwards is the terminal record
2583            // itself, whose disposition is `daemon_shutdown` for every exit
2584            // observed once `begin_daemon_shutdown` has run.
2585            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2586                reason: RouteCloseReason::Restart,
2587                deadline_ms,
2588            })
2589            .expect("module draining serializes");
2590            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2591            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2592            for route in routes {
2593                let client = route.goodbye_target;
2594                if let Some((_, channels)) = clients
2595                    .iter_mut()
2596                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2597                {
2598                    channels.push(client.channel);
2599                } else {
2600                    let channel = client.channel;
2601                    clients.push((client, vec![channel]));
2602                }
2603            }
2604            for (client, mut channels) in clients {
2605                channels.sort_unstable();
2606                channels.dedup();
2607                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2608                    module_id: module_id.clone(),
2609                    channels,
2610                    reason: RouteCloseReason::Restart,
2611                })
2612                .expect("route closing serializes");
2613                recipients.push((client.sink, client.negotiated_ver, closing));
2614            }
2615            for (sink, version, body) in recipients {
2616                notices.spawn(async move {
2617                    let frame = Frame::build_with_version(
2618                        version,
2619                        FrameType::Push,
2620                        control_flags(),
2621                        0,
2622                        0,
2623                        0,
2624                        body,
2625                    )
2626                    .expect("bounded lifecycle notice frame builds");
2627                    sink.send_flushed(frame).await
2628                });
2629            }
2630            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2631            drains.push((module_id, target.endpoint, gauges));
2632        }
2633        // A quiet forwarding table is not proof that queued notices reached the
2634        // socket. Wait for writer flush acknowledgements before testing quiescence.
2635        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2636        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2637            if !matches!(result, Ok(Ok(()))) {
2638                warn!(?result, "daemon shutdown notice delivery failed");
2639            }
2640        }
2641        notices.abort_all();
2642        let deadline = Instant::now() + DRAIN_BUDGET;
2643        let mut waits = tokio::task::JoinSet::new();
2644        for (module_id, endpoint, gauges) in drains {
2645            let forwarding = Arc::clone(forwarding);
2646            let mut runtime = self.runtime_config();
2647            runtime.health.cadence = Duration::from_millis(100);
2648            waits.spawn(async move {
2649                wait_for_forwarding_quiescence(
2650                    &forwarding,
2651                    &module_id,
2652                    &runtime,
2653                    endpoint,
2654                    deadline,
2655                    &gauges,
2656                    DrainScope::Active,
2657                )
2658                .await
2659            });
2660        }
2661        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2662            if !matches!(result, Ok(Ok(true))) {
2663                warn!(?result, "daemon shutdown drain did not reach quiescence");
2664            }
2665        }
2666        Ok(())
2667    }
2668
2669    /// The last step of an announced daemon shutdown, after the notice and the
2670    /// drain: send every registered module a module GOODBYE, the same planned
2671    /// stop signal `ck module stop` gives, then close every connection so each
2672    /// subc module sees EOF and starts its own teardown, then end every
2673    /// supervised child that has not exited
2674    /// by its own deadline (its drain budget, capped). Modules lead their own
2675    /// process groups, so a
2676    /// service manager's group kill no longer reaches them; without this a
2677    /// child that does not stop on EOF (every `protocol: "none"` child, which
2678    /// has no connection) would outlive the daemon. Every wait is bounded (see
2679    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2680    #[cfg(unix)]
2681    pub(crate) async fn end_children_for_daemon_shutdown(
2682        &self,
2683        already_escalated: bool,
2684        escalate: impl std::future::Future<Output = ()>,
2685    ) {
2686        tokio::pin!(escalate);
2687        let mut escalated = already_escalated;
2688        if let Some(forwarding) = &self.forwarding {
2689            let reason = CloseReason::new(
2690                "daemon_shutdown",
2691                "the daemon is exiting after its shutdown notice and drain",
2692            );
2693            if escalated {
2694                // The operator asked to stop waiting: queue the GOODBYEs but
2695                // do not wait for them to be written.
2696                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2697            } else {
2698                tokio::select! {
2699                    biased;
2700                    _ = escalate.as_mut() => {
2701                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2702                        escalated = true;
2703                    }
2704                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2705                }
2706            }
2707            let closed = forwarding.close_all_connections(&reason);
2708            debug!(closed, "closed established connections for daemon shutdown");
2709        }
2710        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2711        // already completed and must not be polled again; the child shutdown
2712        // wait is told it is escalated and gets a future that never fires.
2713        let escalated_here = escalated && !already_escalated;
2714        let remaining_escalate = async move {
2715            if escalated_here {
2716                std::future::pending::<()>().await;
2717            } else {
2718                escalate.await;
2719            }
2720        };
2721        crate::child_roster::end_children_for_daemon_shutdown(
2722            &self.child_roster,
2723            escalated,
2724            remaining_escalate,
2725        )
2726        .await;
2727    }
2728
2729    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2730        Self {
2731            registry,
2732            restart_policy,
2733            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2734            connection_file_path: None,
2735            capture_logs_dir: None,
2736            forwarding: None,
2737            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2738            supervisor_handle: None,
2739            health: HealthConfig::default(),
2740            daemon_start_clock: crate::clock::StartClock::capture(),
2741            terminal_journal: None,
2742            spawn_events: SpawnEventFeed::default(),
2743            provenance_probe: ExecutableIdentityProbe::default(),
2744            child_roster: ChildRoster::default(),
2745            #[cfg(target_os = "linux")]
2746            cgroup_placement: None,
2747            #[cfg(test)]
2748            test_after_first_spawn: AfterFirstSpawnHook::default(),
2749        }
2750    }
2751
2752    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2753        self.drain_timeout = drain_timeout;
2754        self
2755    }
2756
2757    pub fn with_process_liveness(
2758        mut self,
2759        process_liveness: Arc<SupervisorProcessLiveness>,
2760    ) -> Self {
2761        self.process_liveness = process_liveness;
2762        self
2763    }
2764
2765    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2766        self.connection_file_path = Some(connection_file_path.into());
2767        self
2768    }
2769
2770    /// Enables daemon-owned capture files for supervised stdout and stderr.
2771    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2772        self.capture_logs_dir = Some(logs_dir.into());
2773        self
2774    }
2775
2776    /// Names this daemon lifetime in spawn events, independently of whether a
2777    /// terminal journal is configured.
2778    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2779        // A millisecond start stamp can repeat after clock rollback or a rapid
2780        // restart. Use the connection file's random daemon_id instead: it already
2781        // identifies this daemon lifetime independently of the wall clock.
2782        self.spawn_events.configure_incarnation(daemon_incarnation);
2783        self
2784    }
2785
2786    /// Enables best-effort history shared by every supervised module. Without
2787    /// it, terminal history is kept only in each module's in-memory ring.
2788    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2789        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2790        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2791            path,
2792            daemon_incarnation,
2793        )));
2794        this
2795    }
2796
2797    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2798        self.forwarding = Some(forwarding);
2799        self
2800    }
2801
2802    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2803        self.spawn_events = supervisor_handle.spawn_events.clone();
2804        self.supervisor_handle = Some(supervisor_handle);
2805        self
2806    }
2807
2808    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2809        self.health = health;
2810        self
2811    }
2812
2813    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2814    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2815    /// record is kept.
2816    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2817        self.child_roster.record_to(path.into());
2818        self
2819    }
2820
2821    #[cfg(target_os = "linux")]
2822    pub fn with_cgroup_placement(
2823        mut self,
2824        cgroup_placement: Option<subc_cgroup::Placement>,
2825    ) -> Self {
2826        self.cgroup_placement = cgroup_placement;
2827        self
2828    }
2829
2830    /// Spawn `spec.program` and start monitoring it.
2831    ///
2832    /// The child is expected to parse `--subc <connection-file-path>`, read the
2833    /// TCP+key connection file, authenticate to the already-running listener, and
2834    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2835    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2836        validate_spec(&spec)?;
2837        self.establish_identity(&spec);
2838
2839        let runtime = self.runtime_config();
2840        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2841        let spawned = spawn_child(
2842            &spec,
2843            runtime.connection_file_path.as_deref(),
2844            self.supervisor_handle.as_ref(),
2845            &runtime.stderr_ring,
2846            runtime.capture_logs_dir.as_deref(),
2847            &runtime.child_roster,
2848            #[cfg(target_os = "linux")]
2849            runtime.cgroup_placement.as_ref(),
2850        );
2851        #[cfg(test)]
2852        self.test_after_first_spawn.run(&spec.module_id);
2853        let child = match spawned {
2854            Ok(child) => child,
2855            Err(err) => {
2856                // Unlike the configured paths, a failed `spawn` leaves nothing
2857                // on the roster, so the module must not stay marked configured.
2858                self.abandon_unrostered(&spec.module_id);
2859                return Err(err);
2860            }
2861        };
2862        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2863
2864        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2865    }
2866
2867    /// Make `spec`'s module count as configured, with its identity gates
2868    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2869    /// exists.
2870    ///
2871    /// Every path that takes on a new module calls this before `spawn_child`.
2872    /// The order is the point: the child can connect, register, sync its
2873    /// scopes and ask about them as soon as it is spawned, and the module is
2874    /// only put on the roster after `spawn_child` returns. Were the mark set
2875    /// with the roster entry, a fast child would see its own owner reported
2876    /// as not configured, and a scoped `route.open` in that window would be
2877    /// refused as terminal `scope_not_live` ("will never sync") instead of
2878    /// retryable `scope_not_synced`.
2879    fn establish_identity(&self, spec: &ModuleSpec) {
2880        if let Some(supervisor_handle) = &self.supervisor_handle {
2881            supervisor_handle.apply_identity_configuration(spec);
2882            supervisor_handle.mark_configured(&spec.module_id);
2883        }
2884    }
2885
2886    /// Take back [`Self::establish_identity`]'s configured mark when the
2887    /// module will not be put on the roster after all.
2888    fn abandon_unrostered(&self, module_id: &str) {
2889        if let Some(supervisor_handle) = &self.supervisor_handle {
2890            supervisor_handle.unmark_configured_unless_rostered(module_id);
2891        }
2892    }
2893
2894    /// Record a freshly spawned first process as running. On failure the
2895    /// module never reaches the roster, so its configured mark is taken back.
2896    fn mark_first_process_running(
2897        &self,
2898        spec: &ModuleSpec,
2899        runtime: &SupervisorRuntimeConfig,
2900        snapshot: &SharedSnapshot,
2901        child: &SupervisedChild,
2902    ) -> Result<(), SuperviseError> {
2903        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2904            self.abandon_unrostered(&spec.module_id);
2905            return Err(err);
2906        }
2907        self.process_liveness
2908            .track(spec.module_id.clone(), Arc::clone(snapshot));
2909        Ok(())
2910    }
2911
2912    /// Start supervising a module declared in daemon configuration.
2913    ///
2914    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2915    /// failures in the supervisor handle so operator-facing `supervisor.list`
2916    /// reflects every configured module while daemon startup continues.
2917    pub fn supervise_configured(
2918        &self,
2919        spec: ModuleSpec,
2920        enabled: bool,
2921    ) -> Result<SupervisedModule, SuperviseError> {
2922        validate_spec(&spec)?;
2923        self.establish_identity(&spec);
2924
2925        let runtime = self.runtime_config();
2926        if !enabled {
2927            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2928            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2929        }
2930
2931        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2932        let spawned = spawn_child(
2933            &spec,
2934            runtime.connection_file_path.as_deref(),
2935            self.supervisor_handle.as_ref(),
2936            &runtime.stderr_ring,
2937            runtime.capture_logs_dir.as_deref(),
2938            &runtime.child_roster,
2939            #[cfg(target_os = "linux")]
2940            runtime.cgroup_placement.as_ref(),
2941        );
2942        #[cfg(test)]
2943        self.test_after_first_spawn.run(&spec.module_id);
2944        match spawned {
2945            Ok(child) => {
2946                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2947                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2948            }
2949            Err(err) => {
2950                error!(
2951                    module_id = %spec.module_id,
2952                    program = %spec.program.display(),
2953                    error = %err,
2954                    "configured module failed to spawn; marking failed and continuing"
2955                );
2956                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2957                Ok(self.supervised_module(spec, runtime, snapshot, None))
2958            }
2959        }
2960    }
2961
2962    /// Supervise a configured module with its own health, drain, and crash
2963    /// budget. The restart policy is per-module because the config file is:
2964    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2965    /// module that is expensive to restart should not be forced onto the same
2966    /// budget as one that is cheap.
2967    pub fn supervise_configured_with_health(
2968        &self,
2969        spec: ModuleSpec,
2970        enabled: bool,
2971        health: HealthConfig,
2972        drain_timeout_ms: Option<u64>,
2973        restart_policy: RestartPolicy,
2974    ) -> Result<SupervisedModule, SuperviseError> {
2975        validate_spec(&spec)?;
2976        self.establish_identity(&spec);
2977
2978        let mut runtime = self.runtime_config();
2979        runtime.health = health.clone();
2980        runtime.restart_policy = restart_policy;
2981        if let Some(ms) = drain_timeout_ms {
2982            runtime.drain_timeout = Duration::from_millis(ms);
2983            *runtime
2984                .effective_drain_timeout
2985                .lock()
2986                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2987        }
2988        if !enabled {
2989            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2990            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2991        }
2992
2993        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2994        let spawned = spawn_child(
2995            &spec,
2996            runtime.connection_file_path.as_deref(),
2997            self.supervisor_handle.as_ref(),
2998            &runtime.stderr_ring,
2999            runtime.capture_logs_dir.as_deref(),
3000            &runtime.child_roster,
3001            #[cfg(target_os = "linux")]
3002            runtime.cgroup_placement.as_ref(),
3003        );
3004        #[cfg(test)]
3005        self.test_after_first_spawn.run(&spec.module_id);
3006        match spawned {
3007            Ok(child) => {
3008                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3009                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3010            }
3011            Err(err) => {
3012                if health.critical {
3013                    error!(
3014                        module_id = %spec.module_id,
3015                        program = %spec.program.display(),
3016                        error = %err,
3017                        "critical configured module failed to spawn; marking failed and alerting"
3018                    );
3019                } else {
3020                    error!(
3021                        module_id = %spec.module_id,
3022                        program = %spec.program.display(),
3023                        error = %err,
3024                        "configured module failed to spawn; marking failed and continuing"
3025                    );
3026                }
3027                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3028                Ok(self.supervised_module(spec, runtime, snapshot, None))
3029            }
3030        }
3031    }
3032
3033    fn runtime_config(&self) -> SupervisorRuntimeConfig {
3034        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3035        SupervisorRuntimeConfig {
3036            scheduled_respawn: Arc::default(),
3037            deferred_reload_reply: Arc::default(),
3038            restart_policy: self.restart_policy,
3039            drain_timeout: self.drain_timeout,
3040            // Shared with this module's roster copy: daemon shutdown waits on
3041            // each child for the module's own drain budget, as resolved now.
3042            child_roster: self
3043                .child_roster
3044                .for_module(Arc::clone(&effective_drain_timeout)),
3045            effective_drain_timeout,
3046            default_drain_timeout: self.drain_timeout,
3047            health: self.health.clone(),
3048            connection_file_path: self.connection_file_path.clone(),
3049            capture_logs_dir: self.capture_logs_dir.clone(),
3050            forwarding: self.forwarding.clone(),
3051            supervisor_handle: self.supervisor_handle.clone(),
3052            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3053            terminal_ring: Arc::new(Mutex::new(
3054                TerminalRing::new(
3055                    TerminalRingConfig::default(),
3056                    self.daemon_start_clock.started_at_ms(),
3057                )
3058                .with_start_clock(self.daemon_start_clock)
3059                .with_journal(self.terminal_journal.clone())
3060                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3061            )),
3062            spawn_events: self.spawn_events.clone(),
3063            #[cfg(target_os = "linux")]
3064            cgroup_placement: self.cgroup_placement.clone(),
3065            #[cfg(test)]
3066            test_seed_stale_facts_before_enable_spawn: false,
3067            #[cfg(test)]
3068            test_reload_exit_record_gate: None,
3069        }
3070    }
3071
3072    fn supervised_module(
3073        &self,
3074        spec: ModuleSpec,
3075        runtime: SupervisorRuntimeConfig,
3076        snapshot: SharedSnapshot,
3077        child: Option<SupervisedChild>,
3078    ) -> SupervisedModule {
3079        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3080            spec: spec.clone(),
3081            health: runtime.health.clone(),
3082        }));
3083        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3084        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3085        // The module's OWN policy, which may be its per-module config rather than
3086        // the supervisor-wide one; status must report the budget the supervise
3087        // loop actually enforces.
3088        let restart_policy = runtime.restart_policy;
3089        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3090        let (tx, rx) = mpsc::channel(4);
3091        let monitor = tokio::spawn(supervise_loop(
3092            spec.clone(),
3093            runtime,
3094            Arc::clone(&self.registry),
3095            Arc::clone(&self.process_liveness),
3096            Arc::clone(&snapshot),
3097            child,
3098            rx,
3099        ));
3100
3101        let module_id = spec.module_id.clone();
3102        let module = SupervisedModule {
3103            inner: Arc::new(SupervisedModuleInner {
3104                module_id: module_id.clone(),
3105                registry: Arc::clone(&self.registry),
3106                snapshot,
3107                configuration,
3108                stderr_ring,
3109                terminal_ring,
3110                commands: tx,
3111                monitor: Mutex::new(Some(monitor)),
3112                restart_policy,
3113                effective_drain_timeout,
3114                provenance_probe: self.provenance_probe.clone(),
3115            }),
3116        };
3117        // The identity gates and the configured mark were set by
3118        // `establish_identity` before any process was spawned; only the roster
3119        // entry waits for the module handle, which needs the spawned child.
3120        if let Some(supervisor_handle) = &self.supervisor_handle {
3121            supervisor_handle.insert(module.clone());
3122        }
3123        module
3124    }
3125}
3126
3127impl Default for Supervisor {
3128    fn default() -> Self {
3129        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3130    }
3131}
3132
3133/// Handle to one supervised child process.
3134#[derive(Clone)]
3135pub struct SupervisedModule {
3136    inner: Arc<SupervisedModuleInner>,
3137}
3138
3139struct SupervisedModuleInner {
3140    module_id: String,
3141    registry: Arc<Registry>,
3142    snapshot: SharedSnapshot,
3143    configuration: Arc<Mutex<SupervisedConfiguration>>,
3144    stderr_ring: Arc<Mutex<StderrRing>>,
3145    terminal_ring: Arc<Mutex<TerminalRing>>,
3146    commands: mpsc::Sender<SupervisorCommand>,
3147    monitor: Mutex<Option<JoinHandle<()>>>,
3148    /// Copied from the supervisor's runtime config at spawn so `status()` can
3149    /// report the restart budget without reaching back into the supervisor. The
3150    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3151    restart_policy: RestartPolicy,
3152    effective_drain_timeout: Arc<Mutex<Duration>>,
3153    provenance_probe: ExecutableIdentityProbe,
3154}
3155
3156impl fmt::Debug for SupervisedModule {
3157    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3158        f.debug_struct("SupervisedModule")
3159            .field("module_id", &self.inner.module_id)
3160            .field("status", &self.status())
3161            .finish_non_exhaustive()
3162    }
3163}
3164
3165impl SupervisedModule {
3166    pub fn module_id(&self) -> &str {
3167        &self.inner.module_id
3168    }
3169
3170    /// Test-only: put one probe miss on the streak, the way
3171    /// `handle_health_probe_failure` does, so tests can assert what a later
3172    /// event does to the streak without driving the whole probe loop.
3173    #[cfg(test)]
3174    pub(crate) fn record_health_probe_failure_for_test(
3175        &self,
3176        detail: &str,
3177    ) -> Result<(), SuperviseError> {
3178        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3179            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3180            state.health.detail = Some(detail.to_string());
3181        })
3182    }
3183
3184    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3185        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3186    }
3187
3188    /// The module's retained stderr, newest lines last.
3189    ///
3190    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3191    /// module, `supervisor.list` renders every module, and putting it in the
3192    /// shared snapshot would make each status read carry a payload almost nobody
3193    /// asked for. Callers that want the text ask for it.
3194    pub fn stderr_tail(
3195        &self,
3196        max_lines: Option<usize>,
3197        max_bytes: Option<usize>,
3198    ) -> StderrTailSnapshot {
3199        self.inner
3200            .stderr_ring
3201            .lock()
3202            .unwrap_or_else(|poisoned| poisoned.into_inner())
3203            .snapshot(max_lines, max_bytes)
3204    }
3205
3206    /// The module's bounded terminal history, oldest retained exit first.
3207    ///
3208    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3209    /// daemon whose in-memory history was necessarily reset.
3210    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3211        self.inner
3212            .terminal_ring
3213            .lock()
3214            .unwrap_or_else(|poisoned| poisoned.into_inner())
3215            .snapshot()
3216    }
3217
3218    /// Retained observations from the current ring and all journal generations.
3219    ///
3220    /// Blocking: this reads the journal files. Async callers use
3221    /// [`Self::read_durable_terminal_history`].
3222    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3223        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3224    }
3225
3226    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3227    /// read (up to every retained generation) never occupies a runtime worker.
3228    /// Fails only if the blocking task could not finish (runtime shutdown or a
3229    /// panic in the read).
3230    pub(crate) async fn read_durable_terminal_history(
3231        &self,
3232    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3233        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3234        let module_id = self.inner.module_id.clone();
3235        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3236            .await
3237    }
3238
3239    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3240        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3241            .map(|(status, _)| status)
3242    }
3243
3244    pub(crate) fn record_deliberate_severance(
3245        &self,
3246        identity: ProcessIdentity,
3247    ) -> Result<bool, SuperviseError> {
3248        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3249        if snapshot.pid != Some(identity.pid)
3250            || snapshot.process_start_time != Some(identity.start_time)
3251        {
3252            return Ok(false);
3253        }
3254        snapshot.deliberate_severance = Some(identity);
3255        Ok(true)
3256    }
3257
3258    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3259    ///
3260    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3261    /// does not produce reader-observability logs.
3262    pub(crate) fn status_for_control(
3263        &self,
3264        caller: &'static str,
3265    ) -> Result<ModuleStatus, SuperviseError> {
3266        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3267            .map(|(status, _)| status)
3268    }
3269
3270    fn status_with_snapshot_lock(
3271        &self,
3272        snapshot: &SharedSnapshot,
3273        caller: Option<&'static str>,
3274    ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3275        let mut guard = match caller {
3276            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3277            None => lock_snapshot(snapshot)?,
3278        };
3279        // Read the budget through the pruning path so a reader sees the same
3280        // in-window count the restart decision would use, not a stale total.
3281        let restart_count =
3282            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3283        let snapshot = guard.clone();
3284        drop(guard);
3285        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3286            SuperviseError::StatePoisoned {
3287                module_id: Some(self.inner.module_id.clone()),
3288            }
3289        })?;
3290        let registration_active = self
3291            .inner
3292            .registry
3293            .get_module(&self.inner.module_id)
3294            .map_err(SuperviseError::Registry)?
3295            .is_some();
3296        let protocol = snapshot
3297            .spawned_protocol
3298            .unwrap_or(self.declared_protocol()?);
3299        let running_process =
3300            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3301        // Registration is the difference between the two protocols and the only
3302        // one: a subc module that has not registered cannot serve a request even
3303        // though its process is up, and a `none` module never registers at all,
3304        // so requiring it there would pin `live` to false for the whole life of
3305        // a perfectly healthy process.
3306        let live = match protocol {
3307            ModuleProtocol::Subc => running_process && registration_active,
3308            ModuleProtocol::None => running_process,
3309        };
3310
3311        Ok((
3312            ModuleStatus {
3313                module_id: self.inner.module_id.clone(),
3314                state: snapshot.state,
3315                enabled: snapshot.enabled,
3316                process_alive: snapshot.process_alive,
3317                registration_active,
3318                protocol,
3319                live,
3320                restart_count,
3321                lifetime_restarts: snapshot.lifetime_restarts,
3322                spawn_generation: snapshot.spawn_generation,
3323                max_restarts: self.inner.restart_policy.max_restarts,
3324                restart_window: self.inner.restart_policy.window,
3325                drain_timeout,
3326                restart_backoff: self.inner.restart_policy.backoff,
3327                restart_max_backoff: self.inner.restart_policy.max_backoff,
3328                pid: snapshot.reported_pid(),
3329                spawned_at_ms: snapshot.spawned_at_ms,
3330                spawned_from: snapshot.spawned_from,
3331                process_start_time: snapshot.process_start_time,
3332                last_exit: snapshot.last_exit,
3333                health: snapshot.health,
3334            },
3335            snapshot.spawned_file_identity,
3336        ))
3337    }
3338
3339    #[cfg(test)]
3340    pub(crate) fn hold_snapshot_for_test(
3341        &self,
3342        acquired: std::sync::mpsc::Sender<()>,
3343        hold: Duration,
3344    ) -> std::thread::JoinHandle<()> {
3345        let snapshot = Arc::clone(&self.inner.snapshot);
3346        std::thread::spawn(move || {
3347            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3348            acquired
3349                .send(())
3350                .expect("test receiver waits for snapshot lock");
3351            std::thread::sleep(hold);
3352        })
3353    }
3354
3355    /// The status and the running-image check for `supervisor.provenance`,
3356    /// taken from one status read. The exec acknowledgement can land between
3357    /// two separate reads, and the reply would then pair "no pid yet" with an
3358    /// image observed after the module started, which describes no single
3359    /// moment.
3360    pub(crate) async fn status_and_running_image_agreement(
3361        &self,
3362    ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3363        let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3364        let image = self
3365            .inner
3366            .provenance_probe
3367            .observe(
3368                status.pid,
3369                status.spawned_from.as_deref(),
3370                identity,
3371                status.process_start_time,
3372            )
3373            .await;
3374        Ok((status, image))
3375    }
3376
3377    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3378        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3379            Ok(snapshot) => snapshot.clone(),
3380            Err(_) => {
3381                return subc_control::RunningImageAgreement::Unavailable {
3382                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3383                };
3384            }
3385        };
3386        self.inner
3387            .provenance_probe
3388            .observe(
3389                snapshot.reported_pid(),
3390                snapshot.spawned_from.as_deref(),
3391                snapshot.spawned_file_identity,
3392                snapshot.process_start_time,
3393            )
3394            .await
3395    }
3396
3397    /// Memory and CPU time of the module's current process, read now. Only the
3398    /// process the supervisor spawned is read, not processes it has started.
3399    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3400        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3401            Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3402            Err(_) => {
3403                return subc_control::ChildResourceUsage::Unavailable {
3404                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3405                }
3406            }
3407        };
3408        crate::child_resources::read(pid, start_time)
3409    }
3410
3411    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3412        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3413        Ok(match snapshot.state {
3414            ModuleState::Restarting => true,
3415            ModuleState::Failed | ModuleState::Disabled => false,
3416            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3417        })
3418    }
3419
3420    #[cfg(test)]
3421    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3422        self.is_warming_with_snapshot_lock(None)
3423    }
3424
3425    pub(crate) fn is_warming_for_control(
3426        &self,
3427        caller: &'static str,
3428    ) -> Result<bool, SuperviseError> {
3429        self.is_warming_with_snapshot_lock(Some(caller))
3430    }
3431
3432    fn is_warming_with_snapshot_lock(
3433        &self,
3434        caller: Option<&'static str>,
3435    ) -> Result<bool, SuperviseError> {
3436        let snapshot = match caller {
3437            Some(caller) => {
3438                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3439            }
3440            None => lock_snapshot(&self.inner.snapshot)?,
3441        }
3442        .clone();
3443        Ok(matches!(
3444            snapshot.state,
3445            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3446        ))
3447    }
3448
3449    /// Drain the module and stop monitoring it.
3450    pub async fn drain(&self) -> Result<(), SuperviseError> {
3451        self.stop().await
3452    }
3453
3454    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3455        match self.state()? {
3456            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3457            ModuleState::Starting
3458            | ModuleState::Running
3459            | ModuleState::Unresponsive
3460            | ModuleState::Restarting
3461            | ModuleState::Draining
3462            | ModuleState::Disabled => {}
3463        }
3464
3465        let (reply_tx, reply_rx) = oneshot::channel();
3466        self.inner
3467            .commands
3468            .send(SupervisorCommand::Retire { reply: reply_tx })
3469            .await
3470            .map_err(|_| SuperviseError::CommandClosed {
3471                module_id: self.inner.module_id.clone(),
3472            })?;
3473        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3474            module_id: self.inner.module_id.clone(),
3475        })?
3476    }
3477
3478    pub async fn stop(&self) -> Result<(), SuperviseError> {
3479        match self.state()? {
3480            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3481            ModuleState::Starting
3482            | ModuleState::Running
3483            | ModuleState::Unresponsive
3484            | ModuleState::Restarting
3485            | ModuleState::Draining
3486            | ModuleState::Disabled => {}
3487        }
3488
3489        let (reply_tx, reply_rx) = oneshot::channel();
3490        self.inner
3491            .commands
3492            .send(SupervisorCommand::Drain { reply: reply_tx })
3493            .await
3494            .map_err(|_| SuperviseError::CommandClosed {
3495                module_id: self.inner.module_id.clone(),
3496            })?;
3497        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3498            module_id: self.inner.module_id.clone(),
3499        })?
3500    }
3501
3502    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3503        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3504        let (reply_tx, reply_rx) = oneshot::channel();
3505        self.inner
3506            .commands
3507            .send(SupervisorCommand::Restart {
3508                drain_timeout_ms,
3509                received_at_generation,
3510                queued_at: Instant::now(),
3511                reply: reply_tx,
3512            })
3513            .await
3514            .map_err(|_| SuperviseError::CommandClosed {
3515                module_id: self.inner.module_id.clone(),
3516            })?;
3517        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3518            module_id: self.inner.module_id.clone(),
3519        })?
3520    }
3521
3522    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3523    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3524    /// process then drains in the background of the supervise loop) or has
3525    /// failed, leaving the old process serving.
3526    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3527        let (reply_tx, reply_rx) = oneshot::channel();
3528        self.inner
3529            .commands
3530            .send(SupervisorCommand::Swap {
3531                ready_timeout,
3532                reply: reply_tx,
3533            })
3534            .await
3535            .map_err(|_| SuperviseError::CommandClosed {
3536                module_id: self.inner.module_id.clone(),
3537            })?;
3538        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3539            module_id: self.inner.module_id.clone(),
3540        })?
3541    }
3542
3543    pub async fn reload(&self) -> Result<(), SuperviseError> {
3544        let (reply_tx, reply_rx) = oneshot::channel();
3545        self.inner
3546            .commands
3547            .send(SupervisorCommand::Reload { reply: reply_tx })
3548            .await
3549            .map_err(|_| SuperviseError::CommandClosed {
3550                module_id: self.inner.module_id.clone(),
3551            })?;
3552        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3553            module_id: self.inner.module_id.clone(),
3554        })?
3555    }
3556
3557    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3558        let (reply_tx, reply_rx) = oneshot::channel();
3559        self.inner
3560            .commands
3561            .send(SupervisorCommand::SetEnabled {
3562                enabled,
3563                reply: reply_tx,
3564            })
3565            .await
3566            .map_err(|_| SuperviseError::CommandClosed {
3567                module_id: self.inner.module_id.clone(),
3568            })?;
3569        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3570            module_id: self.inner.module_id.clone(),
3571        })?
3572    }
3573
3574    /// The current process's protocol, or the configured protocol when down.
3575    /// A rescan stores the next launch spec without changing how an existing
3576    /// process registers, serves routes, is probed, or exits.
3577    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3578        let configured = self
3579            .inner
3580            .configuration
3581            .lock()
3582            .map_err(|_| SuperviseError::StatePoisoned {
3583                module_id: Some(self.inner.module_id.clone()),
3584            })?
3585            .spec
3586            .protocol;
3587        let state = lock_snapshot(&self.inner.snapshot)?;
3588        Ok(state.spawned_protocol.unwrap_or(configured))
3589    }
3590
3591    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3592        let configuration =
3593            self.inner
3594                .configuration
3595                .lock()
3596                .map_err(|_| SuperviseError::StatePoisoned {
3597                    module_id: Some(self.inner.module_id.clone()),
3598                })?;
3599        Ok((configuration.spec.clone(), configuration.health.clone()))
3600    }
3601
3602    /// Replace this module's launch spec, keeping its health and drain policy,
3603    /// the way a rescan does for a changed config entry. The running process is
3604    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3605    #[cfg(any(test, feature = "test-support"))]
3606    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3607        let (_, health) = self.configuration()?;
3608        let drain_timeout_ms = u64::try_from(
3609            self.inner
3610                .effective_drain_timeout
3611                .lock()
3612                .unwrap_or_else(|poisoned| poisoned.into_inner())
3613                .as_millis(),
3614        )
3615        .ok();
3616        self.update_configuration(spec, health, drain_timeout_ms)
3617            .await
3618    }
3619
3620    pub(crate) async fn update_configuration(
3621        &self,
3622        spec: ModuleSpec,
3623        health: HealthConfig,
3624        drain_timeout_ms: Option<u64>,
3625    ) -> Result<(), SuperviseError> {
3626        if spec.module_id != self.inner.module_id {
3627            return Err(SuperviseError::InvalidSpec {
3628                reason: "a supervised module's module_id cannot be changed".to_string(),
3629            });
3630        }
3631        validate_spec(&spec)?;
3632        let (reply_tx, reply_rx) = oneshot::channel();
3633        self.inner
3634            .commands
3635            .send(SupervisorCommand::UpdateConfiguration {
3636                spec: spec.clone(),
3637                health: health.clone(),
3638                drain_timeout_ms,
3639                reply: reply_tx,
3640            })
3641            .await
3642            .map_err(|_| SuperviseError::CommandClosed {
3643                module_id: self.inner.module_id.clone(),
3644            })?;
3645        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3646            module_id: self.inner.module_id.clone(),
3647        })?;
3648        let mut configuration =
3649            self.inner
3650                .configuration
3651                .lock()
3652                .map_err(|_| SuperviseError::StatePoisoned {
3653                    module_id: Some(self.inner.module_id.clone()),
3654                })?;
3655        configuration.spec = spec;
3656        configuration.health = health;
3657        Ok(())
3658    }
3659}
3660
3661impl Drop for SupervisedModuleInner {
3662    fn drop(&mut self) {
3663        let Ok(mut monitor) = self.monitor.lock() else {
3664            return;
3665        };
3666        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3667            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3668                state.state = ModuleState::Stopped;
3669                clear_current_process_facts(state);
3670            });
3671            monitor.abort();
3672        }
3673        let _ = monitor.take();
3674    }
3675}
3676
3677#[derive(Debug)]
3678enum SupervisorCommand {
3679    Drain {
3680        reply: oneshot::Sender<Result<(), SuperviseError>>,
3681    },
3682    Retire {
3683        reply: oneshot::Sender<Result<(), SuperviseError>>,
3684    },
3685    Restart {
3686        /// Operator override for this one restart's drain budget, in ms. `None`
3687        /// uses the module's configured/default budget; `Some(0)` cuts
3688        /// immediately (wedge bounce: a stuck request never settles, so
3689        /// waiting only delays recovery).
3690        drain_timeout_ms: Option<u64>,
3691        /// The module's `spawn_generation` when the request was received, before
3692        /// it waited in the command queue. A queued restart whose module has
3693        /// since spawned a newer process is already satisfied (see the handler).
3694        received_at_generation: u64,
3695        /// When the request entered the command queue, so the handler can log
3696        /// how long it waited behind the loop's other work.
3697        queued_at: Instant,
3698        reply: oneshot::Sender<Result<(), SuperviseError>>,
3699    },
3700    Reload {
3701        reply: oneshot::Sender<Result<(), SuperviseError>>,
3702    },
3703    SetEnabled {
3704        enabled: bool,
3705        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3706    },
3707    UpdateConfiguration {
3708        spec: ModuleSpec,
3709        health: HealthConfig,
3710        /// Per-module drain override from the new config; `None` re-resolves to
3711        /// the supervisor-wide default.
3712        drain_timeout_ms: Option<u64>,
3713        reply: oneshot::Sender<()>,
3714    },
3715    Swap {
3716        /// How long the candidate may take to register and declare itself
3717        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3718        ready_timeout: Option<Duration>,
3719        /// Answered at cutover or failure; the incumbent's drain follows.
3720        reply: oneshot::Sender<Result<(), SuperviseError>>,
3721    },
3722}
3723
3724#[derive(Debug)]
3725pub enum SuperviseError {
3726    InvalidSpec {
3727        reason: String,
3728    },
3729    Spawn {
3730        program: PathBuf,
3731        source: io::Error,
3732        cgroup_path: Option<PathBuf>,
3733    },
3734    Cgroup {
3735        module_id: String,
3736        source: io::Error,
3737    },
3738    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3739    /// than spawn a reserved module without its identity binding.
3740    LaunchNonce {
3741        reason: String,
3742    },
3743    Wait {
3744        module_id: String,
3745        source: io::Error,
3746    },
3747    Kill {
3748        module_id: String,
3749        source: io::Error,
3750    },
3751    Forwarding(ForwardingError),
3752    Registry(RegistryError),
3753    ReloadUnavailable {
3754        module_id: String,
3755        reason: String,
3756    },
3757    /// An operator restart/reload was requested for a module that is currently
3758    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3759    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3760    /// by a restart, so these commands are rejected instead of re-enabling it.
3761    Disabled {
3762        module_id: String,
3763    },
3764    ReloadFailed {
3765        module_id: String,
3766        reason: String,
3767    },
3768    RegistrationStillActive {
3769        module_id: String,
3770        waited: Duration,
3771    },
3772    StatePoisoned {
3773        module_id: Option<String>,
3774    },
3775    CommandClosed {
3776        module_id: String,
3777    },
3778    /// A restart or reload arrived while a swap's candidate was warming. The
3779    /// swap owns the module until it cuts over or fails; a stop or disable
3780    /// would have aborted it instead.
3781    SwapInProgress {
3782        module_id: String,
3783    },
3784    /// A swap was refused before anything was spawned.
3785    SwapRefused {
3786        module_id: String,
3787        reason: SwapRefusal,
3788    },
3789    /// A swap spawned a candidate and gave up on it. The candidate has been
3790    /// killed and its slot freed; the incumbent was left serving and was never
3791    /// drained, except in the one `CutoverLost` case described on that arm.
3792    SwapFailed {
3793        module_id: String,
3794        arm: SwapFailureArm,
3795        detail: String,
3796        /// How the candidate exited, when it exited on its own before the
3797        /// supervisor gave up on it.
3798        candidate_exit: Option<ExitReport>,
3799    },
3800}
3801
3802/// Why a swap was refused before a candidate was spawned.
3803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3804pub enum SwapRefusal {
3805    /// The module's config does not declare `overlap: "safe"`.
3806    OverlapExclusive,
3807    /// The module is not registered, so there is no incumbent to keep serving
3808    /// and nothing a swap would improve on; a plain restart is the tool.
3809    NotRegistered,
3810    /// The module does not speak the subc wire, so a candidate could never
3811    /// register or declare itself ready.
3812    ProtocolNone,
3813    /// The supervisor lacks the forwarding table (to cut routes over) or the
3814    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3815    NotConfigured,
3816    /// A swap is already open for this module.
3817    AlreadySwapping,
3818}
3819
3820impl SwapRefusal {
3821    pub fn as_str(self) -> &'static str {
3822        match self {
3823            Self::OverlapExclusive => "overlap_exclusive",
3824            Self::NotRegistered => "not_registered",
3825            Self::ProtocolNone => "protocol_none",
3826            Self::NotConfigured => "not_configured",
3827            Self::AlreadySwapping => "already_swapping",
3828        }
3829    }
3830}
3831
3832/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3833/// serving and undrained; see `CutoverLost`.
3834#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3835pub enum SwapFailureArm {
3836    /// The candidate process could not be started.
3837    SpawnFailed,
3838    /// The candidate did not register within the readiness budget.
3839    NeverRegistered,
3840    /// The candidate registered but did not declare itself ready in time.
3841    NeverReady,
3842    /// The candidate exited before cutover.
3843    CandidateExited,
3844    /// The candidate declared itself ready but failed its health probe.
3845    CandidateUnhealthy,
3846    /// An operator stop, disable or retire arrived while the candidate warmed.
3847    /// The candidate was killed and the operator's command then carried out on
3848    /// the incumbent.
3849    Interrupted,
3850    /// The candidate's connection closed at the moment of cutover. If it
3851    /// closed before forwarding moved, the incumbent is untouched. If it closed
3852    /// between the forwarding and registry halves of cutover, forwarding can no
3853    /// longer route to the incumbent, so the module is restarted plainly.
3854    CutoverLost,
3855}
3856
3857impl SwapFailureArm {
3858    pub fn as_str(self) -> &'static str {
3859        match self {
3860            Self::SpawnFailed => "spawn_failed",
3861            Self::NeverRegistered => "never_registered",
3862            Self::NeverReady => "never_ready",
3863            Self::CandidateExited => "candidate_exited",
3864            Self::CandidateUnhealthy => "candidate_unhealthy",
3865            Self::Interrupted => "interrupted",
3866            Self::CutoverLost => "cutover_lost",
3867        }
3868    }
3869}
3870
3871impl fmt::Display for SuperviseError {
3872    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3873        match self {
3874            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3875            Self::Spawn {
3876                program,
3877                source,
3878                cgroup_path: Some(cgroup_path),
3879            } => write!(
3880                f,
3881                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3882                cgroup_path.display(),
3883                program.display()
3884            ),
3885            Self::Spawn {
3886                program,
3887                source,
3888                cgroup_path: None,
3889            } => write!(
3890                f,
3891                "failed to spawn module '{}': {source}",
3892                program.display()
3893            ),
3894            Self::Cgroup { module_id, source } => {
3895                write!(
3896                    f,
3897                    "failed to prepare cgroup for module '{module_id}': {source}"
3898                )
3899            }
3900            Self::LaunchNonce { reason } => {
3901                write!(
3902                    f,
3903                    "failed to generate reserved-module launch nonce: {reason}"
3904                )
3905            }
3906            Self::Wait { module_id, source } => {
3907                write!(f, "failed to wait for module '{module_id}': {source}")
3908            }
3909            Self::Kill { module_id, source } => {
3910                write!(f, "failed to kill module '{module_id}': {source}")
3911            }
3912            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3913            Self::Registry(err) => write!(f, "registry error: {err}"),
3914            Self::ReloadUnavailable { module_id, reason } => {
3915                write!(f, "reload unavailable for module '{module_id}': {reason}")
3916            }
3917            Self::Disabled { module_id } => {
3918                write!(
3919                    f,
3920                    "module '{module_id}' is disabled; enable it before restart or reload"
3921                )
3922            }
3923            Self::ReloadFailed { module_id, reason } => {
3924                write!(f, "reload failed for module '{module_id}': {reason}")
3925            }
3926            Self::RegistrationStillActive { module_id, waited } => write!(
3927                f,
3928                "module '{module_id}' registration remained active after waiting {waited:?}"
3929            ),
3930            Self::StatePoisoned { module_id } => match module_id {
3931                Some(module_id) => {
3932                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3933                }
3934                None => write!(f, "supervisor state was poisoned"),
3935            },
3936            Self::CommandClosed { module_id } => {
3937                write!(
3938                    f,
3939                    "supervisor command channel for module '{module_id}' is closed"
3940                )
3941            }
3942            Self::SwapInProgress { module_id } => write!(
3943                f,
3944                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3945            ),
3946            Self::SwapRefused { module_id, reason } => match reason {
3947                SwapRefusal::OverlapExclusive => write!(
3948                    f,
3949                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3950                ),
3951                SwapRefusal::NotRegistered => write!(
3952                    f,
3953                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3954                ),
3955                SwapRefusal::ProtocolNone => write!(
3956                    f,
3957                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3958                ),
3959                SwapRefusal::NotConfigured => write!(
3960                    f,
3961                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3962                ),
3963                SwapRefusal::AlreadySwapping => {
3964                    write!(f, "module '{module_id}' is already being swapped")
3965                }
3966            },
3967            Self::SwapFailed {
3968                module_id,
3969                arm,
3970                detail,
3971                ..
3972            } => write!(
3973                f,
3974                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3975                arm.as_str()
3976            ),
3977        }
3978    }
3979}
3980
3981impl Error for SuperviseError {
3982    fn source(&self) -> Option<&(dyn Error + 'static)> {
3983        match self {
3984            Self::Spawn { source, .. }
3985            | Self::Cgroup { source, .. }
3986            | Self::Wait { source, .. }
3987            | Self::Kill { source, .. } => Some(source),
3988            Self::Forwarding(err) => Some(err),
3989            Self::Registry(err) => Some(err),
3990            Self::LaunchNonce { .. }
3991            | Self::InvalidSpec { .. }
3992            | Self::ReloadUnavailable { .. }
3993            | Self::Disabled { .. }
3994            | Self::ReloadFailed { .. }
3995            | Self::RegistrationStillActive { .. }
3996            | Self::StatePoisoned { .. }
3997            | Self::CommandClosed { .. }
3998            | Self::SwapInProgress { .. }
3999            | Self::SwapRefused { .. }
4000            | Self::SwapFailed { .. } => None,
4001        }
4002    }
4003}
4004
4005pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4006    if spec.module_id.trim().is_empty() {
4007        return Err(SuperviseError::InvalidSpec {
4008            reason: "module_id must not be empty".to_string(),
4009        });
4010    }
4011
4012    Ok(())
4013}
4014
4015#[derive(Debug, Default)]
4016struct HealthProbeRuntime {
4017    configured_health: Option<HealthConfig>,
4018    registered_connection: Option<crate::ConnectionId>,
4019    advertised: bool,
4020    next_probe_at: Option<Instant>,
4021    probe_index: u64,
4022}
4023
4024fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4025    lock_snapshot(snapshot)
4026        .ok()
4027        .and_then(|state| state.spawned_protocol)
4028        .unwrap_or(spec.protocol)
4029}
4030
4031impl HealthProbeRuntime {
4032    fn refresh_registration(
4033        &mut self,
4034        spec: &ModuleSpec,
4035        runtime: &SupervisorRuntimeConfig,
4036        registry: &Registry,
4037        snapshot: &SharedSnapshot,
4038    ) {
4039        if self.configured_health.as_ref() != Some(&runtime.health) {
4040            self.configured_health = Some(runtime.health.clone());
4041            self.next_probe_at = None;
4042            self.registered_connection = None;
4043            self.probe_index = 0;
4044        }
4045        // A non-wire process never registers. Only an explicitly configured
4046        // HTTP endpoint can arm its health probe; an absent HELLO is not a
4047        // health failure for that kind of process.
4048        if running_protocol(spec, snapshot) == ModuleProtocol::None {
4049            self.registered_connection = None;
4050            self.advertised = runtime.health.http.is_some();
4051            if !self.advertised {
4052                self.next_probe_at = None;
4053                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4054                    state.health = ModuleHealthStatus::default();
4055                });
4056            } else if self.next_probe_at.is_none() {
4057                self.next_probe_at = Some(
4058                    Instant::now()
4059                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4060                );
4061            }
4062            return;
4063        }
4064
4065        let registration = match registry.get_module(&spec.module_id) {
4066            Ok(registration) => registration,
4067            Err(err) => {
4068                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4069                self.advertised = false;
4070                self.next_probe_at = None;
4071                return;
4072            }
4073        };
4074
4075        let Some(registration) = registration else {
4076            self.registered_connection = None;
4077            self.advertised = false;
4078            self.next_probe_at = None;
4079            return;
4080        };
4081
4082        let advertised = registration
4083            .control_ops
4084            .iter()
4085            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4086        if !advertised {
4087            self.registered_connection = Some(registration.connection_id);
4088            self.advertised = false;
4089            self.next_probe_at = None;
4090            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4091                state.health.status = SupervisorHealthStatus::Unknown;
4092                state.health.consecutive_failures = 0;
4093                state.health.last_probe_ms = None;
4094                state.health.detail = None;
4095                state.health.metrics = None;
4096            });
4097            return;
4098        }
4099
4100        let reregistered = self.registered_connection != Some(registration.connection_id);
4101        self.registered_connection = Some(registration.connection_id);
4102        self.advertised = true;
4103        if reregistered || self.next_probe_at.is_none() {
4104            self.probe_index = 0;
4105            self.next_probe_at = Some(
4106                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4107            );
4108            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4109                state.health.status = SupervisorHealthStatus::Unknown;
4110                state.health.consecutive_failures = 0;
4111                state.health.detail = None;
4112                state.health.metrics = None;
4113            });
4114        }
4115    }
4116
4117    fn wake_after(&self) -> Duration {
4118        if !self.advertised {
4119            return REGISTRY_RELEASE_POLL;
4120        }
4121        self.next_probe_at
4122            .map(|next| next.saturating_duration_since(Instant::now()))
4123            .unwrap_or(REGISTRY_RELEASE_POLL)
4124    }
4125
4126    fn due(&self) -> bool {
4127        self.advertised
4128            && self
4129                .next_probe_at
4130                .is_some_and(|next| Instant::now() >= next)
4131    }
4132
4133    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4134        self.probe_index = self.probe_index.wrapping_add(1);
4135        self.next_probe_at = Some(
4136            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4137        );
4138    }
4139}
4140
4141/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4142///
4143/// This was a struct with a single `message: String`, and every one of the
4144/// fifteen construction sites collapsed into it. Each site knows exactly what it
4145/// saw -- the lane is gone, the module did not answer in time, the module
4146/// answered with the wrong thing -- and `handle_health_probe_failure` then
4147/// treated all of them identically: increment a counter, compare to a threshold,
4148/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4149/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4150///
4151/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4152///
4153/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4154///   answer on it again.
4155/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4156///   AND with a perfectly healthy one that lost a CPU race -- which is what
4157///   happens under machine load, and is how this supervisor killed a healthy
4158///   module three times in one day.
4159/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4160///   Restarting on it is defensible, but it is not the silence case and should
4161///   never be counted as one.
4162/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4163///   anything, so it cannot be evidence about the module at all.
4164///
4165/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4166/// one that fires most often, and while every variant collapsed into one string
4167/// it carried the same weight as the strongest.
4168///
4169/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4170/// DESIGN and a reader stopping at it gets the build backwards: the restart
4171/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4172/// probes still increment the failure streak and drive escalation at the
4173/// threshold (see `is_proof_of_death` below for why that is deliberate and
4174/// what gates the change). Absence of evidence restarts modules today.
4175#[derive(Debug)]
4176enum HealthProbeEvidence {
4177    /// The module's control lane is gone. Proof of death.
4178    LaneDead,
4179    /// No reply within the deadline. Proves nothing about the module's state.
4180    NoAnswer,
4181    /// The module replied, but not with a usable health report. Proves it is alive.
4182    BadAnswer,
4183    /// The daemon could not ask. Says nothing about the module.
4184    Misconfigured,
4185}
4186
4187#[derive(Debug)]
4188struct HealthProbeError {
4189    evidence: HealthProbeEvidence,
4190    message: String,
4191}
4192
4193impl HealthProbeError {
4194    fn lane_dead(message: impl Into<String>) -> Self {
4195        Self::with(HealthProbeEvidence::LaneDead, message)
4196    }
4197
4198    fn no_answer(message: impl Into<String>) -> Self {
4199        Self::with(HealthProbeEvidence::NoAnswer, message)
4200    }
4201
4202    fn bad_answer(message: impl Into<String>) -> Self {
4203        Self::with(HealthProbeEvidence::BadAnswer, message)
4204    }
4205
4206    fn misconfigured(message: impl Into<String>) -> Self {
4207        Self::with(HealthProbeEvidence::Misconfigured, message)
4208    }
4209
4210    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4211        Self {
4212            evidence,
4213            message: message.into(),
4214        }
4215    }
4216
4217    /// Whether this observation is proof the module cannot serve.
4218    ///
4219    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4220    /// variant that fires under CPU starvation, and treating it as proof is the
4221    /// defect this enum exists to make impossible to reintroduce silently.
4222    ///
4223    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4224    /// to restart also needs a bound for the case it excludes -- a genuinely
4225    /// wedged module, alive but never answering -- and that bound must come from
4226    /// the distribution of real late-answer latencies, which nothing measures
4227    /// yet. Landing the classification first makes the later change a one-line
4228    /// decision against evidence that already exists, rather than two unproven
4229    /// changes at once.
4230    #[allow(dead_code)]
4231    fn is_proof_of_death(&self) -> bool {
4232        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4233    }
4234
4235    /// Short stable label for logs and the health snapshot.
4236    ///
4237    /// An operator reading `ck health` currently cannot tell "the module is gone"
4238    /// from "the module did not answer in five seconds", because both render as
4239    /// prose in the same field. These labels are what make the two
4240    /// distinguishable at a glance, and they are what a later restart-policy
4241    /// change will be argued from.
4242    fn label(&self) -> &'static str {
4243        match self.evidence {
4244            HealthProbeEvidence::LaneDead => "lane-dead",
4245            HealthProbeEvidence::NoAnswer => "no-answer",
4246            HealthProbeEvidence::BadAnswer => "bad-answer",
4247            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4248        }
4249    }
4250}
4251
4252impl fmt::Display for HealthProbeError {
4253    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4254        f.write_str(&self.message)
4255    }
4256}
4257
4258async fn run_health_probe_cycle(
4259    spec: &ModuleSpec,
4260    runtime: &SupervisorRuntimeConfig,
4261    registry: &Registry,
4262    process_liveness: &SupervisorProcessLiveness,
4263    snapshot: &SharedSnapshot,
4264    child: &mut Option<SupervisedChild>,
4265) {
4266    let now_ms = unix_ms_now();
4267    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4268        .then_some(runtime.health.http.as_deref())
4269        .flatten();
4270    let result = match http {
4271        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4272        None => probe_module_health(&spec.module_id, runtime, None).await,
4273    };
4274    match result {
4275        Ok(report) => {
4276            handle_health_report(
4277                spec,
4278                runtime,
4279                registry,
4280                process_liveness,
4281                snapshot,
4282                child,
4283                report,
4284                now_ms,
4285            )
4286            .await;
4287        }
4288        Err(err) => {
4289            if http.is_some() {
4290                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4291                    state.health.status = SupervisorHealthStatus::Failing;
4292                });
4293            }
4294            handle_health_probe_failure(
4295                spec,
4296                runtime,
4297                registry,
4298                process_liveness,
4299                snapshot,
4300                child,
4301                err,
4302                now_ms,
4303            )
4304            .await;
4305        }
4306    }
4307}
4308
4309pub(crate) struct HttpProbeTarget<'a> {
4310    address: std::net::SocketAddr,
4311    localhost: bool,
4312    authority: &'a str,
4313    path: String,
4314}
4315
4316/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4317/// or TLS. A URL cannot turn a local health check into an outbound connection.
4318pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4319    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4320        return Err("must not contain whitespace, controls, or a fragment".into());
4321    }
4322    let rest = url
4323        .strip_prefix("http://")
4324        .ok_or("must use plain http://")?;
4325    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4326    let (authority, suffix) = rest.split_at(split);
4327    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4328        ("::1", rest)
4329    } else {
4330        let split = authority.find(':').unwrap_or(authority.len());
4331        authority.split_at(split)
4332    };
4333    let ip: std::net::IpAddr = match host {
4334        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4335        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4336        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4337    };
4338    let port = if port.is_empty() {
4339        80
4340    } else {
4341        port.strip_prefix(':')
4342            .and_then(|p| p.parse::<u16>().ok())
4343            .filter(|p| *p > 0)
4344            .ok_or("must have a valid nonzero TCP port")?
4345    };
4346    let path = if suffix.is_empty() {
4347        "/".into()
4348    } else if suffix.starts_with('?') {
4349        format!("/{suffix}")
4350    } else {
4351        suffix.into()
4352    };
4353    Ok(HttpProbeTarget {
4354        address: std::net::SocketAddr::new(ip, port),
4355        localhost: host == "localhost",
4356        authority,
4357        path,
4358    })
4359}
4360
4361async fn probe_http_health(
4362    url: &str,
4363    deadline: Duration,
4364) -> Result<HealthReport, HealthProbeError> {
4365    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4366    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4367    // Keep partial diagnostics outside the timed future so cancellation does
4368    // not discard a status line or body bytes already received.
4369    let mut response_status = String::new();
4370    let mut body = Vec::new();
4371    let probe = async {
4372        // Resolve localhost ourselves so a hosts-file override cannot turn
4373        // this into an outbound request, while IPv6-only local servers work.
4374        let connection = match tokio::net::TcpStream::connect(target.address).await {
4375            Err(_) if target.localhost => {
4376                tokio::net::TcpStream::connect((
4377                    std::net::Ipv6Addr::LOCALHOST,
4378                    target.address.port(),
4379                ))
4380                .await
4381            }
4382            result => result,
4383        };
4384        let mut stream = connection.map_err(|error| {
4385            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4386        })?;
4387        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4388            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4389        let mut reader = BufReader::new(stream);
4390        let mut budget = 16 * 1024;
4391        let status = http_line(&mut reader, &mut budget).await?;
4392        let mut words = status.split_ascii_whitespace();
4393        let version = words.next();
4394        let code = words
4395            .next()
4396            .filter(|word| word.len() == 3)
4397            .and_then(|word| word.parse::<u16>().ok());
4398        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4399            || !code.is_some_and(|code| (100..600).contains(&code))
4400        {
4401            return Err(HealthProbeError::bad_answer(format!(
4402                "invalid HTTP status: {status}"
4403            )));
4404        }
4405        let code = code.expect("validated status code");
4406        response_status = status.clone();
4407        let mut length = None;
4408        let mut chunked = false;
4409        loop {
4410            let line = http_line(&mut reader, &mut budget).await?;
4411            if line.is_empty() {
4412                break;
4413            }
4414            if let Some((name, value)) = line.split_once(':') {
4415                if name.eq_ignore_ascii_case("content-length") {
4416                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4417                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4418                    })?);
4419                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4420                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4421                }
4422            }
4423        }
4424        if chunked {
4425            while body.len() < 200 {
4426                let line = http_line(&mut reader, &mut budget).await?;
4427                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4428                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4429                if size == 0 {
4430                    break;
4431                }
4432                let count = size.min((200 - body.len()) as u64) as usize;
4433                let start = body.len();
4434                (&mut reader)
4435                    .take(count as u64)
4436                    .read_to_end(&mut body)
4437                    .await
4438                    .map_err(|error| {
4439                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4440                    })?;
4441                if body.len() - start != count {
4442                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4443                }
4444                if size > count as u64 || body.len() == 200 {
4445                    break;
4446                }
4447                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4448                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4449                }
4450            }
4451        } else {
4452            reader
4453                .take(length.unwrap_or(200).min(200))
4454                .read_to_end(&mut body)
4455                .await
4456                .map_err(|error| {
4457                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4458                })?;
4459        }
4460        if (200..300).contains(&code) {
4461            Ok(HealthReport::ok())
4462        } else {
4463            Err(HealthProbeError::bad_answer(
4464                "HTTP health endpoint returned non-2xx",
4465            ))
4466        }
4467    };
4468    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4469        Err(HealthProbeError::no_answer(format!(
4470            "HTTP probe timed out after {deadline:?}"
4471        )))
4472    });
4473    if let Err(error) = &mut result {
4474        if !response_status.is_empty() {
4475            error.message = format!(
4476                "{}; {response_status}: {}",
4477                error.message,
4478                String::from_utf8_lossy(&body)
4479            );
4480        }
4481    }
4482    result
4483}
4484
4485async fn http_line(
4486    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4487    remaining: &mut usize,
4488) -> Result<String, HealthProbeError> {
4489    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4490    let mut line = Vec::new();
4491    (&mut *reader)
4492        .take(*remaining as u64)
4493        .read_until(b'\n', &mut line)
4494        .await
4495        .map_err(|error| {
4496            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4497        })?;
4498    *remaining -= line.len();
4499    if !line.ends_with(b"\r\n") {
4500        return Err(HealthProbeError::bad_answer(
4501            "HTTP headers are incomplete or exceed 16 KiB",
4502        ));
4503    }
4504    line.truncate(line.len() - 2);
4505    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4506}
4507
4508async fn probe_module_health(
4509    module_id: &str,
4510    runtime: &SupervisorRuntimeConfig,
4511    drain_deadline: Option<Instant>,
4512) -> Result<HealthReport, HealthProbeError> {
4513    let Some(forwarding) = runtime.forwarding.as_ref() else {
4514        return Err(HealthProbeError::misconfigured(
4515            "supervisor was not configured with a forwarding table",
4516        ));
4517    };
4518    let probe_started_at = Instant::now();
4519    let mut deadline = probe_started_at + runtime.health.deadline;
4520    if let Some(drain_deadline) = drain_deadline {
4521        deadline = deadline.min(drain_deadline);
4522    }
4523    let pending = if drain_deadline.is_some() {
4524        forwarding.begin_drain_health_probe_rpc_for(
4525            module_id,
4526            MODULE_CONTROL_OP_HEALTH_CHECK,
4527            probe_started_at,
4528            deadline,
4529        )
4530    } else {
4531        forwarding.begin_health_probe_rpc_for(
4532            module_id,
4533            MODULE_CONTROL_OP_HEALTH_CHECK,
4534            probe_started_at,
4535            deadline,
4536        )
4537    }
4538    .map_err(|err| {
4539        // The endpoint is not registered, so there is no live control lane to
4540        // ask. That is the module being absent, not slow.
4541        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4542    })?;
4543    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4544}
4545
4546/// [`probe_module_health`] for one endpoint rather than the id's active one.
4547///
4548/// A swap probes two processes that no by-id lookup reaches: its candidate
4549/// before cutover, and its superseded incumbent (for busy gauges) while the
4550/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4551/// bounds the by-id drain probe.
4552async fn probe_endpoint_health(
4553    endpoint: crate::ModuleEndpointId,
4554    runtime: &SupervisorRuntimeConfig,
4555    deadline_cap: Option<Instant>,
4556) -> Result<HealthReport, HealthProbeError> {
4557    let Some(forwarding) = runtime.forwarding.as_ref() else {
4558        return Err(HealthProbeError::misconfigured(
4559            "supervisor was not configured with a forwarding table",
4560        ));
4561    };
4562    let probe_started_at = Instant::now();
4563    let mut deadline = probe_started_at + runtime.health.deadline;
4564    if let Some(cap) = deadline_cap {
4565        deadline = deadline.min(cap);
4566    }
4567    let pending = forwarding
4568        .begin_endpoint_health_probe_rpc_for(
4569            endpoint,
4570            MODULE_CONTROL_OP_HEALTH_CHECK,
4571            probe_started_at,
4572            deadline,
4573        )
4574        .map_err(|err| {
4575            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4576        })?;
4577    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4578}
4579
4580/// Send a begun health probe and classify its answer.
4581async fn await_health_probe(
4582    forwarding: &ForwardingTable,
4583    pending: PendingModuleControlRpc,
4584    deadline: Instant,
4585    probe_budget: Duration,
4586) -> Result<HealthReport, HealthProbeError> {
4587    let PendingModuleControlRpc {
4588        endpoint,
4589        module_sink,
4590        negotiated_ver,
4591        corr,
4592        receiver,
4593    } = pending;
4594    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4595        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4596    })?;
4597    let frame = Frame::build_with_version(
4598        negotiated_ver,
4599        FrameType::Request,
4600        control_flags(),
4601        0,
4602        0,
4603        corr,
4604        body,
4605    )
4606    .map_err(|err| {
4607        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4608    })?;
4609
4610    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4611    // blocks waiting for capacity when the module's egress queue is full, and an
4612    // unbounded await here freezes the whole supervision actor (it stops polling
4613    // Child::wait and supervisor commands), making the module unrecoverable
4614    // in-band. On timeout the probe fails like any transport failure.
4615    match timeout_at(deadline, module_sink.send(frame)).await {
4616        Ok(Ok(())) => {}
4617        Ok(Err(err)) => {
4618            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4619            // A closed sink means the module's egress channel is gone -- the
4620            // receiving half is dropped when its connection tears down. Proof.
4621            return Err(HealthProbeError::lane_dead(format!(
4622                "failed to send health.check: {err}"
4623            )));
4624        }
4625        Err(_elapsed) => {
4626            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4627            // A full egress queue means the module is not draining its socket, which
4628            // is consistent with a wedged module AND with one whose reader is merely
4629            // starved. Silence, not proof.
4630            return Err(HealthProbeError::no_answer(
4631                "health.check send timed out before enqueue (module egress full)",
4632            ));
4633        }
4634    }
4635
4636    match timeout_at(deadline, receiver).await {
4637        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4638        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4639        // and those prove it is alive even though the probe failed.
4640        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4641            response.health_report().ok_or_else(|| {
4642                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4643            })
4644        }
4645        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4646            format!("health.check rejected: {}", body.message),
4647        )),
4648        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4649            Err(HealthProbeError::lane_dead(message))
4650        }
4651        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4652            Err(HealthProbeError::bad_answer(message))
4653        }
4654        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4655            Err(HealthProbeError::bad_answer(format!(
4656                "expected module-control op '{expected}', got '{actual}'"
4657            )))
4658        }
4659        // A reply that crosses the deadline before this waiter observes it is
4660        // still proof of life. The forwarding path records its end-to-end latency
4661        // before delivering this classification.
4662        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4663            "module answered health.check after its daemon deadline",
4664        )),
4665        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4666            "health.check waiter was canceled before the module responded",
4667        )),
4668        Err(_) => {
4669            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4670            Err(HealthProbeError::no_answer(format!(
4671                "module did not answer health.check within {probe_budget:?}"
4672            )))
4673        }
4674    }
4675}
4676
4677#[allow(clippy::too_many_arguments)]
4678async fn handle_health_report(
4679    spec: &ModuleSpec,
4680    runtime: &SupervisorRuntimeConfig,
4681    registry: &Registry,
4682    process_liveness: &SupervisorProcessLiveness,
4683    snapshot: &SharedSnapshot,
4684    child: &mut Option<SupervisedChild>,
4685    report: HealthReport,
4686    now_ms: u64,
4687) {
4688    let status = supervisor_health_status(report.status);
4689    let detail = report.detail.clone();
4690    let metrics = truncate_health_metrics(report.metrics);
4691    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4692        state.health.status = status;
4693        state.health.last_probe_ms = Some(now_ms);
4694        state.health.detail = detail.clone();
4695        state.health.metrics = metrics.clone();
4696        state.health.consecutive_failures = 0;
4697    });
4698
4699    let action = match report.status {
4700        HealthStatus::Ok => return,
4701        HealthStatus::Degraded => runtime.health.on_degraded,
4702        HealthStatus::Failing => runtime.health.on_failing,
4703    };
4704    apply_l3_health_action(
4705        spec,
4706        runtime,
4707        registry,
4708        process_liveness,
4709        snapshot,
4710        child,
4711        status,
4712        detail.as_deref(),
4713        action,
4714        now_ms,
4715    )
4716    .await;
4717}
4718
4719#[allow(clippy::too_many_arguments)]
4720async fn handle_health_probe_failure(
4721    spec: &ModuleSpec,
4722    runtime: &SupervisorRuntimeConfig,
4723    registry: &Registry,
4724    process_liveness: &SupervisorProcessLiveness,
4725    snapshot: &SharedSnapshot,
4726    child: &mut Option<SupervisedChild>,
4727    err: HealthProbeError,
4728    now_ms: u64,
4729) {
4730    let threshold = runtime.health.failure_threshold.max(1);
4731    let mut failures = 0;
4732    // Carry the evidence class into the operator-visible detail. Without it,
4733    // "module did not answer within 5s" and "the control lane is gone" are two
4734    // prose strings in the same field, and the reader has to know the codebase to
4735    // tell which one is proof of anything.
4736    let detail = format!("[{}] {err}", err.label());
4737    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4738        // A failed wire probe invalidates the last report, even before the
4739        // restart threshold. HTTP probes already mark failures as Failing.
4740        if state.spawned_protocol.unwrap_or(spec.protocol) == ModuleProtocol::Subc {
4741            state.health.status = SupervisorHealthStatus::Unknown;
4742        }
4743        state.health.last_probe_ms = Some(now_ms);
4744        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4745        state.health.detail = Some(detail.clone());
4746        state.health.metrics = None;
4747        failures = state.health.consecutive_failures;
4748    });
4749
4750    if failures < threshold {
4751        warn!(
4752            module_id = %spec.module_id,
4753            consecutive_failures = failures,
4754            threshold,
4755            evidence = err.label(),
4756            detail = %detail,
4757            "health.check probe failed"
4758        );
4759        return;
4760    }
4761
4762    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4763        state.state = ModuleState::Unresponsive;
4764        state.health.status = SupervisorHealthStatus::Unresponsive;
4765    });
4766    // The evidence class is logged at the kill site because this is the line an
4767    // operator reads after an unexplained restart. A streak of `no-answer` under
4768    // machine load is the known false-positive shape; a `lane-dead` is not.
4769    if runtime.health.critical {
4770        error!(
4771            module_id = %spec.module_id,
4772            status = "unresponsive",
4773            evidence = err.label(),
4774            detail = %detail,
4775            "critical module health alert"
4776        );
4777    } else {
4778        warn!(
4779            module_id = %spec.module_id,
4780            status = "unresponsive",
4781            evidence = err.label(),
4782            detail = %detail,
4783            "module health threshold breached"
4784        );
4785    }
4786    if let Err(err) = health_restart_child(
4787        spec,
4788        runtime,
4789        registry,
4790        process_liveness,
4791        snapshot,
4792        child,
4793        SupervisorHealthStatus::Unresponsive,
4794        Some(&detail),
4795        now_ms,
4796    )
4797    .await
4798    {
4799        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4800    }
4801}
4802
4803#[allow(clippy::too_many_arguments)]
4804async fn apply_l3_health_action(
4805    spec: &ModuleSpec,
4806    runtime: &SupervisorRuntimeConfig,
4807    registry: &Registry,
4808    process_liveness: &SupervisorProcessLiveness,
4809    snapshot: &SharedSnapshot,
4810    child: &mut Option<SupervisedChild>,
4811    status: SupervisorHealthStatus,
4812    detail: Option<&str>,
4813    action: HealthAction,
4814    now_ms: u64,
4815) {
4816    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4817    match action {
4818        HealthAction::Report => {
4819            info!(
4820                module_id = %spec.module_id,
4821                status = ?status,
4822                detail,
4823                "module reported non-ok health"
4824            );
4825        }
4826        HealthAction::Alert => {
4827            error!(
4828                module_id = %spec.module_id,
4829                status = ?status,
4830                detail,
4831                "module health alert"
4832            );
4833        }
4834        HealthAction::Restart => {
4835            if let Err(err) = health_restart_child(
4836                spec,
4837                runtime,
4838                registry,
4839                process_liveness,
4840                snapshot,
4841                child,
4842                status,
4843                detail,
4844                now_ms,
4845            )
4846            .await
4847            {
4848                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4849            }
4850        }
4851    }
4852}
4853
4854#[allow(clippy::too_many_arguments)]
4855async fn health_restart_child(
4856    spec: &ModuleSpec,
4857    runtime: &SupervisorRuntimeConfig,
4858    registry: &Registry,
4859    process_liveness: &SupervisorProcessLiveness,
4860    snapshot: &SharedSnapshot,
4861    child: &mut Option<SupervisedChild>,
4862    status: SupervisorHealthStatus,
4863    detail: Option<&str>,
4864    now_ms: u64,
4865) -> Result<(), SuperviseError> {
4866    let (enabled, schedule) = {
4867        let mut state = lock_snapshot(snapshot)?;
4868        let enabled = state.enabled;
4869        let schedule = if enabled {
4870            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4871        } else {
4872            None
4873        };
4874        (enabled, schedule)
4875    };
4876
4877    if !enabled {
4878        return Err(SuperviseError::Disabled {
4879            module_id: spec.module_id.clone(),
4880        });
4881    }
4882
4883    if schedule.is_none() {
4884        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4885        error!(
4886            module_id = %spec.module_id,
4887            status = ?status,
4888            detail,
4889            max_restarts = runtime.restart_policy.max_restarts,
4890            window_secs = runtime.restart_policy.window.as_secs(),
4891            reason = %runtime.restart_policy.budget_exhausted_detail(),
4892            "health restart budget exhausted; marking module failed"
4893        );
4894        let stop_notice = begin_forwarding_drain_if_configured(
4895            spec,
4896            runtime,
4897            registry,
4898            snapshot,
4899            Some(true),
4900            RouteCloseReason::Disable,
4901        )
4902        .await?;
4903        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4904            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4905        })?;
4906        drain_optional_child(
4907            &spec.module_id,
4908            spec.protocol,
4909            stop_notice,
4910            registry,
4911            runtime.forwarding.as_deref(),
4912            snapshot,
4913            &runtime.terminal_ring,
4914            &runtime.spawn_events,
4915            child,
4916            runtime.drain_timeout,
4917            ModuleState::Failed,
4918            Some(true),
4919        )
4920        .await?;
4921        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4922        return Ok(());
4923    }
4924
4925    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4926    let mut restart_count = 0;
4927    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4928        restart_count = state.crash_restarts.len();
4929        state.state = ModuleState::Unresponsive;
4930        state.health.status = status;
4931        state.health.last_action = Some(HealthAction::Restart.to_string());
4932        state.health.last_action_ms = Some(now_ms);
4933    })?;
4934    warn!(
4935        module_id = %spec.module_id,
4936        status = ?status,
4937        detail,
4938        restart_count,
4939        restart_in_window = schedule.restart_in_window,
4940        delay_ms = schedule.delay.as_millis() as u64,
4941        "health-triggered module restart"
4942    );
4943
4944    let stop_notice = begin_forwarding_drain_if_configured(
4945        spec,
4946        runtime,
4947        registry,
4948        snapshot,
4949        Some(true),
4950        RouteCloseReason::Restart,
4951    )
4952    .await?;
4953    drain_optional_child(
4954        &spec.module_id,
4955        spec.protocol,
4956        stop_notice,
4957        registry,
4958        runtime.forwarding.as_deref(),
4959        snapshot,
4960        &runtime.terminal_ring,
4961        &runtime.spawn_events,
4962        child,
4963        runtime.drain_timeout,
4964        ModuleState::Restarting,
4965        Some(true),
4966    )
4967    .await?;
4968    schedule_respawn(
4969        runtime,
4970        snapshot,
4971        &spec.module_id,
4972        schedule.delay,
4973        RespawnKind::Spawn,
4974    )
4975}
4976
4977fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4978    if let Some(reply) = runtime
4979        .deferred_reload_reply
4980        .lock()
4981        .unwrap_or_else(|p| p.into_inner())
4982        .take()
4983    {
4984        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4985            module_id: module_id.to_string(),
4986            reason: reason.to_string(),
4987        }));
4988    }
4989}
4990
4991fn schedule_respawn(
4992    runtime: &SupervisorRuntimeConfig,
4993    snapshot: &SharedSnapshot,
4994    module_id: &str,
4995    delay: Duration,
4996    kind: RespawnKind,
4997) -> Result<(), SuperviseError> {
4998    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4999    update_snapshot(snapshot, Some(module_id), |state| {
5000        state.respawn_pending = true
5001    })?;
5002    *runtime
5003        .scheduled_respawn
5004        .lock()
5005        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5006        deadline: Instant::now() + delay,
5007        kind,
5008    });
5009    Ok(())
5010}
5011
5012fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5013    let _ = update_snapshot(snapshot, Some(module_id), |state| {
5014        state.health.last_action = Some(action);
5015        state.health.last_action_ms = Some(now_ms);
5016    });
5017}
5018
5019fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5020    match status {
5021        HealthStatus::Ok => SupervisorHealthStatus::Ok,
5022        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5023        HealthStatus::Failing => SupervisorHealthStatus::Failing,
5024    }
5025}
5026
5027/// Caps the metrics blob stored in the cached supervisor snapshot, which is
5028/// returned to every `supervisor.list` and `supervisor.health` caller.
5029///
5030/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
5031/// path: that request exists to return a module's complete metrics object, and
5032/// `ck health <module-id>` documents it as the way to see what the cached view
5033/// truncates. The asymmetry is the feature.
5034///
5035/// So a new caller must decide which side it is on rather than assume the cap is
5036/// universal. Reaching for it on a fresh-probe path would silently reintroduce
5037/// the truncation that path exists to avoid.
5038fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5039    let metrics = metrics?;
5040    match serde_json::to_vec(&metrics) {
5041        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5042            "truncated": true,
5043            "original_bytes": encoded.len(),
5044        })),
5045        Ok(_) | Err(_) => Some(metrics),
5046    }
5047}
5048
5049/// Spread health probes so a fleet-wide restart does not converge them.
5050///
5051/// The delay is derived from the module id and probe index rather than a random
5052/// source, so it is deterministic per module: a module keeps its own offset
5053/// across daemon restarts instead of re-rolling into a collision.
5054fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5055    if cadence.is_zero() {
5056        return Duration::ZERO;
5057    }
5058    let cadence_ms = cadence.as_millis() as u64;
5059    // This early return is REDUNDANT, deliberately, and a mutation run will show
5060    // it surviving removal. Recording why here so the next person to notice does
5061    // not have to re-derive it:
5062    //
5063    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
5064    //   a zero cadence and builds the Duration from whole milliseconds, so a
5065    //   sub-millisecond cadence cannot come from config.
5066    // - Even if reached it changes no answer. The `.max(1)` below makes the span
5067    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
5068    //   -- exactly what this returns.
5069    //
5070    // Kept as a guard against a future widening of the config parser (accepting
5071    // microseconds, say), which would make the sub-millisecond case reachable.
5072    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
5073    // divides by zero. Remove this and nothing changes.
5074    if cadence_ms == 0 {
5075        return cadence;
5076    }
5077    // Note that this never returns less than one cadence, including for the FIRST
5078    // probe. So a freshly registered module reports health `unknown` for a full
5079    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5080    // ready to answer.
5081    //
5082    // That is a property of the supervisor's schedule, not of any module: an
5083    // operator watching a restart sees `unknown` and cannot tell it from a module
5084    // that is slow to warm. Measured on two unrelated modules, both flipping to
5085    // `ok` between 22s and 32s after restart.
5086    //
5087    // Left as-is because spreading the first probe is what keeps a fleet-wide
5088    // restart from firing fourteen simultaneous probes into a cold machine. The
5089    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5090    // that thundering herd for a faster first reading.
5091    let jitter_span = (cadence_ms / 10).max(1);
5092    let hash = module_id.as_bytes().iter().fold(
5093        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5094        |acc, byte| {
5095            acc.wrapping_mul(1099511628211)
5096                .wrapping_add(u64::from(*byte))
5097        },
5098    );
5099    cadence + Duration::from_millis(hash % jitter_span)
5100}
5101
5102#[cfg(test)]
5103mod tests {
5104    use super::*;
5105
5106    #[test]
5107    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5108        let handle = SupervisorHandle::new();
5109        let module_id = "readded-tombstone";
5110        handle.record_rescan_removal(module_id);
5111        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5112
5113        handle.apply_identity_configuration(&ModuleSpec {
5114            module_id: module_id.to_string(),
5115            program: PathBuf::from("/test/module"),
5116            args: Vec::new(),
5117            env: Vec::new(),
5118            reserved: false,
5119            reserved_prefixes: Vec::new(),
5120            protocol: ModuleProtocol::Subc,
5121            overlap: Default::default(),
5122        });
5123
5124        assert!(
5125            handle.removal_tombstone_age_ms(module_id).is_none(),
5126            "a re-added module must not retain a stale removal tombstone"
5127        );
5128    }
5129
5130    /// What one module's owner looked like from the control plane at the
5131    /// instant after its first process was spawned.
5132    #[derive(Debug, PartialEq, Eq)]
5133    struct OwnerInSpawnWindow {
5134        module_id: String,
5135        configured: bool,
5136        on_roster: bool,
5137        admission_refusal: Option<&'static str>,
5138    }
5139
5140    /// A supervised module's process can connect, register, sync its scopes
5141    /// and describe them as soon as it is spawned, which is BEFORE the
5142    /// supervisor puts the module on the roster. In that window the owner must
5143    /// already read as configured, so a scoped `route.open` against it is
5144    /// refused as retryable `scope_not_synced` and not as terminal
5145    /// `scope_not_live` ("will never sync").
5146    ///
5147    /// The hook runs in exactly that window on every path that takes on a new
5148    /// module, so no race with a real child is needed: `on_roster: false`
5149    /// proves each observation was taken before the roster insert.
5150    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5151    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5152        use crate::scopes::ScopeTable;
5153        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5154
5155        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5156        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5157            module_id: module_id.to_string(),
5158            program,
5159            args: Vec::new(),
5160            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5161                .into_iter()
5162                .map(|key| (key.to_string(), dir.path().display().to_string()))
5163                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5164                .collect(),
5165            reserved: false,
5166            reserved_prefixes: Vec::new(),
5167            protocol: ModuleProtocol::Subc,
5168            overlap: Default::default(),
5169        };
5170        let live = super::terminal_history_tests::fake_aft_stub_path();
5171        let missing = dir.path().join("definitely-missing-module");
5172
5173        let handle = SupervisorHandle::new();
5174        let mut supervisor =
5175            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5176                .with_handle(handle.clone());
5177        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5178        let hook_handle = handle.clone();
5179        let hook_observed = Arc::clone(&observed);
5180        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5181            // Exactly what the control plane computes for a scoped route.open
5182            // naming this module as the owner of a scope it has not synced.
5183            let configured = hook_handle.is_configured(module_id);
5184            let selector = ScopeSelector {
5185                owner: Principal::Reserved {
5186                    module_id: module_id.to_string(),
5187                },
5188                scope_ref: "s".to_string(),
5189                scope_epoch: Some(1),
5190            };
5191            let carrier = Principal::Reserved {
5192                module_id: "carrier".to_string(),
5193            };
5194            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5195                .admit(&carrier, module_id, &selector, configured)
5196            {
5197                Ok(_) => None,
5198                Err(refusal) => Some(refusal.code),
5199            };
5200            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5201                module_id: module_id.to_string(),
5202                configured,
5203                on_roster: hook_handle.get(module_id).is_some(),
5204                admission_refusal,
5205            });
5206        })));
5207
5208        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5209        let configured = supervisor
5210            .supervise_configured(stub("configured", live.clone()), true)
5211            .unwrap();
5212        let with_health = supervisor
5213            .supervise_configured_with_health(
5214                stub("with-health", live.clone()),
5215                true,
5216                HealthConfig::default(),
5217                None,
5218                RestartPolicy::default(),
5219            )
5220            .unwrap();
5221        // The failed-spawn path still puts the module on the roster (as
5222        // failed), so it is configured throughout.
5223        let failed = supervisor
5224            .supervise_configured_with_health(
5225                stub("failed-spawn", missing.clone()),
5226                true,
5227                HealthConfig::default(),
5228                None,
5229                RestartPolicy::default(),
5230            )
5231            .unwrap();
5232        // A failed plain `spawn` puts nothing on the roster, so its mark is
5233        // taken back once the spawn has failed.
5234        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5235
5236        let expected = [
5237            "plain",
5238            "configured",
5239            "with-health",
5240            "failed-spawn",
5241            "spawn-error",
5242        ]
5243        .into_iter()
5244        .map(|module_id| OwnerInSpawnWindow {
5245            module_id: module_id.to_string(),
5246            configured: true,
5247            on_roster: false,
5248            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5249        })
5250        .collect::<Vec<_>>();
5251        assert_eq!(*observed.lock().unwrap(), expected);
5252
5253        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5254            assert!(
5255                handle.get(module_id).is_some(),
5256                "{module_id} is on the roster"
5257            );
5258            assert!(
5259                handle.is_configured(module_id),
5260                "{module_id} stays configured"
5261            );
5262        }
5263        assert!(handle.get("spawn-error").is_none());
5264        assert!(
5265            !handle.is_configured("spawn-error"),
5266            "a plain spawn that failed must not leave its module marked configured"
5267        );
5268
5269        // Leaving the roster clears the mark with it.
5270        handle.retire("failed-spawn");
5271        assert!(!handle.is_configured("failed-spawn"));
5272
5273        for module in [plain, configured, with_health] {
5274            module.stop().await.unwrap();
5275        }
5276        drop(failed);
5277    }
5278
5279    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5280        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5281        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5282            snapshot.process_alive = true;
5283            snapshot.pid = Some(41);
5284            snapshot.spawned_at_ms = Some(42);
5285            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5286            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5287                device: 43,
5288                inode: 44,
5289            });
5290        })
5291        .unwrap();
5292        snapshot
5293    }
5294
5295    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5296        let snapshot = lock_snapshot(snapshot).unwrap();
5297        assert!(!snapshot.process_alive);
5298        assert_eq!(snapshot.pid, None);
5299        assert_eq!(snapshot.spawned_at_ms, None);
5300        assert_eq!(snapshot.spawned_from, None);
5301        assert_eq!(snapshot.spawned_file_identity, None);
5302    }
5303
5304    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5305    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5306        let supervisor =
5307            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5308        let mut runtime = supervisor.runtime_config();
5309        runtime.test_seed_stale_facts_before_enable_spawn = true;
5310        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5311        let mut child = None;
5312        let spec = ModuleSpec {
5313            module_id: "failed-enable-clears-facts".to_string(),
5314            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5315            args: Vec::new(),
5316            env: Vec::new(),
5317            reserved: false,
5318            reserved_prefixes: Vec::new(),
5319            protocol: ModuleProtocol::Subc,
5320            overlap: Default::default(),
5321        };
5322
5323        let result = set_child_enabled(
5324            &spec,
5325            &runtime,
5326            &supervisor.registry,
5327            &supervisor.process_liveness,
5328            &snapshot,
5329            &mut child,
5330            true,
5331        )
5332        .await;
5333
5334        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5335        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5336        assert_snapshot_process_facts_cleared(&snapshot);
5337    }
5338
5339    #[tokio::test]
5340    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5341        let supervisor =
5342            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5343        let runtime = supervisor.runtime_config();
5344        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5345            ModuleState::Restarting,
5346            true,
5347        )));
5348        let spec = ModuleSpec {
5349            module_id: "start-stranded-restarting".to_string(),
5350            program: super::terminal_history_tests::fake_aft_stub_path(),
5351            args: Vec::new(),
5352            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5353            reserved: false,
5354            reserved_prefixes: Vec::new(),
5355            protocol: ModuleProtocol::None,
5356            overlap: Default::default(),
5357        };
5358        let mut child = None;
5359        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5360        assert!(!super::set_child_enabled(
5361            &spec,
5362            &runtime,
5363            &Registry::default(),
5364            &supervisor.process_liveness,
5365            &snapshot,
5366            &mut child,
5367            true
5368        )
5369        .await
5370        .unwrap());
5371        assert!(child.is_none());
5372        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5373        assert!(super::set_child_enabled(
5374            &spec,
5375            &runtime,
5376            &Registry::default(),
5377            &supervisor.process_liveness,
5378            &snapshot,
5379            &mut child,
5380            true
5381        )
5382        .await
5383        .unwrap());
5384        assert_eq!(
5385            lock_snapshot(&snapshot).unwrap().state,
5386            ModuleState::Running
5387        );
5388        let mut child = child.unwrap();
5389        child.start_kill().unwrap();
5390        child.wait().await.unwrap();
5391    }
5392
5393    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5394    async fn failed_reload_spawn_clears_current_process_facts() {
5395        let supervisor =
5396            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5397        let mut runtime = supervisor.runtime_config();
5398        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5399        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5400        let mut child = None;
5401        let spec = ModuleSpec {
5402            module_id: "failed-reload-clears-facts".to_string(),
5403            program: PathBuf::from("/unused/failed-reload-module"),
5404            args: Vec::new(),
5405            env: Vec::new(),
5406            reserved: false,
5407            reserved_prefixes: Vec::new(),
5408            protocol: ModuleProtocol::Subc,
5409            overlap: Default::default(),
5410        };
5411
5412        let result = handle_reload_spawn_failure(
5413            &spec,
5414            &runtime,
5415            &supervisor.process_liveness,
5416            &snapshot,
5417            &mut child,
5418            "forced reload spawn failure".to_string(),
5419        )
5420        .await;
5421
5422        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5423        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5424        assert_snapshot_process_facts_cleared(&snapshot);
5425    }
5426
5427    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5428    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5429        let supervisor =
5430            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5431        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5432        let module = supervisor.supervised_module(
5433            ModuleSpec {
5434                module_id: "drop-clears-facts".to_string(),
5435                program: PathBuf::from("/unused/drop-module"),
5436                args: Vec::new(),
5437                env: Vec::new(),
5438                reserved: false,
5439                reserved_prefixes: Vec::new(),
5440                protocol: ModuleProtocol::Subc,
5441                overlap: Default::default(),
5442            },
5443            supervisor.runtime_config(),
5444            Arc::clone(&snapshot),
5445            None,
5446        );
5447        assert!(!module
5448            .inner
5449            .monitor
5450            .lock()
5451            .unwrap()
5452            .as_ref()
5453            .unwrap()
5454            .is_finished());
5455
5456        drop(module);
5457
5458        assert_eq!(
5459            lock_snapshot(&snapshot).unwrap().state,
5460            ModuleState::Stopped
5461        );
5462        assert_snapshot_process_facts_cleared(&snapshot);
5463    }
5464
5465    #[cfg(unix)]
5466    #[tokio::test]
5467    async fn rescan_preserves_running_protocol_until_respawn() {
5468        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5469        let initial = ModuleSpec {
5470            module_id: "rescan-protocol".into(),
5471            program: PathBuf::from("/bin/sleep"),
5472            args: vec!["60".into()],
5473            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5474                .into_iter()
5475                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5476                .collect(),
5477            reserved: false,
5478            reserved_prefixes: vec![],
5479            protocol: ModuleProtocol::None,
5480            overlap: Default::default(),
5481        };
5482        let supervisor =
5483            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5484        let module = supervisor.spawn(initial.clone()).unwrap();
5485        assert!(module.status().unwrap().live);
5486        let mut next = initial;
5487        next.protocol = ModuleProtocol::Subc;
5488        module
5489            .update_configuration(next.clone(), HealthConfig::default(), None)
5490            .await
5491            .unwrap();
5492        assert!(
5493            module.status().unwrap().live,
5494            "rescan must not require HELLO from the old non-wire process"
5495        );
5496        let runtime = supervisor.runtime_config();
5497        let action = on_child_exit(
5498            &next,
5499            RestartPolicy::default(),
5500            &supervisor.registry,
5501            &module.inner.snapshot,
5502            &runtime.terminal_ring,
5503            &runtime.spawn_events,
5504            &runtime.child_roster,
5505            ExitReport {
5506                kind: ExitKind::Clean,
5507                code: Some(0),
5508                signal: None,
5509                at_ms: unix_ms_now(),
5510            },
5511        )
5512        .await;
5513        assert!(
5514            matches!(action, NextAction::Restart { .. }),
5515            "the old non-wire process's clean exit must restart"
5516        );
5517        module.drain().await.unwrap();
5518    }
5519
5520    #[cfg(unix)]
5521    #[tokio::test]
5522    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5523        use std::os::unix::fs::PermissionsExt;
5524        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5525        let script = dir.join("module.sh");
5526        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5527        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5528        let record_path = dir.join("live-children.json");
5529        let supervisor =
5530            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5531                .with_live_children_record(&record_path);
5532        for (program, args) in [
5533            (PathBuf::from("sleep"), vec!["60".into()]),
5534            (script, vec![]),
5535        ] {
5536            let spec = ModuleSpec {
5537                module_id: "image-identity".into(),
5538                program: program.clone(),
5539                args,
5540                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5541                    .into_iter()
5542                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5543                    .collect(),
5544                reserved: false,
5545                reserved_prefixes: vec![],
5546                protocol: ModuleProtocol::None,
5547                overlap: Default::default(),
5548            };
5549            let module = supervisor.spawn(spec).unwrap();
5550            #[cfg(target_os = "macos")]
5551            {
5552                // SETEXEC confirmation is asynchronous; the orphan record must
5553                // identify the final image, never the intermediate trampoline.
5554                let deadline = Instant::now() + Duration::from_secs(5);
5555                while crate::live_children::read_record(&record_path)
5556                    .unwrap()
5557                    .iter()
5558                    .all(|entry| entry.executable.is_none())
5559                {
5560                    assert!(Instant::now() < deadline, "module image was not confirmed");
5561                    tokio::time::sleep(Duration::from_millis(5)).await;
5562                }
5563            }
5564            let entry = crate::live_children::read_record(&record_path)
5565                .unwrap()
5566                .pop()
5567                .unwrap();
5568            let observed = subc_os::Process::open(entry.pid)
5569                .unwrap()
5570                .unwrap()
5571                .observe()
5572                .unwrap();
5573            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5574            module.drain().await.unwrap();
5575            assert_eq!(
5576                verdict,
5577                crate::live_children::IdentityVerdict::Matches,
5578                "program {program:?}: recorded {entry:?}, observed {observed:?}"
5579            );
5580        }
5581    }
5582
5583    #[cfg(unix)]
5584    fn http_fixture(
5585        dir: &std::path::Path,
5586        url: &str,
5587        threshold: u32,
5588    ) -> crate::daemon_config::ConfiguredModule {
5589        let path = dir.join("subc.jsonc");
5590        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5591            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5592            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5593            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5594            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5595        }}}).to_string()).unwrap();
5596        crate::daemon_config::load(&path)
5597            .unwrap()
5598            .unwrap()
5599            .modules
5600            .pop()
5601            .unwrap()
5602    }
5603
5604    #[cfg(unix)]
5605    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5606        timeout(Duration::from_secs(5), async {
5607            loop {
5608                if module.status().unwrap().health.status == status {
5609                    break;
5610                }
5611                sleep(Duration::from_millis(5)).await;
5612            }
5613        })
5614        .await
5615        .unwrap_or_else(|_| {
5616            panic!(
5617                "expected {status:?}, got {:?}",
5618                module.status().unwrap().health
5619            )
5620        });
5621    }
5622
5623    #[cfg(unix)]
5624    #[tokio::test]
5625    async fn http_health_status_flips_ok_failing_ok() {
5626        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5627        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5628        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5629        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5630        let serving_status = status.clone();
5631        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5632        let server = tokio::spawn(async move {
5633            loop {
5634                let (mut stream, _) = listener.accept().await.unwrap();
5635                let mut request = [0u8; 2048];
5636                let count = stream.read(&mut request).await.unwrap();
5637                assert!(count > 0, "a probe must send an HTTP request");
5638                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5639                let body = if code == 200 {
5640                    "ready"
5641                } else {
5642                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5643                };
5644                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5645                let _ = stream.write_all(response.as_bytes()).await;
5646            }
5647        });
5648        let configured = http_fixture(&dir, &url, 1000);
5649        let module =
5650            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5651                .supervise_configured_with_health(
5652                    configured.module_spec(),
5653                    true,
5654                    configured.health,
5655                    configured.drain_timeout_ms,
5656                    configured.restart,
5657                )
5658                .unwrap();
5659        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5660        status.store(503, std::sync::atomic::Ordering::SeqCst);
5661        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5662        assert!(module
5663            .status()
5664            .unwrap()
5665            .health
5666            .detail
5667            .unwrap()
5668            .contains("scratch failure"));
5669        status.store(200, std::sync::atomic::Ordering::SeqCst);
5670        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5671        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5672        let before = module.status().unwrap();
5673        let (spec, mut health) = module.configuration().unwrap();
5674        health.http = None;
5675        module
5676            .update_configuration(spec.clone(), health.clone(), Some(10))
5677            .await
5678            .unwrap();
5679        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5680        health.http = Some(url);
5681        module
5682            .update_configuration(spec, health, Some(10))
5683            .await
5684            .unwrap();
5685        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5686        assert_eq!(
5687            module.status().unwrap().pid,
5688            before.pid,
5689            "changing a probe must apply live, not restart its process"
5690        );
5691        let (spec, mut health) = module.configuration().unwrap();
5692        health.failure_threshold = 2;
5693        module
5694            .update_configuration(spec, health, Some(10))
5695            .await
5696            .unwrap();
5697        status.store(503, std::sync::atomic::Ordering::SeqCst);
5698        timeout(Duration::from_secs(5), async {
5699            while module.status().unwrap().spawn_generation == before.spawn_generation {
5700                sleep(Duration::from_millis(5)).await;
5701            }
5702        })
5703        .await
5704        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5705        module.drain().await.unwrap();
5706        server.abort();
5707    }
5708
5709    #[cfg(unix)]
5710    #[tokio::test]
5711    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5712        let dir = subc_test_support::TestTempDir::new("http-refused");
5713        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5714        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5715        drop(unused);
5716        let configured = http_fixture(&dir, &url, 2);
5717        let module =
5718            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5719                .supervise_configured_with_health(
5720                    configured.module_spec(),
5721                    true,
5722                    configured.health,
5723                    configured.drain_timeout_ms,
5724                    configured.restart,
5725                )
5726                .unwrap();
5727        let before = module.status().unwrap().spawn_generation;
5728        timeout(Duration::from_secs(5), async {
5729            loop {
5730                let status = module.status().unwrap();
5731                if status.spawn_generation > before {
5732                    assert!(status.lifetime_restarts > 0);
5733                    break;
5734                }
5735                sleep(Duration::from_millis(5)).await;
5736            }
5737        })
5738        .await
5739        .expect("sustained HTTP refusal must trigger the health restart policy");
5740        module.drain().await.unwrap();
5741    }
5742
5743    #[cfg(unix)]
5744    #[tokio::test]
5745    async fn http_health_timeout_honours_deadline() {
5746        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5747        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5748        let server = tokio::spawn(async move {
5749            let _held = listener.accept().await.unwrap();
5750            std::future::pending::<()>().await;
5751        });
5752        let error = timeout(
5753            Duration::from_secs(1),
5754            probe_http_health(&url, Duration::from_millis(10)),
5755        )
5756        .await
5757        .expect("the probe must enforce its own deadline")
5758        .unwrap_err();
5759        server.abort();
5760        assert!(error.to_string().contains("timed out"));
5761    }
5762
5763    #[cfg(unix)]
5764    #[tokio::test]
5765    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5766        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5767        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5768        let url = format!(
5769            "http://localhost:{}/healthz",
5770            listener.local_addr().unwrap().port()
5771        );
5772        let server = tokio::spawn(async move {
5773            let (mut stream, _) = listener.accept().await.unwrap();
5774            let mut request = [0u8; 2048];
5775            assert!(stream.read(&mut request).await.unwrap() > 0);
5776            stream
5777                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5778                .await
5779                .unwrap();
5780        });
5781        // The deadline only bounds a hang. A probe that never tried the IPv6
5782        // address would be refused on 127.0.0.1 and fail at once, so a longer
5783        // deadline does not weaken the assertion; one second timed out under a
5784        // loaded parallel test run.
5785        assert_eq!(
5786            probe_http_health(&url, Duration::from_secs(10))
5787                .await
5788                .unwrap()
5789                .status,
5790            HealthStatus::Ok
5791        );
5792        server.await.unwrap();
5793    }
5794
5795    #[cfg(unix)]
5796    #[tokio::test]
5797    async fn http_health_timeout_keeps_partial_status_and_body() {
5798        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5799        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5800        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5801        let server = tokio::spawn(async move {
5802            let (mut stream, _) = listener.accept().await.unwrap();
5803            let mut request = [0u8; 2048];
5804            assert!(stream.read(&mut request).await.unwrap() > 0);
5805            stream
5806                .write_all(
5807                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5808                )
5809                .await
5810                .unwrap();
5811            std::future::pending::<()>().await;
5812        });
5813        let error = probe_http_health(&url, Duration::from_secs(1))
5814            .await
5815            .unwrap_err()
5816            .to_string();
5817        server.abort();
5818        assert!(
5819            error.contains("timed out")
5820                && error.contains("503 Unavailable")
5821                && error.contains("partial diagnostic"),
5822            "{error}"
5823        );
5824    }
5825
5826    #[cfg(unix)]
5827    #[tokio::test]
5828    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5829        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5830        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5831        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5832        let server = tokio::spawn(async move {
5833            let (mut stream, _) = listener.accept().await.unwrap();
5834            let mut request = [0u8; 2048];
5835            assert!(stream.read(&mut request).await.unwrap() > 0);
5836            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5837            let response = format!(
5838                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5839                body.len()
5840            );
5841            stream.write_all(response.as_bytes()).await.unwrap();
5842        });
5843        let error = probe_http_health(&url, Duration::from_secs(1))
5844            .await
5845            .unwrap_err()
5846            .to_string();
5847        server.await.unwrap();
5848        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5849        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5850        assert!(!error.contains("not-in-diagnostic"));
5851    }
5852
5853    #[cfg(unix)]
5854    #[tokio::test]
5855    async fn http_health_real_nats_server_monitoring() {
5856        if std::process::Command::new("nats-server")
5857            .arg("--version")
5858            .env("XDG_DATA_HOME", std::env::temp_dir())
5859            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5860            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5861            .output()
5862            .is_err()
5863        {
5864            eprintln!(
5865                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5866            );
5867            return;
5868        }
5869        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5870        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5871        let port = monitor.local_addr().unwrap().port();
5872        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5873        let client_port = client.local_addr().unwrap().port();
5874        let config = dir.join("server.conf");
5875        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5876        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5877        configured.program = PathBuf::from("nats-server");
5878        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5879        drop(monitor);
5880        drop(client);
5881        let module =
5882            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5883                .supervise_configured_with_health(
5884                    configured.module_spec(),
5885                    true,
5886                    configured.health,
5887                    configured.drain_timeout_ms,
5888                    configured.restart,
5889                )
5890                .unwrap();
5891        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5892        module.drain().await.unwrap();
5893    }
5894
5895    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5896    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5897        let supervisor =
5898            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5899        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5900        let initial = ModuleSpec {
5901            module_id: "rescan-preserves-spawn-facts".to_string(),
5902            program: PathBuf::from("/spawned/module"),
5903            args: Vec::new(),
5904            env: Vec::new(),
5905            reserved: false,
5906            reserved_prefixes: Vec::new(),
5907            protocol: ModuleProtocol::Subc,
5908            overlap: Default::default(),
5909        };
5910        let module = supervisor.supervised_module(
5911            initial.clone(),
5912            supervisor.runtime_config(),
5913            snapshot,
5914            None,
5915        );
5916        let before = module.status().unwrap();
5917        let mut replacement = initial;
5918        replacement.program = PathBuf::from("/rescanned/replacement-module");
5919
5920        module
5921            .update_configuration(replacement, HealthConfig::default(), None)
5922            .await
5923            .unwrap();
5924
5925        let after = module.status().unwrap();
5926        assert_eq!(after.pid, before.pid);
5927        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5928        assert_eq!(after.spawned_from, before.spawned_from);
5929        drop(module);
5930    }
5931}
5932
5933fn unix_ms_now() -> u64 {
5934    SystemTime::now()
5935        .duration_since(UNIX_EPOCH)
5936        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5937        .unwrap_or(0)
5938}
5939
5940async fn supervise_loop(
5941    mut spec: ModuleSpec,
5942    mut runtime: SupervisorRuntimeConfig,
5943    registry: Arc<Registry>,
5944    process_liveness: Arc<SupervisorProcessLiveness>,
5945    snapshot: SharedSnapshot,
5946    mut child: Option<SupervisedChild>,
5947    mut commands: mpsc::Receiver<SupervisorCommand>,
5948) {
5949    let mut health_probe = HealthProbeRuntime::default();
5950    // All restart backoffs run here, including health and operator requests.
5951    // While one is pending the loop serves commands, so disable or drain can
5952    // cancel the replacement without spawning a process just to stop it.
5953    let mut pending_respawn: Option<PendingRespawn> = None;
5954    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5955    // before anything else so a stop that interrupted a swap runs at once.
5956    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5957    loop {
5958        #[cfg(target_os = "macos")]
5959        if let Some(active) = child.as_mut() {
5960            active.confirm_privacy_exec().await;
5961        }
5962        if let Some(scheduled) = runtime
5963            .scheduled_respawn
5964            .lock()
5965            .unwrap_or_else(|p| p.into_inner())
5966            .take()
5967        {
5968            pending_respawn = Some(scheduled);
5969        }
5970        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5971            pending_respawn = None;
5972            cancel_deferred_reload(
5973                &runtime,
5974                &spec.module_id,
5975                "respawn cancelled by a supervisor command",
5976            );
5977        }
5978        if child.is_none() && pending_respawn.is_none() {
5979            cancel_deferred_reload(
5980                &runtime,
5981                &spec.module_id,
5982                "respawn cancelled before a replacement was spawned",
5983            );
5984            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5985                state.respawn_pending = false;
5986                state.coalesced_restart_pending = false;
5987                if matches!(
5988                    state.state,
5989                    ModuleState::Restarting
5990                        | ModuleState::Starting
5991                        | ModuleState::Draining
5992                        | ModuleState::Unresponsive
5993                ) {
5994                    error!(module_id = %spec.module_id, state = ?state.state,
5995                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5996                    state.state = ModuleState::Failed;
5997                    clear_current_process_facts(state);
5998                }
5999            });
6000        }
6001        if let Some(command) = requeued.pop_front() {
6002            if !handle_supervisor_command(
6003                command,
6004                &mut spec,
6005                &mut runtime,
6006                &registry,
6007                &process_liveness,
6008                &snapshot,
6009                &mut child,
6010                &mut commands,
6011                &mut requeued,
6012            )
6013            .await
6014            {
6015                return;
6016            }
6017            if child.is_some() || !respawn_still_pending(&snapshot) {
6018                pending_respawn = None;
6019                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6020                    state.respawn_pending = false
6021                });
6022            }
6023            continue;
6024        }
6025        if child.is_some() {
6026            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
6027            let probe_sleep = sleep(health_probe.wake_after());
6028            tokio::pin!(probe_sleep);
6029            let active_child = child.as_mut().expect("child checked above");
6030            tokio::select! {
6031                wait_result = active_child.wait() => {
6032                    // Every arm below that gives up on the CHILD must keep the
6033                    // supervision task itself alive (child = None, loop
6034                    // continues into command-serving mode). Returning here
6035                    // closes the command channel, which makes the module
6036                    // permanently unrestartable in-band: a clean child exit
6037                    // of an enabled module once wedged the fleet this way
6038                    // ('supervisor command channel is closed') and required a
6039                    // full daemon restart to recover.
6040                    let exit_report = match wait_result {
6041                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6042                        Err(err) => {
6043                            active_child.drain_stderr(&spec.module_id).await;
6044                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6045                            // Every other exit path (on_child_exit's Clean/Crash arms,
6046                            // the reload-registration-failure path) records a terminal
6047                            // before moving on. Without one here, a module whose wait()
6048                            // itself errored (e.g. already reaped) leaves no terminal
6049                            // record at all -- an empty ring reads as "nothing died".
6050                            record_wait_error_terminal(
6051                                &spec.module_id,
6052                                &runtime.terminal_ring,
6053                                &runtime.spawn_events,
6054                            );
6055                            untrack_if_registration_released(
6056                                &process_liveness,
6057                                &registry,
6058                                &spec.module_id,
6059                                &snapshot,
6060                            );
6061                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6062                            child = None;
6063                            continue;
6064                        }
6065                    };
6066                    active_child.drain_stderr(&spec.module_id).await;
6067
6068                    let next = on_child_exit(
6069                        &spec,
6070                        runtime.restart_policy,
6071                        &registry,
6072                        &snapshot,
6073                        &runtime.terminal_ring,
6074                        &runtime.spawn_events,
6075                        &runtime.child_roster,
6076                        exit_report,
6077                    ).await;
6078                    // The exit is recorded, so a daemon shutdown may stop
6079                    // waiting for this child (see `SupervisedChild::wait`).
6080                    active_child.release_roster();
6081                    match next {
6082                        NextAction::Stop { registration_released } => {
6083                            if registration_released {
6084                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6085                            }
6086                            child = None;
6087                        }
6088                        NextAction::Restart { schedule } => {
6089                            let delay = schedule.map_or(
6090                                runtime.restart_policy.delay_for_restart(0),
6091                                |schedule| schedule.delay,
6092                            );
6093                            if let Some(schedule) = schedule {
6094                                log_crash_respawn(&spec.module_id, schedule);
6095                            }
6096                            // The exited child is fully recorded at this point,
6097                            // so release it and count the backoff down in the
6098                            // command-serving branch below rather than sleeping
6099                            // here: commands cannot be received from inside this
6100                            // select arm, and an operator disable or drain that
6101                            // arrives during the backoff must cancel the pending
6102                            // respawn instead of waiting for it to spawn first.
6103                            child = None;
6104                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6105                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6106                        }
6107                    }
6108                }
6109                command = commands.recv() => {
6110                    let Some(command) = command else {
6111                        return;
6112                    };
6113                    if !handle_supervisor_command(
6114                        command,
6115                        &mut spec,
6116                        &mut runtime,
6117                        &registry,
6118                        &process_liveness,
6119                        &snapshot,
6120                        &mut child,
6121                        &mut commands,
6122                        &mut requeued,
6123                    ).await {
6124                        return;
6125                    }
6126                }
6127                _ = &mut probe_sleep => {
6128                    if health_probe.due() {
6129                        run_health_probe_cycle(
6130                            &spec,
6131                            &runtime,
6132                            &registry,
6133                            &process_liveness,
6134                            &snapshot,
6135                            &mut child,
6136                        ).await;
6137                        if child.is_some() {
6138                            health_probe.schedule_next(&spec, runtime.health.cadence);
6139                        }
6140                    }
6141                }
6142            }
6143        } else if let Some(pending) = pending_respawn {
6144            tokio::select! {
6145                _ = sleep_until(pending.deadline) => {
6146                    pending_respawn = None;
6147                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6148                    // A command handled below while the backoff elapsed may
6149                    // have stopped the module; never respawn past an operator's
6150                    // disable or drain.
6151                    if !respawn_still_pending(&snapshot) {
6152                        continue;
6153                    }
6154                    // The daemon began shutting down during the backoff: the
6155                    // spawn would be refused anyway, and refusing it here
6156                    // leaves the module stopped instead of reporting a
6157                    // failed restart.
6158                    if runtime.child_roster.is_closed() {
6159                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6160                            state.state = ModuleState::Stopped;
6161                        });
6162                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6163                        continue;
6164                    }
6165                    if let Err(err) = release_dead_registration(
6166                        &registry,
6167                        runtime.forwarding.as_deref(),
6168                        &snapshot,
6169                        &spec.module_id,
6170                    ).await {
6171                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6172                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6173                        continue;
6174                    }
6175
6176                    if matches!(pending.kind, RespawnKind::Reload) {
6177                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6178                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6179                        if let Some(reply) = reply { let _ = reply.send(result); }
6180                        continue;
6181                    }
6182                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6183                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6184                        Ok(next_child) => {
6185                            child = Some(next_child);
6186                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6187                        }
6188                        Err(err) => {
6189                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6190                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6191                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6192                        }
6193                    }
6194                }
6195                command = commands.recv() => {
6196                    let Some(command) = command else {
6197                        return;
6198                    };
6199                    if !handle_supervisor_command(
6200                        command,
6201                        &mut spec,
6202                        &mut runtime,
6203                        &registry,
6204                        &process_liveness,
6205                        &snapshot,
6206                        &mut child,
6207                        &mut commands,
6208                        &mut requeued,
6209                    ).await {
6210                        return;
6211                    }
6212                    // Reconcile the pending respawn with what the command did:
6213                    // a start may already have spawned a fresh child,
6214                    // while a disable or drain moved the snapshot out of the
6215                    // state the respawn was counting down from.
6216                    if child.is_some() || !respawn_still_pending(&snapshot) {
6217                        pending_respawn = None;
6218                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6219                    }
6220                }
6221            }
6222        } else {
6223            let Some(command) = commands.recv().await else {
6224                return;
6225            };
6226            if !handle_supervisor_command(
6227                command,
6228                &mut spec,
6229                &mut runtime,
6230                &registry,
6231                &process_liveness,
6232                &snapshot,
6233                &mut child,
6234                &mut commands,
6235                &mut requeued,
6236            )
6237            .await
6238            {
6239                return;
6240            }
6241        }
6242    }
6243}
6244
6245fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6246    info!(
6247        module_id,
6248        restart_in_window = schedule.restart_in_window,
6249        delay_ms = schedule.delay.as_millis() as u64,
6250        "respawning after crash"
6251    );
6252}
6253
6254/// Whether the respawn a backoff was counting down to is still wanted. A
6255/// disable or drain handled while the backoff elapsed moves the snapshot out
6256/// of `Restarting`, and the operator's stop must win over the pending respawn,
6257/// so every sleep-then-spawn path re-validates against the live snapshot
6258/// instead of assuming the state it left behind still holds.
6259fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6260    matches!(
6261        lock_snapshot(snapshot),
6262        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6263    )
6264}
6265
6266enum NextAction {
6267    Stop {
6268        registration_released: bool,
6269    },
6270    Restart {
6271        schedule: Option<CrashRestartSchedule>,
6272    },
6273}
6274
6275#[allow(clippy::too_many_arguments)]
6276async fn handle_supervisor_command(
6277    command: SupervisorCommand,
6278    spec: &mut ModuleSpec,
6279    runtime: &mut SupervisorRuntimeConfig,
6280    registry: &Arc<Registry>,
6281    process_liveness: &SupervisorProcessLiveness,
6282    snapshot: &SharedSnapshot,
6283    child: &mut Option<SupervisedChild>,
6284    commands: &mut mpsc::Receiver<SupervisorCommand>,
6285    requeued: &mut VecDeque<SupervisorCommand>,
6286) -> bool {
6287    match command {
6288        SupervisorCommand::Drain { reply } => {
6289            // A plain stop runs no forwarding drain, so nothing reaches the
6290            // module over its connection before the wait: ask by signal.
6291            let result = drain_optional_child(
6292                &spec.module_id,
6293                spec.protocol,
6294                StopNotice::NotSent,
6295                registry,
6296                runtime.forwarding.as_deref(),
6297                snapshot,
6298                &runtime.terminal_ring,
6299                &runtime.spawn_events,
6300                child,
6301                runtime.drain_timeout,
6302                ModuleState::Stopped,
6303                None,
6304            )
6305            .await;
6306            let registration_released = result.is_ok();
6307            let _ = reply.send(result);
6308            if registration_released {
6309                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6310            }
6311            false
6312        }
6313        SupervisorCommand::Retire { reply } => {
6314            let result = async {
6315                let stop_notice = begin_forwarding_drain_if_configured(
6316                    spec,
6317                    runtime,
6318                    registry,
6319                    snapshot,
6320                    None,
6321                    RouteCloseReason::Disable,
6322                )
6323                .await?;
6324                drain_optional_child(
6325                    &spec.module_id,
6326                    spec.protocol,
6327                    stop_notice,
6328                    registry,
6329                    runtime.forwarding.as_deref(),
6330                    snapshot,
6331                    &runtime.terminal_ring,
6332                    &runtime.spawn_events,
6333                    child,
6334                    runtime.drain_timeout,
6335                    ModuleState::Stopped,
6336                    None,
6337                )
6338                .await
6339            }
6340            .await;
6341            let registration_released = result.is_ok();
6342            let _ = reply.send(result);
6343            if registration_released {
6344                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6345            }
6346            false
6347        }
6348        SupervisorCommand::Restart {
6349            drain_timeout_ms,
6350            received_at_generation,
6351            queued_at,
6352            reply,
6353        } => {
6354            // Without this line a restart that waited in the queue (behind a
6355            // health probe cycle or another command) was invisible: the log
6356            // showed only the drain timing out, minutes after the operator's call.
6357            info!(
6358                module_id = %spec.module_id,
6359                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6360                "restart command dequeued"
6361            );
6362            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6363            // caller whose own request lane rides the module being restarted: the
6364            // caller's in-flight request keeps the drain from quiescing, the drain
6365            // keeps the restart from completing, and the completion keeps the reply
6366            // from releasing the caller — so the drain always timed out and cut the
6367            // initiator with a GOODBYE, even on a healthy module. Replying once the
6368            // restart is validated lets a self-lane caller settle, which is exactly
6369            // what makes the drain succeed. Completion is observable via
6370            // supervisor.list / module status; a post-ack failure lands the module
6371            // in a visible terminal state below rather than in a reply nobody can
6372            // receive.
6373            let validation = match lock_snapshot(snapshot) {
6374                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6375                    module_id: spec.module_id.clone(),
6376                }),
6377                Ok(_) => Ok(()),
6378                Err(err) => Err(err),
6379            };
6380            let initiated = validation.is_ok();
6381            let _ = reply.send(validation);
6382            // A restart asks for a fresh process. Commands run one at a time,
6383            // so a restart queued behind another restart (two operator calls
6384            // in quick succession) is dequeued the moment the first one has
6385            // spawned its replacement -- before that process has sent HELLO.
6386            // Running it would drain and kill the process the first restart
6387            // just produced, which is the opposite of what both callers asked
6388            // for. If a process spawned after this request was received is
6389            // still supervised, the request is already satisfied. Not when the
6390            // configuration changed since that spawn: then the newer process
6391            // predates the spec this restart may exist to apply.
6392            let satisfied_by_generation = if initiated && child.is_some() {
6393                lock_snapshot(snapshot).ok().and_then(|state| {
6394                    (state.spawn_generation > received_at_generation
6395                        && !state.configuration_updated_since_spawn)
6396                        .then_some(state.spawn_generation)
6397                })
6398            } else {
6399                None
6400            };
6401            let satisfied_by_pending = initiated
6402                && child.is_none()
6403                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6404                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6405                    if pending {
6406                        state.coalesced_restart_pending = true;
6407                    }
6408                    pending
6409                });
6410            if satisfied_by_pending {
6411                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6412            } else if let Some(generation) = satisfied_by_generation {
6413                info!(
6414                    module_id = %spec.module_id,
6415                    received_at_generation,
6416                    "restart already satisfied by generation {generation}; not restarting again"
6417                );
6418            } else if initiated {
6419                // Precedence: this restart's operator override, else the module's
6420                // configured budget (already resolved into the runtime).
6421                let drain_timeout = drain_timeout_ms
6422                    .map(Duration::from_millis)
6423                    .unwrap_or(runtime.drain_timeout);
6424                if let Err(err) = restart_child(
6425                    spec,
6426                    runtime,
6427                    registry,
6428                    process_liveness,
6429                    snapshot,
6430                    child,
6431                    drain_timeout,
6432                )
6433                .await
6434                {
6435                    warn!(
6436                        module_id = %spec.module_id,
6437                        error = %err,
6438                        "operator restart failed after initiation ack; module state carries the outcome"
6439                    );
6440                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6441                        state.state = ModuleState::Failed;
6442                        clear_current_process_facts(state);
6443                    });
6444                }
6445            }
6446            true
6447        }
6448        SupervisorCommand::Reload { reply } => {
6449            let result =
6450                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6451            if result.is_ok()
6452                && runtime
6453                    .scheduled_respawn
6454                    .lock()
6455                    .unwrap_or_else(|p| p.into_inner())
6456                    .as_ref()
6457                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6458            {
6459                *runtime
6460                    .deferred_reload_reply
6461                    .lock()
6462                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6463            } else {
6464                let _ = reply.send(result);
6465            }
6466            true
6467        }
6468        SupervisorCommand::SetEnabled { enabled, reply } => {
6469            let result = set_child_enabled(
6470                spec,
6471                runtime,
6472                registry,
6473                process_liveness,
6474                snapshot,
6475                child,
6476                enabled,
6477            )
6478            .await;
6479            let _ = reply.send(result);
6480            true
6481        }
6482        SupervisorCommand::UpdateConfiguration {
6483            spec: next_spec,
6484            health,
6485            drain_timeout_ms,
6486            reply,
6487        } => {
6488            if let Some(handle) = &runtime.supervisor_handle {
6489                handle.apply_identity_configuration(&next_spec);
6490            }
6491            *spec = next_spec;
6492            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6493                state.configuration_updated_since_spawn = true;
6494            });
6495            let health_changed = runtime.health != health;
6496            runtime.health = health;
6497            // Reset the cadence and old endpoint's failure streak on a live
6498            // health-policy change rather than waiting for its old deadline.
6499            if health_changed {
6500                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6501                    state.health = ModuleHealthStatus::default();
6502                });
6503            }
6504            runtime.drain_timeout = drain_timeout_ms
6505                .map(Duration::from_millis)
6506                .unwrap_or(runtime.default_drain_timeout);
6507            *runtime
6508                .effective_drain_timeout
6509                .lock()
6510                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6511            let _ = reply.send(());
6512            true
6513        }
6514        SupervisorCommand::Swap {
6515            ready_timeout,
6516            reply,
6517        } => {
6518            let end = swap::run_swap(
6519                spec,
6520                runtime,
6521                registry,
6522                process_liveness,
6523                snapshot,
6524                child,
6525                commands,
6526                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6527                reply,
6528            )
6529            .await;
6530            requeued.extend(end.requeue);
6531            true
6532        }
6533    }
6534}
6535
6536async fn restart_child(
6537    spec: &ModuleSpec,
6538    runtime: &SupervisorRuntimeConfig,
6539    registry: &Registry,
6540    process_liveness: &SupervisorProcessLiveness,
6541    snapshot: &SharedSnapshot,
6542    child: &mut Option<SupervisedChild>,
6543    drain_timeout: Duration,
6544) -> Result<(), SuperviseError> {
6545    // Restart cycles a running module; it must not silently start a disabled one.
6546    if !lock_snapshot(snapshot)?.enabled {
6547        return Err(SuperviseError::Disabled {
6548            module_id: spec.module_id.clone(),
6549        });
6550    }
6551    let stop_notice = begin_forwarding_drain_with_timeout(
6552        spec,
6553        runtime,
6554        registry,
6555        snapshot,
6556        None,
6557        RouteCloseReason::Restart,
6558        drain_timeout,
6559    )
6560    .await?;
6561
6562    if child.is_some() {
6563        drain_optional_child(
6564            &spec.module_id,
6565            spec.protocol,
6566            stop_notice,
6567            registry,
6568            runtime.forwarding.as_deref(),
6569            snapshot,
6570            &runtime.terminal_ring,
6571            &runtime.spawn_events,
6572            child,
6573            drain_timeout,
6574            ModuleState::Restarting,
6575            Some(true),
6576        )
6577        .await?;
6578    } else {
6579        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6580            state.enabled = true;
6581            state.state = ModuleState::Restarting;
6582            clear_current_process_facts(state);
6583        })?;
6584        release_dead_registration(
6585            registry,
6586            runtime.forwarding.as_deref(),
6587            snapshot,
6588            &spec.module_id,
6589        )
6590        .await?;
6591    }
6592
6593    reset_restart_count(snapshot, &spec.module_id)?;
6594    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6595    schedule_respawn(
6596        runtime,
6597        snapshot,
6598        &spec.module_id,
6599        runtime.restart_policy.backoff,
6600        RespawnKind::Spawn,
6601    )
6602}
6603
6604async fn reload_child(
6605    spec: &ModuleSpec,
6606    runtime: &SupervisorRuntimeConfig,
6607    registry: &Registry,
6608    process_liveness: &SupervisorProcessLiveness,
6609    snapshot: &SharedSnapshot,
6610    child: &mut Option<SupervisedChild>,
6611) -> Result<(), SuperviseError> {
6612    // Reload cycles a running module; it must not silently start a disabled one.
6613    if !lock_snapshot(snapshot)?.enabled {
6614        return Err(SuperviseError::Disabled {
6615            module_id: spec.module_id.clone(),
6616        });
6617    }
6618    let stop_notice = begin_forwarding_drain(
6619        spec,
6620        runtime,
6621        registry,
6622        snapshot,
6623        Some(true),
6624        RouteCloseReason::Reload,
6625    )
6626    .await?;
6627
6628    if child.is_some() {
6629        drain_optional_child(
6630            &spec.module_id,
6631            spec.protocol,
6632            stop_notice,
6633            registry,
6634            runtime.forwarding.as_deref(),
6635            snapshot,
6636            &runtime.terminal_ring,
6637            &runtime.spawn_events,
6638            child,
6639            runtime.drain_timeout,
6640            ModuleState::Restarting,
6641            Some(true),
6642        )
6643        .await?;
6644    } else {
6645        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6646            state.enabled = true;
6647            state.state = ModuleState::Restarting;
6648            clear_current_process_facts(state);
6649        })?;
6650        release_dead_registration(
6651            registry,
6652            runtime.forwarding.as_deref(),
6653            snapshot,
6654            &spec.module_id,
6655        )
6656        .await?;
6657    }
6658
6659    reset_restart_count(snapshot, &spec.module_id)?;
6660    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6661    schedule_respawn(
6662        runtime,
6663        snapshot,
6664        &spec.module_id,
6665        runtime.restart_policy.backoff,
6666        RespawnKind::Reload,
6667    )
6668}
6669
6670async fn finish_reload_child(
6671    spec: &ModuleSpec,
6672    runtime: &SupervisorRuntimeConfig,
6673    registry: &Registry,
6674    process_liveness: &SupervisorProcessLiveness,
6675    snapshot: &SharedSnapshot,
6676    child: &mut Option<SupervisedChild>,
6677) -> Result<(), SuperviseError> {
6678    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6679    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6680        Ok(next_child) => next_child,
6681        Err(err) => {
6682            return handle_reload_spawn_failure(
6683                spec,
6684                runtime,
6685                process_liveness,
6686                snapshot,
6687                child,
6688                format!("new child failed to spawn: {err}"),
6689            )
6690            .await;
6691        }
6692    };
6693    *child = Some(next_child);
6694
6695    let wait_outcome = {
6696        let active_child = child.as_mut().expect("new reload child was just stored");
6697        wait_for_registration_after_reload(
6698            registry,
6699            &spec.module_id,
6700            snapshot,
6701            active_child,
6702            REGISTRY_RELEASE_TIMEOUT,
6703        )
6704        .await?
6705    };
6706
6707    match wait_outcome {
6708        RegistrationWaitOutcome::Registered => {
6709            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6710            Ok(())
6711        }
6712        RegistrationWaitOutcome::Exited(exit_report) => {
6713            if let Some(active_child) = child.as_mut() {
6714                active_child.drain_stderr(&spec.module_id).await;
6715            }
6716            // Keep the reaped child's roster guard until its terminal is written.
6717            // Shutdown waits on that guard, not on the child Option used for respawn.
6718            let mut exited_child = child.take().expect("exited reload child is still stored");
6719            #[cfg(test)]
6720            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6721                gate.reached.notify_one();
6722                gate.resume.notified().await;
6723            }
6724            let result = handle_reload_child_registration_failure(
6725                spec,
6726                runtime,
6727                registry,
6728                process_liveness,
6729                snapshot,
6730                child,
6731                ReloadRegistrationFailure {
6732                    exit_report: registration_failure_exit_report(exit_report),
6733                    reason: exited_child
6734                        .spawn_failure
6735                        .clone()
6736                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6737                },
6738            )
6739            .await;
6740            exited_child.release_roster();
6741            result
6742        }
6743        RegistrationWaitOutcome::TimedOut => {
6744            let mut timed_out_child = child
6745                .take()
6746                .expect("timed-out reload child is still running");
6747            timed_out_child
6748                .start_kill()
6749                .map_err(|source| SuperviseError::Kill {
6750                    module_id: spec.module_id.clone(),
6751                    source,
6752                })?;
6753            let status = timed_out_child
6754                .wait()
6755                .await
6756                .map_err(|source| SuperviseError::Wait {
6757                    module_id: spec.module_id.clone(),
6758                    source,
6759                })?;
6760            timed_out_child.drain_stderr(&spec.module_id).await;
6761            handle_reload_child_registration_failure(
6762                spec,
6763                runtime,
6764                registry,
6765                process_liveness,
6766                snapshot,
6767                child,
6768                ReloadRegistrationFailure {
6769                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6770                        snapshot,
6771                        &timed_out_child,
6772                        &status,
6773                    )),
6774                    reason: format!(
6775                        "new child did not register within {:?}",
6776                        REGISTRY_RELEASE_TIMEOUT
6777                    ),
6778                },
6779            )
6780            .await
6781        }
6782    }
6783}
6784
6785async fn set_child_enabled(
6786    spec: &ModuleSpec,
6787    runtime: &SupervisorRuntimeConfig,
6788    registry: &Registry,
6789    process_liveness: &SupervisorProcessLiveness,
6790    snapshot: &SharedSnapshot,
6791    child: &mut Option<SupervisedChild>,
6792    enabled: bool,
6793) -> Result<bool, SuperviseError> {
6794    let (current_enabled, current_state, respawn_pending) = {
6795        let state = lock_snapshot(snapshot)?;
6796        (state.enabled, state.state, state.respawn_pending)
6797    };
6798    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6799    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6800    // clean (Stopped) has no live process and no other in-band recovery — the
6801    // operator's start is the explicit recovery act and resets the budget. Without
6802    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6803    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6804    // the one providing every agent's shell.
6805    let revive_terminal = enabled
6806        && current_enabled
6807        && child.is_none()
6808        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6809            || (current_state == ModuleState::Restarting && !respawn_pending));
6810    if current_enabled == enabled && !revive_terminal {
6811        return Ok(false);
6812    }
6813
6814    if enabled {
6815        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6816            state.enabled = true;
6817            state.state = ModuleState::Starting;
6818            clear_current_process_facts(state);
6819        })?;
6820        #[cfg(test)]
6821        if runtime.test_seed_stale_facts_before_enable_spawn {
6822            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6823                state.process_alive = true;
6824                state.pid = Some(41);
6825                state.spawned_at_ms = Some(42);
6826                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6827                state.spawned_file_identity = Some(SpawnedFileIdentity {
6828                    device: 43,
6829                    inode: 44,
6830                });
6831            })?;
6832        }
6833        release_dead_registration(
6834            registry,
6835            runtime.forwarding.as_deref(),
6836            snapshot,
6837            &spec.module_id,
6838        )
6839        .await?;
6840        reset_restart_count(snapshot, &spec.module_id)?;
6841        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6842        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6843            Ok(next_child) => next_child,
6844            Err(err) => {
6845                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6846                    state.state = ModuleState::Failed;
6847                    clear_current_process_facts(state);
6848                }) {
6849                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6850                }
6851                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6852                return Err(err);
6853            }
6854        };
6855        *child = Some(next_child);
6856        debug!(module_id = %spec.module_id, "supervised module enabled");
6857        Ok(true)
6858    } else {
6859        let stop_notice = begin_forwarding_drain_if_configured(
6860            spec,
6861            runtime,
6862            registry,
6863            snapshot,
6864            Some(false),
6865            RouteCloseReason::Disable,
6866        )
6867        .await?;
6868        drain_optional_child(
6869            &spec.module_id,
6870            spec.protocol,
6871            stop_notice,
6872            registry,
6873            runtime.forwarding.as_deref(),
6874            snapshot,
6875            &runtime.terminal_ring,
6876            &runtime.spawn_events,
6877            child,
6878            runtime.drain_timeout,
6879            ModuleState::Disabled,
6880            Some(false),
6881        )
6882        .await?;
6883        debug!(module_id = %spec.module_id, "supervised module disabled");
6884        Ok(true)
6885    }
6886}
6887
6888#[allow(clippy::too_many_arguments)]
6889async fn on_child_exit(
6890    spec: &ModuleSpec,
6891    policy: RestartPolicy,
6892    registry: &Registry,
6893    snapshot: &SharedSnapshot,
6894    terminal_ring: &Arc<Mutex<TerminalRing>>,
6895    spawn_events: &SpawnEventFeed,
6896    roster: &ChildRoster,
6897    exit_report: ExitReport,
6898) -> NextAction {
6899    // Once the daemon has begun shutting down, no exit is a crash to recover
6900    // from: the module is exiting because the daemon is going away (EOF on its
6901    // connection, or a service manager signalling the whole cgroup). Record it
6902    // as such and never schedule a respawn, which would only start a process
6903    // for the shutdown to end again.
6904    if roster.is_closed() {
6905        return on_child_exit_during_daemon_shutdown(
6906            spec,
6907            registry,
6908            snapshot,
6909            terminal_ring,
6910            spawn_events,
6911            exit_report,
6912        )
6913        .await;
6914    }
6915    // Every stop the supervisor itself asks for (operator stop, disable,
6916    // restart, reload, swap, a health restart, a drain that runs out of budget)
6917    // takes the child out of the supervise loop and reaps it in
6918    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
6919    // that reaches this point was not requested by the daemon.
6920    //
6921    // For a subc-wire module a clean exit is still a stop: those modules are
6922    // written to re-raise SIGTERM, so a stray outside signal already reads as a
6923    // crash, and exiting 0 is a deliberate choice the module made. A
6924    // `protocol: "none"` module is a stock program we cannot change, and many
6925    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
6926    // stop would leave the module down for good after any stray signal, so it
6927    // goes through the crash path instead: it spends restart budget, respawns
6928    // with the crash backoff, and ends `failed` when the budget runs out.
6929    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6930        && running_protocol(spec, snapshot) == ModuleProtocol::None;
6931    match exit_report.kind {
6932        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6933            info!(
6934                module_id = %spec.module_id,
6935                exit_code = ?exit_report.code,
6936                exit_signal = ?exit_report.signal,
6937                "supervised module exited cleanly"
6938            );
6939            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6940                state.state = ModuleState::Stopped;
6941                clear_current_process_facts(state);
6942                state.last_exit = Some(exit_report.clone());
6943            }) {
6944                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6945            }
6946            record_terminal(
6947                &spec.module_id,
6948                terminal_ring,
6949                spawn_events,
6950                &exit_report,
6951                TerminalDisposition::Stopped,
6952            );
6953            let registration_released = match wait_for_registration_release(
6954                registry,
6955                &spec.module_id,
6956                REGISTRY_RELEASE_TIMEOUT,
6957            )
6958            .await
6959            {
6960                Ok(()) => true,
6961                Err(err) => {
6962                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6963                    false
6964                }
6965            };
6966            NextAction::Stop {
6967                registration_released,
6968            }
6969        }
6970        ExitKind::Clean | ExitKind::Crash => {
6971            if unrequested_clean_exit_of_protocol_none {
6972                warn!(
6973                    module_id = %spec.module_id,
6974                    exit_code = ?exit_report.code,
6975                    exit_signal = ?exit_report.signal,
6976                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
6977                );
6978            } else {
6979                warn!(
6980                    module_id = %spec.module_id,
6981                    exit_code = ?exit_report.code,
6982                    exit_signal = ?exit_report.signal,
6983                    "supervised module exited abnormally (crash)"
6984                );
6985            }
6986            let mut restart_schedule = None;
6987            let mut disposition = TerminalDisposition::Disabled;
6988            // Set only when the budget is what stopped the module, so the
6989            // terminal record says which limit was hit rather than leaving
6990            // `failed` to be read as "crashed once, badly".
6991            let mut disposition_detail = lock_snapshot(snapshot)
6992                .ok()
6993                .and_then(|mut state| state.spawn_failure.take());
6994            let now = Instant::now();
6995            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6996                clear_current_process_facts(state);
6997                state.last_exit = Some(exit_report.clone());
6998                if state.enabled {
6999                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
7000                        state.state = ModuleState::Restarting;
7001                        restart_schedule = Some(schedule);
7002                        disposition = TerminalDisposition::Restarting;
7003                    } else {
7004                        disposition = TerminalDisposition::Failed;
7005                        let budget = policy.budget_exhausted_detail();
7006                        disposition_detail =
7007                            Some(disposition_detail.take().map_or_else(
7008                                || budget.clone(),
7009                                |cause| format!("{cause}; {budget}"),
7010                            ));
7011                    }
7012                } else {
7013                    state.state = ModuleState::Disabled;
7014                    disposition = TerminalDisposition::Disabled;
7015                }
7016            }) {
7017                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7018                return NextAction::Stop {
7019                    registration_released: false,
7020                };
7021            }
7022            if disposition == TerminalDisposition::Failed {
7023                // The window is in the message, not only in the fields: this line
7024                // is read in a scrollback where a bare `max_restarts=3` reads as a
7025                // lifetime cap and sends the operator looking for three crashes
7026                // that never happened together.
7027                error!(
7028                    module_id = %spec.module_id,
7029                    max_restarts = policy.max_restarts,
7030                    window_secs = policy.window.as_secs(),
7031                    "module stopped: {}",
7032                    policy.budget_exhausted_detail()
7033                );
7034            }
7035            let budget_exhausted = disposition == TerminalDisposition::Failed;
7036            let record_exit = || {
7037                record_terminal_with_detail(
7038                    &spec.module_id,
7039                    terminal_ring,
7040                    spawn_events,
7041                    &exit_report,
7042                    disposition,
7043                    disposition_detail,
7044                );
7045            };
7046            if budget_exhausted {
7047                // Publish Failed only after its terminal record is available.
7048                // Recording takes the event-feed lock, then ring -> journal
7049                // writer (with file I/O), all without the hot snapshot lock.
7050                // No lock is held when the final snapshot update runs, nor
7051                // across the registration-release await below. Commands and
7052                // health actions run on this same supervisor task, so none can
7053                // act on the old state during the write; process facts already
7054                // say the child is dead to concurrent liveness readers.
7055                // A journal error is retained in history, not returned. If
7056                // recording panics, still publish Failed before resuming the
7057                // original unwind rather than leaving a dead child Running.
7058                let recorded = std::panic::catch_unwind(std::panic::AssertUnwindSafe(record_exit));
7059                if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7060                    state.state = ModuleState::Failed;
7061                }) {
7062                    error!(module_id = %spec.module_id, error = %err, "failed to publish exhausted restart budget");
7063                }
7064                if let Err(panic) = recorded {
7065                    std::panic::resume_unwind(panic);
7066                }
7067            } else {
7068                record_exit();
7069            }
7070
7071            if let Some(schedule) = restart_schedule {
7072                NextAction::Restart {
7073                    schedule: Some(schedule),
7074                }
7075            } else {
7076                let registration_released = match wait_for_registration_release(
7077                    registry,
7078                    &spec.module_id,
7079                    REGISTRY_RELEASE_TIMEOUT,
7080                )
7081                .await
7082                {
7083                    Ok(()) => true,
7084                    Err(err) => {
7085                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7086                        false
7087                    }
7088                };
7089                NextAction::Stop {
7090                    registration_released,
7091                }
7092            }
7093        }
7094        ExitKind::DeliberateSeverance => {
7095            warn!(
7096                module_id = %spec.module_id,
7097                exit_code = ?exit_report.code,
7098                exit_signal = ?exit_report.signal,
7099                "supervised module exited after deliberate connection severance"
7100            );
7101            let mut should_restart = false;
7102            let mut disposition = TerminalDisposition::Disabled;
7103            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7104                clear_current_process_facts(state);
7105                state.last_exit = Some(exit_report.clone());
7106                state.lifetime_restarts += 1;
7107                if state.enabled {
7108                    state.state = ModuleState::Restarting;
7109                    should_restart = true;
7110                    disposition = TerminalDisposition::Restarting;
7111                } else {
7112                    state.state = ModuleState::Disabled;
7113                }
7114            }) {
7115                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7116                return NextAction::Stop {
7117                    registration_released: false,
7118                };
7119            }
7120            record_terminal(
7121                &spec.module_id,
7122                terminal_ring,
7123                spawn_events,
7124                &exit_report,
7125                disposition,
7126            );
7127
7128            if should_restart {
7129                NextAction::Restart { schedule: None }
7130            } else {
7131                let registration_released = match wait_for_registration_release(
7132                    registry,
7133                    &spec.module_id,
7134                    REGISTRY_RELEASE_TIMEOUT,
7135                )
7136                .await
7137                {
7138                    Ok(()) => true,
7139                    Err(err) => {
7140                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7141                        false
7142                    }
7143                };
7144                NextAction::Stop {
7145                    registration_released,
7146                }
7147            }
7148        }
7149    }
7150}
7151
7152async fn on_child_exit_during_daemon_shutdown(
7153    spec: &ModuleSpec,
7154    registry: &Registry,
7155    snapshot: &SharedSnapshot,
7156    terminal_ring: &Arc<Mutex<TerminalRing>>,
7157    spawn_events: &SpawnEventFeed,
7158    exit_report: ExitReport,
7159) -> NextAction {
7160    info!(
7161        module_id = %spec.module_id,
7162        exit_code = ?exit_report.code,
7163        exit_signal = ?exit_report.signal,
7164        exit_kind = ?exit_report.kind,
7165        "supervised module exited during daemon shutdown; not restarting it"
7166    );
7167    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7168        state.state = ModuleState::Stopped;
7169        clear_current_process_facts(state);
7170        state.last_exit = Some(exit_report.clone());
7171    }) {
7172        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7173    }
7174    record_terminal(
7175        &spec.module_id,
7176        terminal_ring,
7177        spawn_events,
7178        &exit_report,
7179        TerminalDisposition::DaemonShutdown,
7180    );
7181    let registration_released =
7182        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7183            .await
7184            .is_ok();
7185    NextAction::Stop {
7186        registration_released,
7187    }
7188}
7189
7190fn record_wait_error_terminal(
7191    module_id: &str,
7192    terminal_ring: &Arc<Mutex<TerminalRing>>,
7193    spawn_events: &SpawnEventFeed,
7194) {
7195    record_terminal(
7196        module_id,
7197        terminal_ring,
7198        spawn_events,
7199        &wait_error_exit_report(),
7200        TerminalDisposition::Failed,
7201    );
7202}
7203
7204fn record_terminal(
7205    module_id: &str,
7206    terminal_ring: &Arc<Mutex<TerminalRing>>,
7207    spawn_events: &SpawnEventFeed,
7208    exit_report: &ExitReport,
7209    disposition: TerminalDisposition,
7210) {
7211    record_terminal_with_detail(
7212        module_id,
7213        terminal_ring,
7214        spawn_events,
7215        exit_report,
7216        disposition,
7217        None,
7218    );
7219}
7220
7221/// The ring lock is held only to capture the read (see
7222/// `TerminalJournal::capture_read`), so this module's exits keep recording
7223/// while the journal files are read. Blocking: it reads files.
7224fn durable_terminal_history_of(
7225    terminal_ring: &Mutex<TerminalRing>,
7226    module_id: &str,
7227) -> subc_control::TerminalHistory {
7228    let read = terminal_ring
7229        .lock()
7230        .unwrap_or_else(|p| p.into_inner())
7231        .capture_durable_history();
7232    read.read(module_id)
7233}
7234
7235fn record_terminal_with_detail(
7236    module_id: &str,
7237    terminal_ring: &Arc<Mutex<TerminalRing>>,
7238    spawn_events: &SpawnEventFeed,
7239    exit_report: &ExitReport,
7240    disposition: TerminalDisposition,
7241    disposition_detail: Option<String>,
7242) {
7243    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7244    let record = TerminalRecord {
7245        exit_code: exit_report.code,
7246        exit_signal: exit_report.signal,
7247        at_ms: exit_report.at_ms,
7248        disposition,
7249        exit_kind: exit_report.kind.into(),
7250        disposition_detail,
7251    };
7252    terminal_ring
7253        .lock()
7254        .unwrap_or_else(|poisoned| poisoned.into_inner())
7255        .record_exit(module_id, record);
7256}
7257
7258fn untrack_if_registration_released(
7259    process_liveness: &SupervisorProcessLiveness,
7260    registry: &Registry,
7261    module_id: &str,
7262    snapshot: &SharedSnapshot,
7263) {
7264    match registry.get_module(module_id) {
7265        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7266        Ok(Some(_)) => {}
7267        Err(err) => {
7268            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7269        }
7270    }
7271}
7272
7273/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7274/// then apply the module's configured entries minus daemon-private capture keys.
7275///
7276/// Separated from `spawn_child` only so it can be asserted without spawning a
7277/// process — a duplicate of this logic in a test would pass while the real one
7278/// drifted, which is the defect class this function exists to avoid.
7279/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7280/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7281/// either and the argument would stop a stock binary from starting at all.
7282/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7283///
7284/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7285/// through [`apply_wire_spawn_args_for_role`].
7286#[cfg(test)]
7287fn apply_wire_spawn_args(
7288    command: &mut Command,
7289    spec: &ModuleSpec,
7290    connection_file_path: Option<&std::path::Path>,
7291    handle: Option<&SupervisorHandle>,
7292) -> Result<Option<NonceHandoff>, SuperviseError> {
7293    apply_wire_spawn_args_for_role(
7294        command,
7295        spec,
7296        connection_file_path,
7297        handle,
7298        SpawnRole::Plain,
7299    )
7300}
7301
7302/// The read end of a spawn's launch-nonce pipe, prepared by
7303/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7304/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7305/// handoff and keeps only the environment copy.
7306#[cfg(unix)]
7307type NonceHandoff = subc_os::LaunchNonceHandoff;
7308#[cfg(not(unix))]
7309type NonceHandoff = std::convert::Infallible;
7310
7311/// Prepare wire identity for a plain spawn or a swap candidate.
7312///
7313/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7314/// a separate candidate token so the still-serving incumbent and its consumers
7315/// keep their nonce. Both records are installed before the process exists, so
7316/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7317///
7318/// On Unix the nonce is delivered only through a pipe. It is written into
7319/// a pipe whose read end the child gets as descriptor 3, named by
7320/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7321/// process of the same user cannot read it with `ps eww`. That handoff is
7322/// returned rather than installed here, because installing it replaces
7323/// whatever the child has at descriptor 3 and so must be the last pre-exec
7324/// step, after the Linux cgroup placement that the caller registers later.
7325/// Windows retains the environment handoff until restricted handle inheritance
7326/// can be implemented outside std's process primitives.
7327fn apply_wire_spawn_args_for_role(
7328    command: &mut Command,
7329    spec: &ModuleSpec,
7330    connection_file_path: Option<&std::path::Path>,
7331    handle: Option<&SupervisorHandle>,
7332    role: SpawnRole,
7333) -> Result<Option<NonceHandoff>, SuperviseError> {
7334    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7335    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7336    // included: a daemon started from a module's process tree inherits it,
7337    // and passing it on would point the child at a descriptor it does not
7338    // have.
7339    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7340    // Remove inherited or configured copies too: withholding must mean absent.
7341    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7342    if spec.protocol == ModuleProtocol::None {
7343        return Ok(None);
7344    }
7345    if let Some(connection_file_path) = connection_file_path {
7346        command.arg(SUBC_ARG).arg(connection_file_path);
7347    }
7348
7349    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7350    // route.open attestation. Reserved modules additionally use the same nonce
7351    // for HELLO id-squatting protection. A respawn rotates both records.
7352    let nonce = generate_launch_nonce()?;
7353    if let Some(handle) = handle {
7354        match role {
7355            SpawnRole::Plain => {
7356                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7357                if spec.reserved {
7358                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7359                }
7360            }
7361            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7362        }
7363    }
7364    #[cfg(unix)]
7365    let handoff = {
7366        let handoff =
7367            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7368                program: spec.program.clone(),
7369                source,
7370                cgroup_path: None,
7371            })?;
7372        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7373        Some(handoff)
7374    };
7375    #[cfg(not(unix))]
7376    let handoff = None;
7377    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7378    // handle to this child without leaking it to concurrently spawned processes.
7379    #[cfg(not(unix))]
7380    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7381    Ok(handoff)
7382}
7383
7384fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7385    command.env_remove(CK_LOG_ENV);
7386    // The spawn role is the supervisor's to set, and only on a swap candidate
7387    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7388    // it, is what makes it absent on a plain spawn: the daemon's own
7389    // environment could carry it, and so could a spec built outside daemon
7390    // config (config refuses it as an `env` key). A module reading it on a
7391    // plain restart would pick the long swap budget and leave callers waiting.
7392    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7393    for (key, value) in &spec.env {
7394        // cortexkit-log currently exposes retention only as a Rust struct, not
7395        // environment names. These values are daemon-private sink metadata and
7396        // must never become a public child-process contract by being inherited.
7397        if matches!(
7398            key.as_str(),
7399            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7400        ) || key == SUBC_SPAWN_ROLE_ENV
7401        {
7402            continue;
7403        }
7404        command.env(key, value);
7405    }
7406}
7407
7408/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7409/// of a blue/green swap.
7410#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7411enum SpawnRole {
7412    Plain,
7413    SwapCandidate,
7414}
7415
7416/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7417/// `apply_child_env` has already removed the variable for every spawn.
7418fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7419    if role == SpawnRole::SwapCandidate {
7420        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7421    }
7422}
7423
7424fn spawn_child(
7425    spec: &ModuleSpec,
7426    connection_file_path: Option<&std::path::Path>,
7427    handle: Option<&SupervisorHandle>,
7428    ring: &Arc<Mutex<StderrRing>>,
7429    capture_logs_dir: Option<&std::path::Path>,
7430    roster: &ChildRoster,
7431    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7432) -> Result<SupervisedChild, SuperviseError> {
7433    spawn_child_in_slot(
7434        spec,
7435        connection_file_path,
7436        handle,
7437        ring,
7438        capture_logs_dir,
7439        roster,
7440        #[cfg(target_os = "linux")]
7441        cgroup_placement,
7442        SpawnRole::Plain,
7443        false,
7444    )
7445}
7446
7447/// Spawn one process of `spec` into a slot.
7448///
7449/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7450/// A swap candidate needs a different cgroup from the process it is replacing,
7451/// which is still alive: in the same cgroup the two would be one kill domain,
7452/// and killing a failed candidate could take the incumbent with it.
7453///
7454/// The stderr capture file is `<module_id>.stderr.log` for every process of
7455/// the module, whichever slot it is in, because that is the one file
7456/// `ck module logs` reads. During a swap's overlap both processes append to it;
7457/// the daemon writes whole lines, so the two interleave by line, which is also
7458/// the merged view an operator wants while a swap runs.
7459#[allow(clippy::too_many_arguments)]
7460fn spawn_child_in_slot(
7461    spec: &ModuleSpec,
7462    connection_file_path: Option<&std::path::Path>,
7463    handle: Option<&SupervisorHandle>,
7464    ring: &Arc<Mutex<StderrRing>>,
7465    capture_logs_dir: Option<&std::path::Path>,
7466    roster: &ChildRoster,
7467    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7468    role: SpawnRole,
7469    alternate_slot: bool,
7470) -> Result<SupervisedChild, SuperviseError> {
7471    if roster.is_closed() {
7472        return Err(SuperviseError::Spawn {
7473            program: spec.program.clone(),
7474            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7475            cgroup_path: None,
7476        });
7477    }
7478    #[cfg(target_os = "linux")]
7479    let cgroup_name = {
7480        // Slot names alone are not kill domains: a retired incumbent may still
7481        // be draining when a later enable/restart spawns into the same slot.
7482        // Decimal entropy keeps the suffix unambiguous; Placement performs
7483        // the module-id escaping and constructs the filesystem path.
7484        if cgroup_placement.is_none() {
7485            swap::cgroup_name(&spec.module_id, alternate_slot)
7486        } else {
7487            let nonce = generate_launch_nonce()?;
7488            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7489            // Leave room for byte escaping and the suffix under NAME_MAX. The
7490            // label is only for humans; the nonce identifies the kill domain.
7491            let mut end = spec.module_id.len().min(64);
7492            while !spec.module_id.is_char_boundary(end) {
7493                end -= 1;
7494            }
7495            format!(
7496                "{}_{suffix}",
7497                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7498            )
7499        }
7500    };
7501    #[cfg(not(target_os = "linux"))]
7502    let _ = alternate_slot;
7503    #[cfg(target_os = "macos")]
7504    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7505    #[cfg(not(target_os = "macos"))]
7506    let mut command = Command::new(&spec.program);
7507    command.args(&spec.args);
7508    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7509    // that is the whole of the intent, so remove that one key rather than the
7510    // environment.
7511    //
7512    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7513    // and took the POSIX environment with it. Modules spawned that way had no
7514    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7515    // logging:
7516    //
7517    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7518    //     both unset it fell back to the temp dir alone and `ck` could not find
7519    //     a daemon running on the same machine from inside any module's process
7520    //     tree — reporting a path the file has never lived at, which reads as
7521    //     "the daemon did not write its file".
7522    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7523    //     the RELATIVE `.local/share`, so a module deriving its own store path
7524    //     resolved it against its own CWD. That is the store-fragmentation
7525    //     defect the daemon already refuses in config (`parse_doc` rejects a
7526    //     relative `storage.data_home`) arriving by derivation instead.
7527    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7528    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7529    //     quietly rather than erroring.
7530    //
7531    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7532    // offered one candidate under /tmp while the file sat in /run/user/1000.
7533    //
7534    // A configured module is unaffected either way: `module_spec()` puts the
7535    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7536    // wins over anything ambient.
7537    apply_child_env(&mut command, spec);
7538    apply_spawn_role(&mut command, role);
7539    let nonce_handoff =
7540        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7541
7542    #[cfg(target_os = "linux")]
7543    let cgroup_path = cgroup_placement
7544        .map(|placement| placement.module_path(&cgroup_name))
7545        .transpose()
7546        .map_err(|source| SuperviseError::Cgroup {
7547            module_id: spec.module_id.clone(),
7548            source,
7549        })?;
7550    #[cfg(not(target_os = "linux"))]
7551    let cgroup_path: Option<PathBuf> = None;
7552    #[cfg(target_os = "linux")]
7553    if let Some(path) = &cgroup_path {
7554        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7555            if let Some(placement) = cgroup_placement {
7556                remove_module_cgroup(placement, &cgroup_name);
7557            }
7558            return Err(error);
7559        }
7560    }
7561
7562    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7563        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7564        match ChildOutputSink::open(&path, capture_retention(spec)) {
7565            Ok(sink) => sink,
7566            Err(error) => {
7567                warn!(
7568                    module_id = %spec.module_id,
7569                    path = %path.display(),
7570                    error = %error,
7571                    "could not open child output capture file; forwarding to stderr"
7572                );
7573                ChildOutputSink::Stderr
7574            }
7575        }
7576    } else {
7577        ChildOutputSink::Stderr
7578    };
7579
7580    command.stdout(Stdio::piped());
7581    command.stderr(Stdio::piped());
7582    command.kill_on_drop(true);
7583    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7584    // before exec). In the daemon's group, a service manager that kills the
7585    // job's process group when the daemon exits (launchd's default) killed
7586    // every module at the same moment its control connection closed, so no
7587    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7588    // module is reached only by the daemon: the EOF it sees when its
7589    // connection closes, and the bounded stop in `child_roster` for anything
7590    // still running after that. On Linux this composes with the cgroup
7591    // placement above: that is a pre_exec write to cgroup.procs, std performs
7592    // setpgid in the child before running pre_exec callbacks, and the two
7593    // change independent process attributes.
7594    //
7595    // stdin is /dev/null because a process outside the terminal's foreground
7596    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7597    // by hand would otherwise hand down. Under a service manager stdin is
7598    // already /dev/null.
7599    #[cfg(unix)]
7600    command.process_group(0);
7601    command.stdin(Stdio::null());
7602    // The LAST pre-exec step, after the cgroup placement above: installing the
7603    // nonce at descriptor 3 replaces whatever the child had there, which could
7604    // be the descriptor an earlier step writes through.
7605    #[cfg(unix)]
7606    if let Some(handoff) = nonce_handoff {
7607        handoff.install_last(command.as_std_mut());
7608    }
7609    #[cfg(not(unix))]
7610    let _ = nonce_handoff;
7611
7612    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7613    // cannot run a single instruction -- and therefore cannot spawn a
7614    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7615    // other two steps and why the window matters.
7616    #[cfg(windows)]
7617    subc_jobobject::suspend_on_create_async(&mut command);
7618    let mut child = match command.spawn() {
7619        Ok(child) => child,
7620        Err(source) => {
7621            #[cfg(target_os = "linux")]
7622            if let Some(placement) = cgroup_placement {
7623                remove_module_cgroup(placement, &cgroup_name);
7624            }
7625            return Err(SuperviseError::Spawn {
7626                program: spec.program.clone(),
7627                source,
7628                cgroup_path,
7629            });
7630        }
7631    };
7632    // The parent must close its writer now: the acknowledgement pipe reports EOF
7633    // only when every writer is gone, and the child's copy closes when the
7634    // trampoline replaces itself with the module. Command holds only an integer
7635    // in its pre_exec callback, not another writer.
7636    #[cfg(target_os = "macos")]
7637    drop(exec_ack);
7638
7639    // Containment, steps 2 and 3: assign while suspended, then resume.
7640    #[cfg(windows)]
7641    let job = contain_spawned_child(&child, spec)?;
7642    let spawned_at_ms = unix_ms_now();
7643    let spawned_from = spec.program.clone();
7644    let spawned_file_identity = spawned_file_identity(&spawned_from);
7645    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7646        program: spec.program.clone(),
7647        source: io::Error::other("spawned child exposed no live pid"),
7648        cgroup_path: cgroup_path.clone(),
7649    })?;
7650    let process_start_time = crate::provenance::process_start_time(pid);
7651    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7652    #[cfg(all(test, target_os = "macos"))]
7653    privacy_exec_boundary_tests::before_image_sample(spec, pid);
7654    // Unix spawn returns after exec's error pipe closes. The kernel image is
7655    // therefore the executable to compare during a future orphan sweep: PATH
7656    // lookup and shebang interpretation may select a different file from the
7657    // configured program. Keep the literal program's identity for provenance,
7658    // but never use it as proof that a recorded pid may be signalled.
7659    let recorded_image = observe_spawned_image(pid);
7660    // spawn() confirms only the first exec, into the trampoline. Never persist
7661    // the trampoline image; the asynchronous acknowledgement publishes the
7662    // module image once the trampoline has replaced itself with the module.
7663    #[cfg(target_os = "macos")]
7664    let recorded_image = if privacy_exec.is_some() {
7665        None
7666    } else {
7667        recorded_image
7668    };
7669    #[cfg(target_os = "linux")]
7670    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7671    #[cfg(not(target_os = "linux"))]
7672    let recorded_cgroup_name = None;
7673    let roster_guard = roster.admit(
7674        spec.module_id.clone(),
7675        pid,
7676        spec.protocol,
7677        process_start_time,
7678        crate::child_roster::RecordedIdentity {
7679            start_time: recorded_image.map(|image| image.start_time),
7680            executable: recorded_image
7681                .and_then(|image| image.executable)
7682                .map(crate::live_children::ExecutableIdentity::from),
7683            cgroup_name: recorded_cgroup_name,
7684            #[cfg(target_os = "linux")]
7685            cgroup_placement: cgroup_placement.cloned(),
7686        },
7687    );
7688    // The check at the top of this function can pass just before daemon
7689    // shutdown begins, and the process is only in the roster from here on.
7690    // The shutdown stop returns as soon as it finds the roster empty, so a
7691    // process admitted after that look would outlive the daemon. The roster
7692    // is closed before the stop first reads it and admission happens under
7693    // the roster's lock, so either the stop sees this process or this check
7694    // sees the roster closed: end the process now rather than start a module
7695    // the daemon is about to stop.
7696    if roster.is_closed() {
7697        // This child was never admitted, so there is no module protocol shutdown to wait for.
7698        #[cfg(target_os = "linux")]
7699        kill_module_cgroup(cgroup_placement, &cgroup_name);
7700        if let Err(error) = child.start_kill() {
7701            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7702        }
7703        #[cfg(target_os = "linux")]
7704        if let Some(placement) = cgroup_placement {
7705            // This spawn was never admitted, so shutdown has no roster entry
7706            // to await. Do not detach its cleanup: the runtime could exit
7707            // before that task reaps the rejected child and removes its group.
7708            while matches!(child.try_wait(), Ok(None)) {
7709                std::thread::yield_now();
7710            }
7711            if matches!(
7712                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7713                subc_cgroup::KillOutcome::Killed
7714            ) {
7715                if let Ok(path) = placement.module_path(&cgroup_name) {
7716                    while std::fs::read_to_string(path.join("cgroup.events"))
7717                        .ok()
7718                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7719                    {
7720                        std::thread::yield_now();
7721                    }
7722                }
7723            }
7724            remove_module_cgroup(placement, &cgroup_name);
7725        }
7726        drop(roster_guard);
7727        return Err(SuperviseError::Spawn {
7728            program: spec.program.clone(),
7729            source: io::Error::other(
7730                "the daemon began shutting down while this process was starting; ended it",
7731            ),
7732            cgroup_path,
7733        });
7734    }
7735
7736    let stdout_pump = match child.stdout.take() {
7737        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7738        None => {
7739            warn!(
7740                module_id = %spec.module_id,
7741                "spawned child exposed no stdout pipe; file capture will be incomplete"
7742            );
7743            None
7744        }
7745    };
7746    let stderr_pump = match child.stderr.take() {
7747        Some(stderr) => {
7748            let generation = ring
7749                .lock()
7750                .unwrap_or_else(|poisoned| poisoned.into_inner())
7751                .begin_process();
7752            Some(StderrPump {
7753                task: tokio::spawn(pump_stderr_to(
7754                    stderr,
7755                    Arc::clone(ring),
7756                    generation,
7757                    output_sink,
7758                )),
7759                generation,
7760            })
7761        }
7762        None => {
7763            // Spawning succeeded but the pipe did not materialise. Recording it as
7764            // uncaptured keeps the tail honest: the alternative is an empty tail
7765            // that reads as a module which printed nothing.
7766            ring.lock()
7767                .unwrap_or_else(|poisoned| poisoned.into_inner())
7768                .mark_not_captured("stderr pipe was not available on spawn");
7769            warn!(
7770                module_id = %spec.module_id,
7771                "spawned child exposed no stderr pipe; tail will be unavailable"
7772            );
7773            None
7774        }
7775    };
7776
7777    Ok(SupervisedChild {
7778        child,
7779        protocol: spec.protocol,
7780        #[cfg(target_os = "linux")]
7781        module_id: cgroup_name,
7782        #[cfg(target_os = "linux")]
7783        cgroup_placement: cgroup_placement.cloned(),
7784        #[cfg(windows)]
7785        job,
7786        stdout_pump,
7787        stderr_pump,
7788        stderr_ring: Arc::clone(ring),
7789        spawned_at_ms,
7790        spawned_from,
7791        spawned_file_identity,
7792        process_start_time,
7793        process_identity,
7794        pid,
7795        roster_guard: Some(roster_guard),
7796        #[cfg(target_os = "macos")]
7797        privacy_exec,
7798        #[cfg(target_os = "macos")]
7799        report_ready: Arc::new(OnceLock::new()),
7800        spawn_failure: None,
7801    })
7802}
7803
7804#[cfg(target_os = "linux")]
7805pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7806    use subc_cgroup::KillOutcome;
7807    match subc_cgroup::kill_module(placement, module_id) {
7808        KillOutcome::Killed => {}
7809        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7810            debug!(
7811                module_id,
7812                "cgroup tree kill unavailable; using direct-child kill"
7813            );
7814        }
7815        KillOutcome::IoError { path, error } => {
7816            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7817        }
7818    }
7819}
7820
7821/// Contain a freshly spawned Windows child and start it.
7822///
7823/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7824/// child assigned **while it is still suspended** (step 1 is
7825/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7826///
7827/// A child that is never resumed hangs forever holding a pid, so a resume
7828/// failure kills the child and fails the spawn rather than returning a
7829/// `SupervisedChild` that can never run.
7830///
7831/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7832/// it did before this existed, whereas refusing to start one would be a new
7833/// outage. It is logged at warn because it means a helper process could leak.
7834#[cfg(windows)]
7835fn contain_spawned_child(
7836    child: &Child,
7837    spec: &ModuleSpec,
7838) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7839    let module_id = spec.module_id.as_str();
7840    let Some(pid) = child.id() else {
7841        // The child exited between spawn and here. Its tree, if it made one,
7842        // needs no containment: nothing is left to contain.
7843        warn!(
7844            module_id,
7845            "spawned child had already exited before containment; no job object attached"
7846        );
7847        return Ok(None);
7848    };
7849
7850    let job = match subc_jobobject::JobObject::new() {
7851        Ok(job) => job,
7852        Err(source) => {
7853            warn!(
7854                module_id,
7855                error = %source,
7856                "could not create a job object; this module's helper processes will not be \
7857                 reaped on teardown"
7858            );
7859            // Resume regardless: leaving the child suspended would turn a
7860            // containment gap into a hung module.
7861            resume_suspended_child(pid, spec)?;
7862            return Ok(None);
7863        }
7864    };
7865
7866    if let Err(source) = job.assign(child) {
7867        warn!(
7868            module_id,
7869            error = %source,
7870            "could not assign the child to its job object; this module's helper processes \
7871             will not be reaped on teardown"
7872        );
7873        resume_suspended_child(pid, spec)?;
7874        return Ok(None);
7875    }
7876
7877    resume_suspended_child(pid, spec)?;
7878    Ok(Some(job))
7879}
7880
7881/// Resume a suspended child, killing it if it cannot be started.
7882///
7883/// A suspended process holds a pid and does nothing, so there is no useful
7884/// state to return: the caller gets an error and the spawn fails.
7885#[cfg(windows)]
7886fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7887    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7888        // Kill it here rather than leaving a suspended process for the caller
7889        // to notice; `kill_on_drop` would eventually do this, but the module
7890        // would have been reported as running in between.
7891        let _ = std::process::Command::new("taskkill.exe")
7892            .args(["/PID", &pid.to_string(), "/T", "/F"])
7893            .stdin(Stdio::null())
7894            .stdout(Stdio::null())
7895            .stderr(Stdio::null())
7896            .status();
7897        return Err(SuperviseError::Spawn {
7898            program: spec.program.clone(),
7899            source,
7900            cgroup_path: None,
7901        });
7902    }
7903    Ok(())
7904}
7905
7906#[cfg(target_os = "linux")]
7907fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7908    match placement.remove_module(module_id) {
7909        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7910        Err(error) => warn!(
7911            module_id,
7912            error = %error,
7913            "could not remove module cgroup after process exit; continuing teardown"
7914        ),
7915    }
7916}
7917
7918#[cfg(target_os = "linux")]
7919async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7920    // Reaping the direct child is not proof its descendants exited. End the
7921    // residual tree and wait for the kernel's population fact before rmdir;
7922    // otherwise a successful parent wait leaks a directory on each restart.
7923    if matches!(
7924        subc_cgroup::kill_module(Some(placement), module_id),
7925        subc_cgroup::KillOutcome::Killed
7926    ) {
7927        if let Ok(path) = placement.module_path(module_id) {
7928            while std::fs::read_to_string(path.join("cgroup.events"))
7929                .ok()
7930                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7931            {
7932                sleep(Duration::from_millis(1)).await;
7933            }
7934        }
7935    }
7936    remove_module_cgroup(placement, module_id);
7937}
7938
7939#[cfg(target_os = "linux")]
7940fn apply_cgroup_placement(
7941    command: &mut Command,
7942    spec: &ModuleSpec,
7943    path: &std::path::Path,
7944) -> Result<(), SuperviseError> {
7945    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7946        module_id: spec.module_id.clone(),
7947        source,
7948    })
7949}
7950
7951fn capture_retention(spec: &ModuleSpec) -> Retention {
7952    let defaults = Retention::default();
7953    let value = |name: &str| {
7954        spec.env
7955            .iter()
7956            .rev()
7957            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7958    };
7959    Retention {
7960        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7961            .and_then(|value| value.parse().ok())
7962            .unwrap_or(defaults.max_file_mb),
7963        keep: value(CAPTURE_KEEP_ENV)
7964            .and_then(|value| value.parse().ok())
7965            .unwrap_or(defaults.keep),
7966        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7967            .and_then(|value| value.parse().ok())
7968            .unwrap_or(defaults.max_age_days),
7969    }
7970}
7971
7972/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
7973/// module's registration to the exact process the supervisor spawned.
7974fn generate_launch_nonce() -> Result<String, SuperviseError> {
7975    let mut bytes = [0u8; 32];
7976    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7977        reason: source.to_string(),
7978    })?;
7979    let mut hex = String::with_capacity(64);
7980    for b in bytes {
7981        use std::fmt::Write;
7982        let _ = write!(hex, "{b:02x}");
7983    }
7984    Ok(hex)
7985}
7986
7987/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
7988/// signal about how many leading bytes matched.
7989fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7990    if a.len() != b.len() {
7991        return false;
7992    }
7993    let mut diff = 0u8;
7994    for (x, y) in a.iter().zip(b.iter()) {
7995        diff |= x ^ y;
7996    }
7997    diff == 0
7998}
7999
8000/// The kernel's image after an acknowledged exec, shared by ordinary launches
8001/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
8002/// not identities inferred from a configured pathname.
8003fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
8004    subc_os::Process::open(pid)
8005        .ok()
8006        .flatten()
8007        .and_then(|process| process.observe())
8008}
8009
8010fn spawn_and_mark_running(
8011    spec: &ModuleSpec,
8012    runtime: &SupervisorRuntimeConfig,
8013    snapshot: &SharedSnapshot,
8014) -> Result<SupervisedChild, SuperviseError> {
8015    let child = spawn_child(
8016        spec,
8017        runtime.connection_file_path.as_deref(),
8018        runtime.supervisor_handle.as_ref(),
8019        &runtime.stderr_ring,
8020        runtime.capture_logs_dir.as_deref(),
8021        &runtime.child_roster,
8022        #[cfg(target_os = "linux")]
8023        runtime.cgroup_placement.as_ref(),
8024    )?;
8025    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
8026    Ok(child)
8027}
8028
8029enum RegistrationWaitOutcome {
8030    Registered,
8031    Exited(ExitReport),
8032    TimedOut,
8033}
8034
8035struct ReloadRegistrationFailure {
8036    exit_report: ExitReport,
8037    reason: String,
8038}
8039
8040#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8041enum BusyGaugeObservation {
8042    Quiescent,
8043    Busy,
8044    Omitted,
8045}
8046
8047fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8048    let Some(metrics) = metrics.and_then(Value::as_object) else {
8049        return BusyGaugeObservation::Omitted;
8050    };
8051    let mut sum = 0u128;
8052    for gauge in gauges {
8053        let Some(value) = metrics.get(gauge) else {
8054            return BusyGaugeObservation::Omitted;
8055        };
8056        let Some(value) = value.as_u64() else {
8057            return BusyGaugeObservation::Busy;
8058        };
8059        sum = sum.saturating_add(u128::from(value));
8060    }
8061    if sum == 0 {
8062        BusyGaugeObservation::Quiescent
8063    } else {
8064        BusyGaugeObservation::Busy
8065    }
8066}
8067
8068fn declared_busy_gauges(
8069    registry: &Registry,
8070    module_id: &str,
8071) -> Result<Vec<String>, SuperviseError> {
8072    busy_gauges_of(
8073        registry
8074            .get_module(module_id)
8075            .map_err(SuperviseError::Registry)?,
8076    )
8077}
8078
8079/// [`declared_busy_gauges`] for the registration a connection holds, in any
8080/// slot: after cutover the incumbent is no longer the id's active
8081/// registration, and its own manifest is the one that names its gauges.
8082fn declared_busy_gauges_for_connection(
8083    registry: &Registry,
8084    connection_id: ConnectionId,
8085) -> Result<Vec<String>, SuperviseError> {
8086    busy_gauges_of(
8087        registry
8088            .get_module_by_connection(connection_id)
8089            .map_err(SuperviseError::Registry)?,
8090    )
8091}
8092
8093fn busy_gauges_of(
8094    registration: Option<crate::registry::ModuleRegistration>,
8095) -> Result<Vec<String>, SuperviseError> {
8096    let Some(registration) = registration else {
8097        return Ok(Vec::new());
8098    };
8099    let Some(self_signals) = registration.manifest.self_signals else {
8100        return Ok(Vec::new());
8101    };
8102
8103    let mut gauges = Vec::new();
8104    for declaration in self_signals {
8105        if declaration.kind != SelfSignalKind::Busy {
8106            continue;
8107        }
8108        match declaration.anchored_to {
8109            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8110                gauges.extend(declared)
8111            }
8112            _ => {
8113                // An invalid Busy anchor is fail-safe: the empty name cannot be
8114                // present in a conforming health report, so this drain stays busy.
8115                gauges.push(String::new());
8116            }
8117        }
8118    }
8119    Ok(gauges)
8120}
8121
8122/// Wait for `endpoint` to have nothing in flight and, when the module declares
8123/// busy gauges, for a health probe to report them quiet. The probe is addressed
8124/// by `scope`: a swap's superseded incumbent must be asked about its own
8125/// gauges, and by module id the probe would reach the promoted candidate.
8126async fn wait_for_forwarding_quiescence(
8127    forwarding: &ForwardingTable,
8128    module_id: &str,
8129    runtime: &SupervisorRuntimeConfig,
8130    endpoint: crate::ModuleEndpointId,
8131    deadline: Instant,
8132    busy_gauges: &[String],
8133    scope: DrainScope,
8134) -> Result<bool, SuperviseError> {
8135    let mut gauges_quiescent = busy_gauges.is_empty();
8136    let mut next_probe_at = Instant::now();
8137    let mut omission_counted = false;
8138
8139    loop {
8140        let now = Instant::now();
8141        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8142            let report = match scope {
8143                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8144                DrainScope::Endpoint(endpoint) => {
8145                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8146                }
8147            };
8148            gauges_quiescent = match report {
8149                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8150                    BusyGaugeObservation::Quiescent => true,
8151                    BusyGaugeObservation::Busy => false,
8152                    BusyGaugeObservation::Omitted => {
8153                        if !omission_counted {
8154                            forwarding
8155                                .counters()
8156                                .increment_drains_with_undeclared_gauge();
8157                            omission_counted = true;
8158                        }
8159                        false
8160                    }
8161                },
8162                Err(err) => {
8163                    warn!(
8164                        module_id,
8165                        error = %err,
8166                        "drain health.check did not produce declared busy gauges; treating module as busy"
8167                    );
8168                    false
8169                }
8170            };
8171            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8172        }
8173
8174        let in_flight = forwarding
8175            .endpoint_in_flight_count(endpoint)
8176            .map_err(SuperviseError::Forwarding)?;
8177        if in_flight == 0 && gauges_quiescent {
8178            return Ok(true);
8179        }
8180
8181        let now = Instant::now();
8182        if now >= deadline {
8183            return Ok(false);
8184        }
8185        let mut wait = deadline
8186            .saturating_duration_since(now)
8187            .min(REGISTRY_RELEASE_POLL);
8188        if !busy_gauges.is_empty() {
8189            wait = wait.min(next_probe_at.saturating_duration_since(now));
8190        }
8191        sleep(wait).await;
8192    }
8193}
8194
8195/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8196///
8197/// `Ok` is always honest and passed straight through -- the wait actually measured
8198/// in-flight state. `Err` means the wait produced no measurement at all (the
8199/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8200/// constant: the drain did not complete. Never recomputed from route state, never a
8201/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8202fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8203    match wait_result {
8204        Ok(drained) => *drained,
8205        Err(_) => false,
8206    }
8207}
8208
8209fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8210    for released in released_routes {
8211        let frame = match Frame::build_with_version(
8212            released.negotiated_ver,
8213            FrameType::Goodbye,
8214            control_flags(),
8215            released.channel,
8216            released.epoch,
8217            0,
8218            Vec::new(),
8219        ) {
8220            Ok(frame) => frame,
8221            Err(err) => {
8222                warn!(
8223                    route_channel = released.channel,
8224                    error = %err,
8225                    "failed to build supervisor drain route GOODBYE frame"
8226                );
8227                continue;
8228            }
8229        };
8230        if !released.close_on_delivery_failure() {
8231            crate::forwarding::send_module_route_goodbye(
8232                &forwarding.counters(),
8233                &released.sink,
8234                frame,
8235                released.module_id.as_deref(),
8236                "supervisor drain",
8237            );
8238            continue;
8239        }
8240        if let Err(err) = released.sink.try_send(frame) {
8241            warn!(
8242                target_connection_id = released.connection_id.get(),
8243                route_channel = released.channel,
8244                error = %err,
8245                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8246            );
8247            let _ = forwarding.escalate_client_delivery_failure(
8248                released.connection_id,
8249                released.channel,
8250                released.epoch,
8251                CloseReason::new(
8252                    "route_goodbye_delivery_failed",
8253                    format!(
8254                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8255                        released.channel
8256                    ),
8257                ),
8258                crate::forwarding::UndeliveredFrame {
8259                    module_id: released.module_id.as_deref(),
8260                    sink: &released.sink,
8261                },
8262            );
8263        }
8264    }
8265}
8266
8267fn send_module_draining(
8268    module_id: &str,
8269    reason: RouteCloseReason,
8270    deadline_ms: u64,
8271    target: &ModuleDrainTarget,
8272) {
8273    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8274        reason,
8275        deadline_ms,
8276    }) {
8277        Ok(body) => body,
8278        Err(err) => {
8279            warn!(
8280                module_id,
8281                error = %err,
8282                "failed to encode module draining command"
8283            );
8284            return;
8285        }
8286    };
8287    let frame = match Frame::build_with_version(
8288        target.negotiated_ver,
8289        FrameType::Push,
8290        control_flags(),
8291        0,
8292        0,
8293        0,
8294        body,
8295    ) {
8296        Ok(frame) => frame,
8297        Err(err) => {
8298            warn!(
8299                module_id,
8300                error = %err,
8301                "failed to build module draining command frame"
8302            );
8303            return;
8304        }
8305    };
8306    if let Err(err) = target.sink.try_send(frame) {
8307        warn!(
8308            module_id,
8309            target_connection_id = target.endpoint.connection_id.get(),
8310            error = %err,
8311            "module draining command was not delivered to peer"
8312        );
8313    }
8314}
8315
8316/// The channel-0 GOODBYE that tells a module its stop is planned.
8317fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8318    match Frame::build_with_version(
8319        negotiated_ver,
8320        FrameType::Goodbye,
8321        control_flags(),
8322        0,
8323        0,
8324        0,
8325        Vec::new(),
8326    ) {
8327        Ok(frame) => Some(frame),
8328        Err(err) => {
8329            warn!(
8330                module_id,
8331                error = %err,
8332                "failed to build module GOODBYE frame"
8333            );
8334            None
8335        }
8336    }
8337}
8338
8339/// Send every registered module connection its module GOODBYE at daemon
8340/// shutdown, then request that connection's close.
8341///
8342/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8343/// before EOF, so the GOODBYE must reach the socket before the close. A close
8344/// request does not wait for the connection's queued frames: its writer gets a
8345/// bounded grace after the close, is aborted if it overruns it, and the daemon
8346/// process may exit before that grace ends. So with `wait_for_flush`, each
8347/// connection is closed only after its writer has acknowledged writing the
8348/// GOODBYE, or once a short shared budget runs out, so one module that is not
8349/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8350/// are only queued, for a shutdown the operator has told to stop waiting.
8351/// A connection that is already gone is skipped.
8352#[cfg(unix)]
8353async fn send_module_goodbyes_for_daemon_shutdown(
8354    forwarding: &Arc<ForwardingTable>,
8355    reason: &CloseReason,
8356    wait_for_flush: bool,
8357) {
8358    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8359    let targets = match forwarding.module_connections() {
8360        Ok(targets) => targets,
8361        Err(err) => {
8362            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8363            return;
8364        }
8365    };
8366    let deadline = Instant::now() + GOODBYE_BUDGET;
8367    let mut sends = tokio::task::JoinSet::new();
8368    for target in targets {
8369        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8370            continue;
8371        };
8372        if !wait_for_flush {
8373            if let Err(err) = target.sink.try_send(frame) {
8374                debug!(
8375                    module_id = %target.module_id,
8376                    error = %err,
8377                    "shutdown module GOODBYE was not queued"
8378                );
8379            }
8380            continue;
8381        }
8382        let forwarding = Arc::clone(forwarding);
8383        let reason = reason.clone();
8384        sends.spawn(async move {
8385            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8386                Ok(Ok(())) => {}
8387                Ok(Err(err)) => debug!(
8388                    module_id = %target.module_id,
8389                    error = %err,
8390                    "module connection closed before its shutdown GOODBYE was written"
8391                ),
8392                Err(_) => warn!(
8393                    module_id = %target.module_id,
8394                    budget = ?GOODBYE_BUDGET,
8395                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8396                ),
8397            }
8398            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8399        });
8400    }
8401    // Every task ends by the shared deadline, so this wait is bounded too.
8402    while sends.join_next().await.is_some() {}
8403}
8404
8405fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8406    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8407        return;
8408    };
8409    if let Err(err) = target.sink.try_send(frame) {
8410        warn!(
8411            module_id,
8412            target_connection_id = target.endpoint.connection_id.get(),
8413            error = %err,
8414            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8415        );
8416        forwarding.request_connection_close(
8417            target.endpoint.connection_id,
8418            CloseReason::new(
8419                "module_goodbye_delivery_failed",
8420                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8421            ),
8422        );
8423    }
8424}
8425
8426#[derive(Clone, Copy)]
8427struct ForwardingDrainContext<'a> {
8428    spec: &'a ModuleSpec,
8429    runtime: &'a SupervisorRuntimeConfig,
8430    registry: &'a Registry,
8431    scope: DrainScope,
8432}
8433
8434/// Which process a forwarding drain addresses.
8435#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8436enum DrainScope {
8437    /// Whatever endpoint is active for the module id: every plain stop,
8438    /// restart and reload. Also moves the module's state to `Draining`.
8439    Active,
8440    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8441    /// module id would resolve to the promoted candidate and leave neither
8442    /// process routable. The module's state is left alone, since the promoted
8443    /// candidate is what it describes and that process is running.
8444    Endpoint(crate::ModuleEndpointId),
8445}
8446
8447/// Whether a child being drained has already been asked to stop by the time
8448/// its drain wait starts.
8449///
8450/// The drain wait is the same budget whatever this says. What it decides is
8451/// whether the supervisor must ask by signal before that wait begins: a child
8452/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8453/// healthy or not.
8454#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8455enum StopNotice {
8456    /// The module was sent `module.draining` and a module GOODBYE over its own
8457    /// registered connection, and stops itself.
8458    SentOverConnection,
8459    /// The forwarding drain found no registered connection for the module: a
8460    /// subc child spawned moments ago that has not sent HELLO yet, or a
8461    /// `protocol: "none"` child, which never registers.
8462    NoConnection,
8463    /// This path sends nothing over the module's connection: the supervisor has
8464    /// no forwarding table, or the caller stops the child without a forwarding
8465    /// drain.
8466    NotSent,
8467}
8468
8469async fn begin_forwarding_drain(
8470    spec: &ModuleSpec,
8471    runtime: &SupervisorRuntimeConfig,
8472    registry: &Registry,
8473    snapshot: &SharedSnapshot,
8474    enabled: Option<bool>,
8475    reason: RouteCloseReason,
8476) -> Result<StopNotice, SuperviseError> {
8477    let Some(forwarding) = runtime.forwarding.as_ref() else {
8478        return Err(SuperviseError::ReloadUnavailable {
8479            module_id: spec.module_id.clone(),
8480            reason: "supervisor was not configured with a forwarding table".to_string(),
8481        });
8482    };
8483
8484    begin_forwarding_drain_with(
8485        forwarding,
8486        ForwardingDrainContext {
8487            spec,
8488            runtime,
8489            registry,
8490            scope: DrainScope::Active,
8491        },
8492        snapshot,
8493        enabled,
8494        reason,
8495        runtime.drain_timeout,
8496    )
8497    .await
8498}
8499
8500async fn begin_forwarding_drain_if_configured(
8501    spec: &ModuleSpec,
8502    runtime: &SupervisorRuntimeConfig,
8503    registry: &Registry,
8504    snapshot: &SharedSnapshot,
8505    enabled: Option<bool>,
8506    reason: RouteCloseReason,
8507) -> Result<StopNotice, SuperviseError> {
8508    begin_forwarding_drain_with_timeout(
8509        spec,
8510        runtime,
8511        registry,
8512        snapshot,
8513        enabled,
8514        reason,
8515        runtime.drain_timeout,
8516    )
8517    .await
8518}
8519
8520/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8521/// budget, for paths where the operator overrides the module's configured one
8522/// (`supervisor.restart{drain_timeout_ms}`).
8523async fn begin_forwarding_drain_with_timeout(
8524    spec: &ModuleSpec,
8525    runtime: &SupervisorRuntimeConfig,
8526    registry: &Registry,
8527    snapshot: &SharedSnapshot,
8528    enabled: Option<bool>,
8529    reason: RouteCloseReason,
8530    drain_timeout: Duration,
8531) -> Result<StopNotice, SuperviseError> {
8532    let Some(forwarding) = runtime.forwarding.as_ref() else {
8533        return Ok(StopNotice::NotSent);
8534    };
8535
8536    begin_forwarding_drain_with(
8537        forwarding,
8538        ForwardingDrainContext {
8539            spec,
8540            runtime,
8541            registry,
8542            scope: DrainScope::Active,
8543        },
8544        snapshot,
8545        enabled,
8546        reason,
8547        drain_timeout,
8548    )
8549    .await
8550}
8551
8552async fn begin_forwarding_drain_with(
8553    forwarding: &ForwardingTable,
8554    context: ForwardingDrainContext<'_>,
8555    snapshot: &SharedSnapshot,
8556    enabled: Option<bool>,
8557    reason: RouteCloseReason,
8558    drain_timeout: Duration,
8559) -> Result<StopNotice, SuperviseError> {
8560    let ForwardingDrainContext {
8561        spec,
8562        runtime,
8563        registry,
8564        scope,
8565    } = context;
8566    debug_assert_ne!(reason, RouteCloseReason::Crash);
8567    let terminal = matches!(reason, RouteCloseReason::Disable);
8568    let drain_started_at = Instant::now();
8569    let drain_deadline = drain_started_at + drain_timeout;
8570    let deadline_ms =
8571        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8572    let busy_gauges = match scope {
8573        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8574        DrainScope::Endpoint(endpoint) => {
8575            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8576        }
8577    };
8578
8579    // Admission gate first: route.open/commit and route REQUEST admission are closed
8580    // before the first quiescence check, so the outstanding count can only fall.
8581    let gate_started = Instant::now();
8582    let drain_target = match scope {
8583        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8584        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8585    }
8586    .map_err(SuperviseError::Forwarding)?;
8587    // The instant admission closed, and how long taking the forwarding write
8588    // lock to close it took. The timeout line reports only the quiescence
8589    // wait, so without this a drain that started late looked like one that
8590    // started on time.
8591    info!(
8592        module_id = %spec.module_id,
8593        ?reason,
8594        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8595        connected = drain_target.is_some(),
8596        "module drain began; route admission closed"
8597    );
8598    if scope == DrainScope::Active {
8599        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8600            state.state = ModuleState::Draining;
8601            state.draining_to_replace =
8602                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8603            if let Some(enabled) = enabled {
8604                state.enabled = enabled;
8605            }
8606        })?;
8607    }
8608
8609    let Some(target) = drain_target.as_ref() else {
8610        // Nothing was sent: the module has no registered connection to carry
8611        // `module.draining` or a GOODBYE. The caller must not assume the child
8612        // was asked to stop.
8613        return Ok(StopNotice::NoConnection);
8614    };
8615    {
8616        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8617        let routes = forwarding
8618            .endpoint_routes(target.endpoint)
8619            .map_err(SuperviseError::Forwarding)?;
8620        let routes_notified = routes.len();
8621        crate::control::send_route_control_pushes(
8622            forwarding,
8623            routes.clone(),
8624            ClientControlPush::RouteClosing {
8625                module_id: spec.module_id.clone(),
8626                channels: Vec::new(),
8627                reason,
8628            },
8629        );
8630        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8631
8632        // `route.closing` was just sent above: from here on every return path,
8633        // including an early one, MUST send `route.closed` before propagating
8634        // anything else. A client holds `closing` as a promise that a verdict is
8635        // coming; leaving early without `closed` strands it waiting forever, since
8636        // `closing` carries no timeout of its own.
8637        let wait_result = wait_for_forwarding_quiescence(
8638            forwarding,
8639            &spec.module_id,
8640            runtime,
8641            target.endpoint,
8642            drain_deadline,
8643            &busy_gauges,
8644            scope,
8645        )
8646        .await;
8647        let drained = drained_after_quiescence_wait(&wait_result);
8648        if let Err(err) = &wait_result {
8649            error!(
8650                module_id = %spec.module_id,
8651                ?reason,
8652                error = %err,
8653                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8654            );
8655        } else if !drained {
8656            // Name what the drain waited on. Without it the line says only that
8657            // something did not settle, and "one wedged call" and "every
8658            // session's held stream" read the same; the first is a module bug,
8659            // the second is a module that should end its streams on
8660            // module.draining. Read before teardown releases the routes.
8661            let holdouts = forwarding
8662                .endpoint_drain_holdouts(target.endpoint)
8663                .unwrap_or_default();
8664            warn!(
8665                module_id = %spec.module_id,
8666                waited = ?drain_timeout,
8667                ?reason,
8668                held_requests = holdouts.requests,
8669                held_routes = holdouts.routes,
8670                total_routes = holdouts.total_routes,
8671                top_connections = ?holdouts.top_connections,
8672                // `module_channel:corr`, so the module can find each held request
8673                // in its own log; capped, so `held_requests` is the full count.
8674                held = %holdouts
8675                    .held
8676                    .iter()
8677                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8678                    .collect::<Vec<_>>()
8679                    .join(","),
8680                "route drain timed out before request quiescence; forcing teardown"
8681            );
8682        }
8683        crate::control::send_route_control_pushes(
8684            forwarding,
8685            routes,
8686            ClientControlPush::RouteClosed {
8687                module_id: spec.module_id.clone(),
8688                channels: Vec::new(),
8689                reason,
8690                drained,
8691                abandoned: target.abandoned_bindings.len() as u32,
8692                excluded_subscriptions: target.excluded_subscriptions,
8693                terminal: Some(terminal),
8694            },
8695        );
8696        wait_result?;
8697
8698        // `route.closed` has now been sent unconditionally above. From here the
8699        // remaining steps are cleanup (route + module GOODBYE) rather than a
8700        // promise the client is waiting on, but a lock-poisoned
8701        // `release_module_endpoint_routes` would otherwise skip the module
8702        // GOODBYE silently too -- send it before propagating the error.
8703        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8704            Ok(routes) => routes,
8705            Err(err) => {
8706                warn!(
8707                    module_id = %spec.module_id,
8708                    ?reason,
8709                    error = %err,
8710                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8711                );
8712                send_module_goodbye(&spec.module_id, forwarding, target);
8713                return Err(SuperviseError::Forwarding(err));
8714            }
8715        };
8716        let route_goodbye_count = released_routes.len();
8717        send_route_goodbyes(forwarding, released_routes);
8718        send_module_goodbye(&spec.module_id, forwarding, target);
8719
8720        // The drain's happy path was previously silent: every emission above is
8721        // best-effort with only its failure arm logged, so "were consumers told"
8722        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8723        // hang where the open question was exactly whether teardown notice went
8724        // out). One summary line makes that class decidable in one grep.
8725        info!(
8726            module_id = %spec.module_id,
8727            ?reason,
8728            routes_notified,
8729            route_goodbyes = route_goodbye_count,
8730            abandoned_reservations = target.abandoned_bindings.len(),
8731            excluded_subscriptions = target.excluded_subscriptions,
8732            drained,
8733            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8734        );
8735    }
8736
8737    Ok(StopNotice::SentOverConnection)
8738}
8739
8740/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8741/// the only slot a plain (non-swap) spawn can register into.
8742async fn wait_for_registration_after_reload(
8743    registry: &Registry,
8744    module_id: &str,
8745    snapshot: &SharedSnapshot,
8746    child: &mut SupervisedChild,
8747    wait: Duration,
8748) -> Result<RegistrationWaitOutcome, SuperviseError> {
8749    wait_for_slot_registration(
8750        registry,
8751        crate::registry::RegistrationSlot::Active(module_id),
8752        module_id,
8753        snapshot,
8754        child,
8755        wait,
8756    )
8757    .await
8758}
8759
8760/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8761///
8762/// Keyed on the slot rather than the bare module id because during a swap the
8763/// id's active slot is already held by the incumbent: an id-keyed wait would
8764/// report the incumbent's registration as the candidate's and a candidate that
8765/// never registers would look registered. A swap candidate waits on
8766/// `crate::registry::RegistrationSlot::Candidate`.
8767async fn wait_for_slot_registration(
8768    registry: &Registry,
8769    slot: crate::registry::RegistrationSlot<'_>,
8770    module_id: &str,
8771    snapshot: &SharedSnapshot,
8772    child: &mut SupervisedChild,
8773    wait: Duration,
8774) -> Result<RegistrationWaitOutcome, SuperviseError> {
8775    let deadline = Instant::now() + wait;
8776    loop {
8777        if registry
8778            .registration(slot)
8779            .map_err(SuperviseError::Registry)?
8780            .is_some()
8781        {
8782            return Ok(RegistrationWaitOutcome::Registered);
8783        }
8784
8785        let now = Instant::now();
8786        if now >= deadline {
8787            return Ok(RegistrationWaitOutcome::TimedOut);
8788        }
8789        let remaining = deadline.saturating_duration_since(now);
8790        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8791
8792        tokio::select! {
8793            wait_result = child.wait() => {
8794                let status = wait_result.map_err(|source| SuperviseError::Wait {
8795                    module_id: module_id.to_string(),
8796                    source,
8797                })?;
8798                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8799                    snapshot,
8800                    child,
8801                    &status,
8802                )));
8803            }
8804            _ = sleep(poll) => {}
8805        }
8806    }
8807}
8808
8809fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8810    // A replacement process that exits before HELLO did not provide service, even
8811    // if it used status 0. Count it against the restart cap as a new-binary failure.
8812    if exit_report.kind != ExitKind::DeliberateSeverance {
8813        exit_report.kind = ExitKind::Crash;
8814    }
8815    exit_report
8816}
8817
8818async fn handle_reload_child_registration_failure(
8819    spec: &ModuleSpec,
8820    runtime: &SupervisorRuntimeConfig,
8821    registry: &Registry,
8822    process_liveness: &SupervisorProcessLiveness,
8823    snapshot: &SharedSnapshot,
8824    _child: &mut Option<SupervisedChild>,
8825    failure: ReloadRegistrationFailure,
8826) -> Result<(), SuperviseError> {
8827    let ReloadRegistrationFailure {
8828        exit_report,
8829        reason,
8830    } = failure;
8831    match on_child_exit(
8832        spec,
8833        runtime.restart_policy,
8834        registry,
8835        snapshot,
8836        &runtime.terminal_ring,
8837        &runtime.spawn_events,
8838        &runtime.child_roster,
8839        exit_report,
8840    )
8841    .await
8842    {
8843        NextAction::Stop {
8844            registration_released,
8845        } => {
8846            if registration_released {
8847                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8848            }
8849        }
8850        NextAction::Restart { schedule } => {
8851            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8852                schedule.delay
8853            });
8854            if let Some(schedule) = schedule {
8855                log_crash_respawn(&spec.module_id, schedule);
8856            }
8857            schedule_respawn(
8858                runtime,
8859                snapshot,
8860                &spec.module_id,
8861                delay,
8862                RespawnKind::Spawn,
8863            )?;
8864        }
8865    }
8866    Err(SuperviseError::ReloadFailed {
8867        module_id: spec.module_id.clone(),
8868        reason,
8869    })
8870}
8871
8872async fn handle_reload_spawn_failure(
8873    spec: &ModuleSpec,
8874    runtime: &SupervisorRuntimeConfig,
8875    process_liveness: &SupervisorProcessLiveness,
8876    snapshot: &SharedSnapshot,
8877    _child: &mut Option<SupervisedChild>,
8878    reason: String,
8879) -> Result<(), SuperviseError> {
8880    let now = Instant::now();
8881    let mut schedule = None;
8882    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8883        clear_current_process_facts(state);
8884        if state.enabled {
8885            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8886            state.state = if schedule.is_some() {
8887                ModuleState::Restarting
8888            } else {
8889                ModuleState::Failed
8890            };
8891        } else {
8892            state.state = ModuleState::Disabled;
8893        }
8894    })?;
8895    if let Some(schedule) = schedule {
8896        schedule_respawn(
8897            runtime,
8898            snapshot,
8899            &spec.module_id,
8900            schedule.delay,
8901            RespawnKind::Spawn,
8902        )?;
8903    } else {
8904        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8905    }
8906    Err(SuperviseError::ReloadFailed {
8907        module_id: spec.module_id.clone(),
8908        reason,
8909    })
8910}
8911
8912fn control_flags() -> Flags {
8913    Flags::new(false, Priority::Passive, false)
8914}
8915
8916#[allow(clippy::too_many_arguments)]
8917async fn drain_optional_child(
8918    module_id: &str,
8919    protocol: ModuleProtocol,
8920    stop_notice: StopNotice,
8921    registry: &Registry,
8922    forwarding: Option<&ForwardingTable>,
8923    snapshot: &SharedSnapshot,
8924    terminal_ring: &Arc<Mutex<TerminalRing>>,
8925    spawn_events: &SpawnEventFeed,
8926    child: &mut Option<SupervisedChild>,
8927    drain_timeout: Duration,
8928    final_state: ModuleState,
8929    enabled: Option<bool>,
8930) -> Result<(), SuperviseError> {
8931    if let Some(child) = child.take() {
8932        drain_child_to_state(
8933            module_id,
8934            protocol,
8935            stop_notice,
8936            registry,
8937            forwarding,
8938            snapshot,
8939            terminal_ring,
8940            spawn_events,
8941            child,
8942            drain_timeout,
8943            final_state,
8944            enabled,
8945        )
8946        .await
8947    } else {
8948        update_snapshot(snapshot, Some(module_id), |state| {
8949            state.state = final_state;
8950            if let Some(enabled) = enabled {
8951                state.enabled = enabled;
8952            }
8953            clear_current_process_facts(state);
8954        })?;
8955        release_dead_registration(registry, forwarding, snapshot, module_id).await
8956    }
8957}
8958
8959#[allow(clippy::too_many_arguments)]
8960async fn drain_child_to_state(
8961    module_id: &str,
8962    _protocol: ModuleProtocol,
8963    stop_notice: StopNotice,
8964    registry: &Registry,
8965    forwarding: Option<&ForwardingTable>,
8966    snapshot: &SharedSnapshot,
8967    terminal_ring: &Arc<Mutex<TerminalRing>>,
8968    spawn_events: &SpawnEventFeed,
8969    mut child: SupervisedChild,
8970    drain_timeout: Duration,
8971    final_state: ModuleState,
8972    enabled: Option<bool>,
8973) -> Result<(), SuperviseError> {
8974    let protocol = child.protocol;
8975    update_snapshot(snapshot, Some(module_id), |state| {
8976        state.state = ModuleState::Draining;
8977        state.draining_to_replace = final_state == ModuleState::Restarting;
8978        if let Some(enabled) = enabled {
8979            state.enabled = enabled;
8980        }
8981    })?;
8982
8983    // The wait below is the same budget in every case; what differs is
8984    // whether anything has ASKED the child to stop before it starts. Only a
8985    // forwarding drain that reached the module's registered connection has
8986    // (`module.draining`, then a module GOODBYE). Every other child was told
8987    // nothing: a `protocol: "none"` module, which never registers; a subc
8988    // module spawned moments ago that has not sent HELLO yet; or a stop that
8989    // runs no forwarding drain. Without a signal the budget is only a delay
8990    // in front of SIGKILL -- and the not-yet-registered child is the worst
8991    // case, because it registers into a module that is already draining,
8992    // is never told, and is killed while healthy.
8993    if stop_notice != StopNotice::SentOverConnection {
8994        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8995            info!(
8996                module_id,
8997                pid = child.pid,
8998                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8999                "module has no connection yet; requesting stop by signal"
9000            );
9001        }
9002        request_graceful_stop(module_id, &child);
9003    }
9004
9005    let exit_report = match timeout(drain_timeout, child.wait()).await {
9006        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
9007        Ok(Err(source)) => {
9008            fail_snapshot(snapshot, Some(module_id), None);
9009            return Err(SuperviseError::Wait {
9010                module_id: module_id.to_string(),
9011                source,
9012            });
9013        }
9014        Err(_) => {
9015            // Mirror the sibling arm above: state is already `Draining`, and an
9016            // error propagated from here would strand it there -- a state
9017            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
9018            // `Failed | Stopped`), leaving an operator Restart as the only exit.
9019            // `Failed` before `?` keeps the module operator-visible and
9020            // revivable. Trigger is an ESRCH race (process exits between the
9021            // drain timeout firing and the kill) or a post-kill wait failure
9022            // (issue #34).
9023            //
9024            // Logged because the kill is otherwise visible only as signal 9 in
9025            // the terminal ring, and the budget it follows can be long enough
9026            // that consumers see a stretch of refusals with no stated cause.
9027            warn!(
9028                module_id,
9029                pid = child.pid,
9030                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9031                reason = ?final_state,
9032                ?stop_notice,
9033                "drain budget expired before the module exited; killing it"
9034            );
9035            child.start_kill().map_err(|source| {
9036                fail_snapshot(snapshot, Some(module_id), None);
9037                SuperviseError::Kill {
9038                    module_id: module_id.to_string(),
9039                    source,
9040                }
9041            })?;
9042            let status = child.wait().await.map_err(|source| {
9043                fail_snapshot(snapshot, Some(module_id), None);
9044                SuperviseError::Wait {
9045                    module_id: module_id.to_string(),
9046                    source,
9047                }
9048            })?;
9049            classify_reaped_child_exit(snapshot, &child, &status)
9050        }
9051    };
9052
9053    update_snapshot(snapshot, Some(module_id), |state| {
9054        state.state = final_state;
9055        if let Some(enabled) = enabled {
9056            state.enabled = enabled;
9057        }
9058        clear_current_process_facts(state);
9059        state.last_exit = Some(exit_report.clone());
9060        if exit_report.kind == ExitKind::DeliberateSeverance {
9061            state.lifetime_restarts += 1;
9062        }
9063    })?;
9064    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9065    record_terminal_with_detail(
9066        module_id,
9067        terminal_ring,
9068        spawn_events,
9069        &exit_report,
9070        terminal_disposition(final_state),
9071        detail,
9072    );
9073    child.drain_stderr(module_id).await;
9074
9075    release_dead_registration(registry, forwarding, snapshot, module_id).await
9076}
9077
9078/// Ask a child that nothing else has asked to stop, by signal.
9079///
9080/// A registered subc module is asked over its own connection: the drain sends
9081/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
9082/// module GOODBYE, and the module stops itself. A module that speaks no subc
9083/// wire receives none of that, and neither does a subc module that has not
9084/// registered yet, so for them the drain budget would be pure delay in front of
9085/// a SIGKILL -- and for a process with a store to flush (JetStream is the
9086/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
9087/// into a recovery on the next start.
9088///
9089/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
9090/// rule rather than an optimisation: that module's graceful stop is already
9091/// running by the time its child is drained, and a signal would race it.
9092///
9093/// Best-effort by construction. A child that has already exited is the ordinary
9094/// case rather than an error (the kill lands on a reaped or exiting pid), so a
9095/// failure is logged at debug and the wait-then-kill below still decides the
9096/// outcome.
9097#[cfg(unix)]
9098fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9099    let Some(pid) = child
9100        .id()
9101        .and_then(|pid| i32::try_from(pid).ok())
9102        .and_then(rustix::process::Pid::from_raw)
9103    else {
9104        debug!(
9105            module_id,
9106            "no pid to signal for teardown; falling through to the drain wait"
9107        );
9108        return;
9109    };
9110    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9111        Ok(()) => debug!(
9112            module_id,
9113            "sent SIGTERM to a module nothing else asked to stop"
9114        ),
9115        Err(err) => debug!(
9116            module_id,
9117            error = %err,
9118            "SIGTERM to module failed; the drain wait and kill still apply"
9119        ),
9120    }
9121}
9122
9123/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9124/// Windows does offer need cooperation this supervisor cannot assume: a console
9125/// control event requires sharing a console with the child, and `WM_CLOSE`
9126/// requires the child to pump a message loop. A supervised server process does
9127/// neither, so there is nothing to send and teardown is the wait followed by the
9128/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9129/// the thing `protocol: "none"` exists to avoid.
9130#[cfg(not(unix))]
9131fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9132    debug!(
9133        module_id,
9134        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9135    );
9136}
9137
9138fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9139    match final_state {
9140        ModuleState::Stopped => TerminalDisposition::Stopped,
9141        ModuleState::Disabled => TerminalDisposition::Disabled,
9142        ModuleState::Restarting => TerminalDisposition::Restarting,
9143        ModuleState::Failed => TerminalDisposition::Failed,
9144        ModuleState::Starting
9145        | ModuleState::Running
9146        | ModuleState::Unresponsive
9147        | ModuleState::Draining => {
9148            unreachable!("terminal exits only finish in terminal or restarting states")
9149        }
9150    }
9151}
9152
9153/// Release a reaped child's registration before allowing another spawn.
9154///
9155/// EOF is not a process-lifetime signal: an inherited socket can stay open
9156/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9157/// reading EOF. After the normal release grace, request connection close (which
9158/// cancels both reads and dispatch), then allow one more release grace for the
9159/// connection guard's forwarding cleanup. Never evict a different connection.
9160async fn release_dead_registration(
9161    registry: &Registry,
9162    forwarding: Option<&ForwardingTable>,
9163    snapshot: &SharedSnapshot,
9164    module_id: &str,
9165) -> Result<(), SuperviseError> {
9166    let result = async {
9167        let registration = registry
9168            .get_module(module_id)
9169            .map_err(SuperviseError::Registry)?;
9170        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9171            Ok(()) => return Ok(()),
9172            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9173            Err(err) => return Err(err),
9174        }
9175        let pid = lock_snapshot(snapshot)?.reaped_pid;
9176        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9177            warn!(
9178                module_id,
9179                pid,
9180                connection_id = registration.connection_id.get(),
9181                "reaped module registration outlived release grace; closing dead connection"
9182            );
9183            forwarding.request_connection_close(
9184                registration.connection_id,
9185                CloseReason::new(
9186                    "supervised_process_reaped",
9187                    format!("module '{module_id}' pid {pid} exited"),
9188                ),
9189            );
9190            wait_for_slot_registration_release(
9191                registry,
9192                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9193                REGISTRY_RELEASE_TIMEOUT,
9194            )
9195            .await?;
9196        }
9197        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9198    }
9199    .await;
9200    if let Err(err) = &result {
9201        fail_snapshot(snapshot, Some(module_id), None);
9202        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9203    }
9204    result
9205}
9206
9207/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9208/// plain stop or restart waits for before it spawns a replacement.
9209async fn wait_for_registration_release(
9210    registry: &Registry,
9211    module_id: &str,
9212    wait: Duration,
9213) -> Result<(), SuperviseError> {
9214    wait_for_slot_registration_release(
9215        registry,
9216        crate::registry::RegistrationSlot::Active(module_id),
9217        wait,
9218    )
9219    .await
9220}
9221
9222/// Wait for the registration in `slot` to go away.
9223///
9224/// Keyed on the slot rather than the bare module id because a successful swap
9225/// never empties the id's active slot (the promoted candidate is in it), so an
9226/// id-keyed wait for the incumbent's release would always time out. Draining a
9227/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9228/// incumbent's connection instead.
9229async fn wait_for_slot_registration_release(
9230    registry: &Registry,
9231    slot: crate::registry::RegistrationSlot<'_>,
9232    wait: Duration,
9233) -> Result<(), SuperviseError> {
9234    let deadline = Instant::now() + wait;
9235    let mut release_events = registration_release_events().subscribe();
9236    let still_active = |registration: &crate::registry::ModuleRegistration| {
9237        SuperviseError::RegistrationStillActive {
9238            module_id: registration.manifest.module_id.clone(),
9239            waited: wait,
9240        }
9241    };
9242    loop {
9243        let _observed_generation = *release_events.borrow_and_update();
9244        let Some(registration) = registry
9245            .registration(slot)
9246            .map_err(SuperviseError::Registry)?
9247        else {
9248            return Ok(());
9249        };
9250
9251        let now = Instant::now();
9252        if now >= deadline {
9253            return Err(still_active(&registration));
9254        }
9255
9256        let remaining = deadline.saturating_duration_since(now);
9257        match timeout(remaining, release_events.changed()).await {
9258            Ok(Ok(())) | Ok(Err(_)) => {}
9259            Err(_) => return Err(still_active(&registration)),
9260        }
9261    }
9262}
9263
9264#[cfg(test)]
9265mod slot_registration_wait_tests {
9266    use super::*;
9267    use crate::registry::{ConnectionId, RegistrationSlot};
9268    use subc_protocol::manifest::ModuleManifest;
9269
9270    #[tokio::test]
9271    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9272        let registry = Arc::new(Registry::default());
9273        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9274        let runtime = supervisor.runtime_config();
9275        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9276        let spec = ModuleSpec {
9277            module_id: "enable-stale-registration".to_string(),
9278            program: PathBuf::from("/missing/enable-retry-test"),
9279            args: Vec::new(),
9280            env: Vec::new(),
9281            reserved: false,
9282            reserved_prefixes: Vec::new(),
9283            protocol: ModuleProtocol::Subc,
9284            overlap: Default::default(),
9285        };
9286        let connection = ConnectionId::new(90);
9287        registry
9288            .register_with_control_ops(
9289                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9290                1,
9291                connection,
9292                Vec::new(),
9293            )
9294            .unwrap();
9295        let mut child = None;
9296        let err = set_child_enabled(
9297            &spec,
9298            &runtime,
9299            &registry,
9300            &supervisor.process_liveness,
9301            &snapshot,
9302            &mut child,
9303            true,
9304        )
9305        .await
9306        .unwrap_err();
9307        assert!(matches!(
9308            err,
9309            SuperviseError::RegistrationStillActive { .. }
9310        ));
9311        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9312        assert!(child.is_none());
9313        registry.deregister_connection(connection).unwrap();
9314        let err = set_child_enabled(
9315            &spec,
9316            &runtime,
9317            &registry,
9318            &supervisor.process_liveness,
9319            &snapshot,
9320            &mut child,
9321            true,
9322        )
9323        .await
9324        .unwrap_err();
9325        assert!(
9326            matches!(err, SuperviseError::Spawn { .. }),
9327            "second enable must attempt a spawn: {err}"
9328        );
9329        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9330    }
9331
9332    const INCUMBENT: u64 = 1;
9333    const CANDIDATE: u64 = 2;
9334
9335    fn swapped_registry() -> Arc<Registry> {
9336        let registry = Arc::new(Registry::default());
9337        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9338        registry
9339            .register_with_control_ops(
9340                manifest.clone(),
9341                1,
9342                ConnectionId::new(INCUMBENT),
9343                Vec::new(),
9344            )
9345            .unwrap();
9346        registry
9347            .register_candidate_with_control_ops(
9348                manifest,
9349                1,
9350                ConnectionId::new(CANDIDATE),
9351                Vec::new(),
9352            )
9353            .unwrap();
9354        registry
9355    }
9356
9357    /// After a promotion the id's active slot is held by the new process, so an
9358    /// id-keyed wait for the incumbent's release can never succeed; the
9359    /// connection-keyed wait completes as soon as the incumbent deregisters.
9360    #[tokio::test]
9361    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9362        let registry = swapped_registry();
9363        registry.promote_candidate("m").unwrap().unwrap();
9364
9365        assert!(matches!(
9366            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9367            Err(SuperviseError::RegistrationStillActive { .. })
9368        ));
9369
9370        // Still held while the incumbent's connection has not deregistered.
9371        assert!(matches!(
9372            wait_for_slot_registration_release(
9373                &registry,
9374                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9375                Duration::from_millis(50),
9376            )
9377            .await,
9378            Err(SuperviseError::RegistrationStillActive { .. })
9379        ));
9380
9381        let releaser = Arc::clone(&registry);
9382        let release = tokio::spawn(async move {
9383            sleep(Duration::from_millis(20)).await;
9384            releaser
9385                .deregister_connection(ConnectionId::new(INCUMBENT))
9386                .unwrap();
9387            notify_registration_release();
9388        });
9389        wait_for_slot_registration_release(
9390            &registry,
9391            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9392            Duration::from_secs(5),
9393        )
9394        .await
9395        .expect("the incumbent's own registration is released");
9396        release.await.unwrap();
9397        assert!(registry.get_module("m").unwrap().is_some());
9398    }
9399
9400    /// The candidate slot is waited on separately from the active slot: the
9401    /// incumbent's registration neither holds up nor stands in for it.
9402    #[tokio::test]
9403    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9404        let registry = swapped_registry();
9405        assert!(matches!(
9406            wait_for_slot_registration_release(
9407                &registry,
9408                RegistrationSlot::Candidate("m"),
9409                Duration::from_millis(50),
9410            )
9411            .await,
9412            Err(SuperviseError::RegistrationStillActive { .. })
9413        ));
9414        registry
9415            .deregister_connection(ConnectionId::new(CANDIDATE))
9416            .unwrap();
9417        wait_for_slot_registration_release(
9418            &registry,
9419            RegistrationSlot::Candidate("m"),
9420            Duration::from_millis(50),
9421        )
9422        .await
9423        .expect("a candidate slot with no candidate is released");
9424        assert!(registry
9425            .registration(RegistrationSlot::Active("m"))
9426            .unwrap()
9427            .is_some());
9428    }
9429}
9430
9431fn classify_exit(status: &ExitStatus) -> ExitReport {
9432    ExitReport {
9433        kind: if status.success() {
9434            ExitKind::Clean
9435        } else {
9436            ExitKind::Crash
9437        },
9438        code: status.code(),
9439        signal: exit_signal(status),
9440        at_ms: unix_ms_now(),
9441    }
9442}
9443
9444/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9445/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9446/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9447/// disposition still must be `Failed` so the terminal ring is not silently missing
9448/// an entry, matching what `fail_snapshot` records for this same arm.
9449fn wait_error_exit_report() -> ExitReport {
9450    ExitReport {
9451        kind: ExitKind::Crash,
9452        code: None,
9453        signal: None,
9454        at_ms: unix_ms_now(),
9455    }
9456}
9457
9458#[cfg(unix)]
9459fn exit_signal(status: &ExitStatus) -> Option<i32> {
9460    use std::os::unix::process::ExitStatusExt;
9461
9462    status.signal()
9463}
9464
9465#[cfg(not(unix))]
9466fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9467    None
9468}
9469
9470/// Give an operator-touched module its full crash budget back.
9471///
9472/// Named for the counter it used to zero; it now empties the in-window ring,
9473/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9474/// ledger of what happened survives every operator action.
9475fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9476    update_snapshot(snapshot, Some(module_id), |state| {
9477        state.clear_crash_restarts();
9478    })
9479}
9480
9481fn set_running(
9482    snapshot: &SharedSnapshot,
9483    child: &SupervisedChild,
9484    module_id: &str,
9485    spawn_events: &SpawnEventFeed,
9486) -> Result<(), SuperviseError> {
9487    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9488        module_id: Some(module_id.to_string()),
9489    })?;
9490    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9491    if std::mem::take(&mut state.coalesced_restart_pending) {
9492        let generation = state.spawn_generation;
9493        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9494    }
9495    state.drain_disposition_detail = None;
9496    state.spawn_failure = None;
9497    // Every caller of this is a plain spawn, which always uses the primary key;
9498    // a promoted swap candidate sets the flag itself after this returns.
9499    state.in_alternate_slot = false;
9500    state.configuration_updated_since_spawn = false;
9501    state.spawned_protocol = Some(child.protocol);
9502    state.state = ModuleState::Running;
9503    state.enabled = true;
9504    state.process_alive = true;
9505    state.pid = child.id();
9506    #[cfg(target_os = "macos")]
9507    {
9508        state.report_ready = Some(Arc::clone(&child.report_ready));
9509    }
9510    state.spawned_at_ms = Some(child.spawned_at_ms);
9511    state.spawned_from = Some(child.spawned_from.clone());
9512    state.spawned_file_identity = child.spawned_file_identity;
9513    state.process_start_time = child.process_start_time;
9514    Ok(())
9515}
9516
9517fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9518    state.process_alive = false;
9519    state.spawned_protocol = None;
9520    state.pid = None;
9521    #[cfg(target_os = "macos")]
9522    {
9523        state.report_ready = None;
9524    }
9525    state.spawned_at_ms = None;
9526    state.spawned_from = None;
9527    state.spawned_file_identity = None;
9528    state.process_start_time = None;
9529    state.deliberate_severance = None;
9530}
9531
9532#[cfg(test)]
9533fn record_deliberate_severance(
9534    snapshot: &SharedSnapshot,
9535    identity: ProcessIdentity,
9536) -> Result<(), SuperviseError> {
9537    update_snapshot(snapshot, None, |state| {
9538        state.deliberate_severance = Some(identity);
9539    })
9540}
9541
9542fn apply_deliberate_severance_marker(
9543    snapshot: &SharedSnapshot,
9544    exited_identity: Option<ProcessIdentity>,
9545    mut exit_report: ExitReport,
9546) -> ExitReport {
9547    let marker = lock_snapshot(snapshot)
9548        .ok()
9549        .and_then(|mut state| state.deliberate_severance.take());
9550    if marker.is_some() && marker == exited_identity {
9551        exit_report.kind = ExitKind::DeliberateSeverance;
9552    }
9553    exit_report
9554}
9555
9556fn classify_reaped_child_exit(
9557    snapshot: &SharedSnapshot,
9558    child: &SupervisedChild,
9559    status: &ExitStatus,
9560) -> ExitReport {
9561    let _ = update_snapshot(snapshot, None, |state| {
9562        state.reaped_pid = Some(child.pid);
9563        state.spawn_failure = child.spawn_failure.clone();
9564    });
9565    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9566}
9567
9568fn fail_snapshot(
9569    snapshot: &SharedSnapshot,
9570    module_id: Option<&str>,
9571    last_exit: Option<ExitReport>,
9572) {
9573    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9574        state.state = ModuleState::Failed;
9575        clear_current_process_facts(state);
9576        if let Some(last_exit) = last_exit {
9577            state.last_exit = Some(last_exit);
9578        }
9579    }) {
9580        error!(error = %err, "failed to mark supervisor state failed");
9581    }
9582}
9583
9584fn update_snapshot(
9585    snapshot: &SharedSnapshot,
9586    module_id: Option<&str>,
9587    update: impl FnOnce(&mut SupervisorSnapshot),
9588) -> Result<(), SuperviseError> {
9589    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9590        module_id: module_id.map(ToOwned::to_owned),
9591    })?;
9592    update(&mut state);
9593    Ok(())
9594}
9595
9596const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9597
9598fn lock_snapshot_for_control<'a>(
9599    snapshot: &'a SharedSnapshot,
9600    module_id: &str,
9601    caller: &'static str,
9602) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9603    let started_at = Instant::now();
9604    let guard = lock_snapshot(snapshot)?;
9605    let waited = started_at.elapsed();
9606    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9607        warn!(
9608            module_id = %module_id,
9609            waited_ms = waited.as_millis() as u64,
9610            caller = %caller,
9611            "slow snapshot lock"
9612        );
9613    }
9614    Ok(guard)
9615}
9616
9617fn lock_snapshot(
9618    snapshot: &SharedSnapshot,
9619) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9620    snapshot
9621        .lock()
9622        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9623}
9624
9625#[cfg(test)]
9626mod terminal_history_tests {
9627    use std::{
9628        path::PathBuf,
9629        sync::Arc,
9630        time::{Duration, Instant},
9631    };
9632
9633    use tokio::{sync::mpsc, time::sleep};
9634
9635    use super::{
9636        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9637        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9638        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9639        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9640        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9641        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedConfiguration,
9642        SupervisedModule, SupervisedModuleInner, Supervisor, SupervisorHandle,
9643        SupervisorHealthStatus, SupervisorSnapshot,
9644    };
9645    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9646    // use for their own wall-clock deadlines: crash-restart instants must be on
9647    // the same clock the production code stamps them with, which is tokio's (and
9648    // is what `start_paused` tests can move).
9649    use super::Instant as ClockInstant;
9650    use crate::{
9651        registry::Registry,
9652        terminal_ring::{TerminalRing, TerminalRingConfig},
9653    };
9654    use std::sync::Mutex;
9655    use subc_control::TerminalDisposition;
9656
9657    /// See the twin in `control.rs` for why this derives the path from
9658    /// `current_exe()` and why the existence check is here: `--lib` alone does
9659    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9660    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9661    pub(super) fn fake_aft_stub_path() -> PathBuf {
9662        let mut path = std::env::current_exe().expect("current_exe available in tests");
9663        path.pop();
9664        path.pop();
9665        path.push(if cfg!(windows) {
9666            "fake-aft-stub.exe"
9667        } else {
9668            "fake-aft-stub"
9669        });
9670        assert!(
9671            path.exists(),
9672            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9673             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9674            path.display()
9675        );
9676        path
9677    }
9678
9679    #[test]
9680    fn reserved_never_spawned_refuses_every_hello() {
9681        // The canary hole: a reserved id whose module has never spawned had NO
9682        // gate entry and admitted anyone -- the reservation protected the nonce
9683        // holder, not the NAME. Now the entry is present with no legitimate
9684        // holder and refuses all comers.
9685        let supervisor = SupervisorHandle::default();
9686        supervisor.apply_identity_configuration(&ModuleSpec {
9687            module_id: "never-spawned".to_string(),
9688            program: PathBuf::from("/usr/bin/false"),
9689            args: Vec::new(),
9690            env: Vec::new(),
9691            reserved: true,
9692            reserved_prefixes: Vec::new(),
9693            protocol: ModuleProtocol::Subc,
9694            overlap: Default::default(),
9695        });
9696        assert!(
9697            supervisor
9698                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9699                .is_some(),
9700            "forged nonce must refuse on a reserved never-spawned id"
9701        );
9702        assert!(
9703            supervisor
9704                .reserved_hello_rejection("never-spawned", None)
9705                .is_some(),
9706            "absent nonce must refuse on a reserved never-spawned id"
9707        );
9708        // And a real spawn nonce minted later admits exactly that nonce.
9709        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9710        supervisor.apply_identity_configuration(&ModuleSpec {
9711            module_id: "never-spawned".to_string(),
9712            program: PathBuf::from("/usr/bin/false"),
9713            args: Vec::new(),
9714            env: Vec::new(),
9715            reserved: true,
9716            reserved_prefixes: Vec::new(),
9717            protocol: ModuleProtocol::Subc,
9718            overlap: Default::default(),
9719        });
9720        assert!(supervisor
9721            .reserved_hello_rejection("never-spawned", Some("minted"))
9722            .is_none());
9723        assert!(supervisor
9724            .reserved_hello_rejection("never-spawned", Some("forged"))
9725            .is_some());
9726    }
9727
9728    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9729    /// happened, which is what "spent budget" looks like to every reader.
9730    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9731        let now = ClockInstant::now();
9732        for _ in 0..count {
9733            state.crash_restarts.push_back(now);
9734        }
9735    }
9736
9737    /// Age the oldest recorded restart out of `window`, standing in for the hours
9738    /// that would otherwise have to pass. Injecting the instant is the point: a
9739    /// test that slept a real window would take ten minutes and still prove less.
9740    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9741        let aged = state
9742            .crash_restarts
9743            .front()
9744            .expect("a crash restart must be recorded before it can be aged")
9745            .checked_sub(window + Duration::from_secs(1))
9746            .expect("the test clock is far enough from its origin to age an instant");
9747        state.crash_restarts[0] = aged;
9748    }
9749
9750    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9751        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9752        seed_crash_restarts(&mut state, count);
9753        state
9754    }
9755
9756    #[test]
9757    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9758        let policy = RestartPolicy::new(3, Duration::ZERO);
9759        let now = ClockInstant::now();
9760        assert!(daemon_will_restart(
9761            &mut snapshot_with_restarts(true, 2),
9762            &policy,
9763            now
9764        ));
9765        assert!(!daemon_will_restart(
9766            &mut snapshot_with_restarts(true, 3),
9767            &policy,
9768            now
9769        ));
9770        assert!(!daemon_will_restart(
9771            &mut snapshot_with_restarts(false, 0),
9772            &policy,
9773            now
9774        ));
9775    }
9776
9777    #[test]
9778    fn crash_restart_backoff_escalates_with_in_window_count() {
9779        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9780            .with_max_backoff(Duration::from_secs(30));
9781        let now = ClockInstant::now();
9782        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9783        let schedules = (0..4)
9784            .map(|_| {
9785                state
9786                    .next_crash_restart(&policy, now)
9787                    .expect("the test policy allows four crash restarts")
9788            })
9789            .collect::<Vec<_>>();
9790
9791        assert_eq!(
9792            schedules
9793                .iter()
9794                .map(|schedule| schedule.restart_in_window)
9795                .collect::<Vec<_>>(),
9796            vec![0, 1, 2, 3]
9797        );
9798        assert_eq!(
9799            schedules
9800                .iter()
9801                .map(|schedule| schedule.delay)
9802                .collect::<Vec<_>>(),
9803            vec![
9804                Duration::from_millis(100),
9805                Duration::from_secs(1),
9806                Duration::from_secs(10),
9807                Duration::from_secs(30),
9808            ]
9809        );
9810    }
9811
9812    #[test]
9813    fn crash_restart_backoff_resets_after_ring_clear() {
9814        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9815        let now = ClockInstant::now();
9816        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9817        assert_eq!(
9818            state.next_crash_restart(&policy, now).unwrap().delay,
9819            Duration::from_millis(100)
9820        );
9821        assert_eq!(
9822            state.next_crash_restart(&policy, now).unwrap().delay,
9823            Duration::from_secs(1)
9824        );
9825
9826        state.clear_crash_restarts();
9827        let schedule = state
9828            .next_crash_restart(&policy, now)
9829            .expect("a cleared ring must allow another restart");
9830        assert_eq!(schedule.restart_in_window, 0);
9831        assert_eq!(schedule.delay, Duration::from_millis(100));
9832    }
9833
9834    #[test]
9835    fn crash_restart_backoff_ignores_aged_restarts() {
9836        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9837        let now = ClockInstant::now();
9838        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9839        state
9840            .next_crash_restart(&policy, now)
9841            .expect("the first restart is allowed");
9842        state
9843            .next_crash_restart(&policy, now)
9844            .expect("the second restart is allowed");
9845        state.crash_restarts[0] = now
9846            .checked_sub(policy.window + Duration::from_secs(1))
9847            .expect("the fake clock can age a restart past the window");
9848
9849        let schedule = state
9850            .next_crash_restart(&policy, now)
9851            .expect("an aged restart must release its slot");
9852        assert_eq!(schedule.restart_in_window, 1);
9853        assert_eq!(schedule.delay, Duration::from_secs(1));
9854        assert_eq!(state.crash_restarts.len(), 2);
9855    }
9856
9857    /// The budget is a rate: the same three spent restarts refuse a respawn
9858    /// while they are recent and allow one once they have aged past the window.
9859    /// Nothing about the module changed in between, which is the whole point.
9860    #[test]
9861    fn a_budget_spent_before_the_window_no_longer_refuses() {
9862        let policy = RestartPolicy::new(3, Duration::ZERO);
9863        let mut state = snapshot_with_restarts(true, 3);
9864        let now = ClockInstant::now();
9865        assert!(!daemon_will_restart(&mut state, &policy, now));
9866
9867        assert!(daemon_will_restart(
9868            &mut state,
9869            &policy,
9870            now + policy.window + Duration::from_secs(1)
9871        ));
9872        assert!(
9873            state.crash_restarts.is_empty(),
9874            "reading the budget must drop the instants that left the window"
9875        );
9876    }
9877
9878    fn module_with_recovery_snapshot(
9879        state: ModuleState,
9880        enabled: bool,
9881        restart_count: u32,
9882    ) -> SupervisedModule {
9883        let registry = Arc::new(Registry::default());
9884        let supervisor =
9885            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9886        let runtime = supervisor.runtime_config();
9887        let spec = ModuleSpec {
9888            module_id: "recovery-snapshot".to_string(),
9889            program: fake_aft_stub_path(),
9890            args: Vec::new(),
9891            env: Vec::new(),
9892            reserved: false,
9893            reserved_prefixes: Vec::new(),
9894            protocol: ModuleProtocol::Subc,
9895            overlap: Default::default(),
9896        };
9897        let mut snapshot = SupervisorSnapshot::new(state, enabled);
9898        seed_crash_restarts(&mut snapshot, restart_count);
9899        // These tests read synthetic snapshots. A real child and monitor would
9900        // race those reads by replacing the requested state during startup.
9901        let (commands, _rx) = mpsc::channel(4);
9902        SupervisedModule {
9903            inner: Arc::new(SupervisedModuleInner {
9904                module_id: spec.module_id.clone(),
9905                registry,
9906                snapshot: Arc::new(Mutex::new(snapshot)),
9907                configuration: Arc::new(Mutex::new(SupervisedConfiguration {
9908                    spec,
9909                    health: runtime.health,
9910                })),
9911                stderr_ring: runtime.stderr_ring,
9912                terminal_ring: runtime.terminal_ring,
9913                commands,
9914                monitor: Mutex::new(None),
9915                restart_policy: runtime.restart_policy,
9916                effective_drain_timeout: runtime.effective_drain_timeout,
9917                provenance_probe: supervisor.provenance_probe.clone(),
9918            }),
9919        }
9920    }
9921
9922    #[cfg(target_os = "linux")]
9923    #[tokio::test]
9924    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9925        let supervisor =
9926            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
9927                .with_cgroup_placement(None);
9928        let result = supervisor.spawn(ModuleSpec {
9929            module_id: "no-cgroup-placement".to_string(),
9930            program: fake_aft_stub_path(),
9931            args: Vec::new(),
9932            env: Vec::new(),
9933            reserved: false,
9934            reserved_prefixes: Vec::new(),
9935            protocol: ModuleProtocol::Subc,
9936            overlap: Default::default(),
9937        });
9938
9939        assert!(
9940            result.is_ok(),
9941            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9942        );
9943    }
9944
9945    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9946    async fn undecided_snapshot_uses_shared_restart_predicate() {
9947        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9948            .will_recover_after_connection_loss()
9949            .unwrap());
9950        assert!(
9951            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9952                .will_recover_after_connection_loss()
9953                .unwrap()
9954        );
9955    }
9956
9957    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9958    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9959        assert!(
9960            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9961                .will_recover_after_connection_loss()
9962                .unwrap()
9963        );
9964    }
9965
9966    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9967    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9968        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9969            .will_recover_after_connection_loss()
9970            .unwrap());
9971        assert!(
9972            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9973                .will_recover_after_connection_loss()
9974                .unwrap()
9975        );
9976    }
9977
9978    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9979    async fn warming_snapshot_is_limited_to_startup_phases() {
9980        for state in [
9981            ModuleState::Starting,
9982            ModuleState::Running,
9983            ModuleState::Restarting,
9984        ] {
9985            assert!(
9986                module_with_recovery_snapshot(state, true, 0)
9987                    .is_warming()
9988                    .unwrap(),
9989                "{state:?} should be warming"
9990            );
9991        }
9992        for state in [
9993            ModuleState::Unresponsive,
9994            ModuleState::Draining,
9995            ModuleState::Stopped,
9996            ModuleState::Failed,
9997            ModuleState::Disabled,
9998        ] {
9999            assert!(
10000                !module_with_recovery_snapshot(state, true, 0)
10001                    .is_warming()
10002                    .unwrap(),
10003                "{state:?} should not be warming"
10004            );
10005        }
10006    }
10007
10008    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10009    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
10010        let registry = Arc::new(Registry::default());
10011        let supervisor =
10012            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
10013        let module = supervisor
10014            .spawn(ModuleSpec {
10015                module_id: "terminal-history".to_string(),
10016                program: fake_aft_stub_path(),
10017                args: Vec::new(),
10018                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10019                reserved: false,
10020                reserved_prefixes: Vec::new(),
10021                protocol: ModuleProtocol::Subc,
10022                overlap: Default::default(),
10023            })
10024            .unwrap();
10025
10026        let deadline = Instant::now() + Duration::from_secs(5);
10027        loop {
10028            let history = module.terminal_history();
10029            if history.entries.len() == 2 {
10030                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
10031                assert_eq!(history.dropped, 0);
10032                assert_eq!(
10033                    history
10034                        .entries
10035                        .iter()
10036                        .map(|entry| entry.exit_code)
10037                        .collect::<Vec<_>>(),
10038                    vec![Some(23), Some(23)]
10039                );
10040                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
10041                return;
10042            }
10043            assert!(
10044                Instant::now() < deadline,
10045                "module did not retain two terminal exits: {history:?}"
10046            );
10047            sleep(Duration::from_millis(10)).await;
10048        }
10049    }
10050
10051    /// A disable issued while a crash respawn is still backing off must preempt
10052    /// that respawn: the operator's stop wins, the disable must not queue behind
10053    /// the backoff, and the module must never come back up afterwards.
10054    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10055    async fn disable_during_crash_backoff_cancels_pending_respawn() {
10056        let backoff = Duration::from_secs(2);
10057        let supervisor = Supervisor::new_for_test(
10058            Arc::new(Registry::default()),
10059            RestartPolicy::new(10, backoff),
10060        );
10061        let module = supervisor
10062            .spawn(ModuleSpec {
10063                module_id: "disable-during-backoff".to_string(),
10064                program: fake_aft_stub_path(),
10065                args: Vec::new(),
10066                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10067                reserved: false,
10068                reserved_prefixes: Vec::new(),
10069                protocol: ModuleProtocol::Subc,
10070                overlap: Default::default(),
10071            })
10072            .unwrap();
10073
10074        // Wait for the first crash to put the module into its backoff window.
10075        let deadline = Instant::now() + Duration::from_secs(5);
10076        loop {
10077            if module.status().unwrap().state == ModuleState::Restarting {
10078                break;
10079            }
10080            assert!(
10081                Instant::now() < deadline,
10082                "module never entered the crash backoff"
10083            );
10084            sleep(Duration::from_millis(10)).await;
10085        }
10086
10087        let started = Instant::now();
10088        module.set_enabled(false).await.unwrap();
10089        let waited = started.elapsed();
10090
10091        assert!(
10092            waited < backoff / 2,
10093            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10094        );
10095        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10096
10097        // Outlast the backoff: the respawn it was counting down to must never run.
10098        sleep(backoff + Duration::from_millis(500)).await;
10099        let status = module.status().unwrap();
10100        assert_eq!(status.state, ModuleState::Disabled);
10101        assert_eq!(
10102            status.spawn_generation, 1,
10103            "module respawned after the operator disabled it"
10104        );
10105    }
10106
10107    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
10108    /// the shape of nats-server, the program this rule exists for.
10109    #[cfg(unix)]
10110    fn protocol_none_sigterm_exits_clean_spec(
10111        module_id: &str,
10112        dir: &std::path::Path,
10113    ) -> (ModuleSpec, PathBuf, PathBuf) {
10114        let ready = dir.join("ready");
10115        let marker = dir.join("sigterm");
10116        let spec = ModuleSpec {
10117            module_id: module_id.to_string(),
10118            program: fake_aft_stub_path(),
10119            args: Vec::new(),
10120            env: vec![
10121                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10122                (
10123                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10124                    marker.display().to_string(),
10125                ),
10126                (
10127                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10128                    ready.display().to_string(),
10129                ),
10130            ],
10131            reserved: false,
10132            reserved_prefixes: Vec::new(),
10133            protocol: ModuleProtocol::None,
10134            overlap: Default::default(),
10135        };
10136        (spec, ready, marker)
10137    }
10138
10139    /// Wait for a file the child writes, so a signal is never sent before the
10140    /// child's SIGTERM handler is installed (the default disposition would
10141    /// kill it by signal and the exit would not be clean).
10142    #[cfg(unix)]
10143    async fn wait_for_file(path: &std::path::Path) {
10144        let deadline = Instant::now() + Duration::from_secs(10);
10145        while !path.exists() {
10146            assert!(
10147                Instant::now() < deadline,
10148                "{} never appeared",
10149                path.display()
10150            );
10151            sleep(Duration::from_millis(10)).await;
10152        }
10153    }
10154
10155    /// A protocol-none module that exits 0 because something OUTSIDE the
10156    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10157    /// the crash-path disposition rather than `stopped`.
10158    #[cfg(unix)]
10159    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10160    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10161        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10162        let (spec, ready, marker) =
10163            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10164        let supervisor = Supervisor::new_for_test(
10165            Arc::new(Registry::default()),
10166            RestartPolicy::new(3, Duration::ZERO),
10167        );
10168        let module = supervisor.spawn(spec).unwrap();
10169        wait_for_file(&ready).await;
10170        // The ready file proves the child installed its SIGTERM handler, not
10171        // that the supervisor has processed the privacy trampoline's exec
10172        // acknowledgement. On macOS status withholds the pid until then.
10173        let deadline = Instant::now() + Duration::from_secs(10);
10174        let first_pid = loop {
10175            let status = module.status().unwrap();
10176            if status.state == ModuleState::Running {
10177                if let Some(pid) = status.pid {
10178                    break pid;
10179                }
10180            }
10181            assert!(
10182                Instant::now() < deadline,
10183                "a running module must report its pid after exec confirmation: {status:?}"
10184            );
10185            sleep(Duration::from_millis(10)).await;
10186        };
10187
10188        rustix::process::kill_process(
10189            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10190            rustix::process::Signal::TERM,
10191        )
10192        .unwrap();
10193
10194        let deadline = Instant::now() + Duration::from_secs(10);
10195        let respawned = loop {
10196            let status = module.status().unwrap();
10197            if status.state == ModuleState::Running
10198                && status.pid.is_some_and(|pid| pid != first_pid)
10199            {
10200                break status;
10201            }
10202            assert!(
10203                Instant::now() < deadline,
10204                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10205            );
10206            sleep(Duration::from_millis(10)).await;
10207        };
10208        assert_eq!(respawned.spawn_generation, 2);
10209        assert!(
10210            marker.exists(),
10211            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10212        );
10213
10214        let history = module.terminal_history();
10215        assert_eq!(history.entries.len(), 1, "{history:?}");
10216        let entry = &history.entries[0];
10217        assert_eq!(entry.exit_code, Some(0));
10218        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10219        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10220
10221        module.stop().await.unwrap();
10222    }
10223
10224    /// Repeated unrequested clean exits of a protocol-none module spend the
10225    /// restart budget exactly as crashes do, and the module ends `failed` with
10226    /// the budget named.
10227    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10228    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10229        let supervisor = Supervisor::new_for_test(
10230            Arc::new(Registry::default()),
10231            RestartPolicy::new(1, Duration::ZERO),
10232        );
10233        let module = supervisor
10234            .spawn(ModuleSpec {
10235                module_id: "none-clean-exit-budget".to_string(),
10236                program: fake_aft_stub_path(),
10237                args: Vec::new(),
10238                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10239                reserved: false,
10240                reserved_prefixes: Vec::new(),
10241                protocol: ModuleProtocol::None,
10242                overlap: Default::default(),
10243            })
10244            .unwrap();
10245
10246        // Failed follows the terminal write; this deadline only bounds a hang,
10247        // not an assumed duration for the two launches or their exit recording.
10248        let deadline = Instant::now() + Duration::from_secs(10);
10249        loop {
10250            let status = module.status().unwrap();
10251            if status.state == ModuleState::Failed {
10252                break;
10253            }
10254            assert!(
10255                Instant::now() < deadline,
10256                "module never exhausted its budget: {status:?} {:?}",
10257                module.terminal_history()
10258            );
10259            sleep(Duration::from_millis(10)).await;
10260        }
10261        let history = module.terminal_history();
10262        assert_eq!(
10263            history
10264                .entries
10265                .iter()
10266                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10267                .collect::<Vec<_>>(),
10268            vec![
10269                (Some(0), TerminalDisposition::Restarting),
10270                (Some(0), TerminalDisposition::Failed),
10271            ]
10272        );
10273        let detail = history.entries[1]
10274            .disposition_detail
10275            .as_deref()
10276            .expect("a budget failure names the budget");
10277        assert!(detail.contains("max_restarts=1"), "{detail}");
10278        assert_eq!(module.status().unwrap().spawn_generation, 2);
10279    }
10280
10281    #[test]
10282    fn restart_budget_failure_is_published_after_its_terminal_record() {
10283        let supervisor = Supervisor::new_for_test(
10284            Arc::new(Registry::default()),
10285            RestartPolicy::new(0, Duration::ZERO),
10286        );
10287        let runtime = supervisor.runtime_config();
10288        let spec = ModuleSpec {
10289            protocol: ModuleProtocol::None,
10290            ..windowed_crash_spec("budget-publication-order")
10291        };
10292        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
10293            ModuleState::Running,
10294            true,
10295        )));
10296        let reaped_snapshot = snapshot.clone();
10297        let events = runtime.spawn_events.clone();
10298        let history = runtime.terminal_ring.clone();
10299
10300        // Exit recording takes the event-feed lock before the history lock.
10301        // Holding it pauses the writer after choosing a disposition but before
10302        // recording history, without assuming anything about scheduler timing.
10303        let before_record = events.0.lock().unwrap();
10304        let reap = std::thread::spawn(move || {
10305            tokio::runtime::Builder::new_current_thread()
10306                .enable_all()
10307                .build()
10308                .unwrap()
10309                .block_on(on_child_exit(
10310                    &spec,
10311                    runtime.restart_policy,
10312                    &supervisor.registry,
10313                    &reaped_snapshot,
10314                    &runtime.terminal_ring,
10315                    &runtime.spawn_events,
10316                    &runtime.child_roster,
10317                    ExitReport {
10318                        kind: ExitKind::Clean,
10319                        code: Some(0),
10320                        signal: None,
10321                        at_ms: 1,
10322                    },
10323                ))
10324        });
10325        // The deadline bounds a hung writer only; last_exit is the handshake.
10326        let deadline = Instant::now() + Duration::from_secs(10);
10327        let before_state = loop {
10328            let state = lock_snapshot(&snapshot).unwrap();
10329            if state.last_exit.is_some() {
10330                break state.state;
10331            }
10332            drop(state);
10333            assert!(Instant::now() < deadline, "exit decision was not reached");
10334            std::thread::yield_now();
10335        };
10336        let before_history = history.lock().unwrap().snapshot();
10337        drop(before_record);
10338        assert!(matches!(reap.join().unwrap(), NextAction::Stop { .. }));
10339        assert!(before_history.entries.is_empty());
10340        assert_ne!(
10341            before_state,
10342            ModuleState::Failed,
10343            "Failed was visible before its terminal record could be written"
10344        );
10345        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10346        let history = history.lock().unwrap().snapshot();
10347        assert_eq!(history.entries.len(), 1);
10348        assert_eq!(history.entries[0].exit_code, Some(0));
10349        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10350    }
10351
10352    #[tokio::test]
10353    async fn restart_budget_failure_remains_failed_when_journal_append_fails() {
10354        let dir = subc_test_support::TestTempDir::new("budget-journal-failure");
10355        let path = dir.join("terminals.jsonl");
10356        std::fs::create_dir(&path).unwrap();
10357        let supervisor = Supervisor::new_for_test(
10358            Arc::new(Registry::default()),
10359            RestartPolicy::new(0, Duration::ZERO),
10360        )
10361        .with_terminal_journal(path, "budget-journal-failure".into());
10362        let runtime = supervisor.runtime_config();
10363        let spec = windowed_crash_spec("budget-journal-failure");
10364        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10365        assert!(matches!(
10366            on_child_exit(
10367                &spec,
10368                runtime.restart_policy,
10369                &supervisor.registry,
10370                &snapshot,
10371                &runtime.terminal_ring,
10372                &runtime.spawn_events,
10373                &runtime.child_roster,
10374                crash_exit_report(1),
10375            )
10376            .await,
10377            NextAction::Stop { .. }
10378        ));
10379        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10380        let history = runtime
10381            .terminal_ring
10382            .lock()
10383            .unwrap()
10384            .durable_history(&spec.module_id);
10385        assert!(history.journal_write_failures > 0);
10386        assert_eq!(history.entries.len(), 1);
10387        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10388    }
10389
10390    #[test]
10391    fn restart_budget_failure_remains_failed_when_exit_recording_panics() {
10392        let supervisor = Supervisor::new_for_test(
10393            Arc::new(Registry::default()),
10394            RestartPolicy::new(0, Duration::ZERO),
10395        );
10396        let runtime = supervisor.runtime_config();
10397        let spec = windowed_crash_spec("budget-recording-panic");
10398        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10399        runtime.spawn_events.emit_spawned(&spec.module_id, 1, 1);
10400        // Exhausting the event sequence makes emit_exited panic before the
10401        // terminal write, exercising publication on the recording unwind.
10402        runtime.spawn_events.0.lock().unwrap().seq = u64::MAX;
10403        let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
10404            tokio::runtime::Builder::new_current_thread()
10405                .enable_all()
10406                .build()
10407                .unwrap()
10408                .block_on(on_child_exit(
10409                    &spec,
10410                    runtime.restart_policy,
10411                    &supervisor.registry,
10412                    &snapshot,
10413                    &runtime.terminal_ring,
10414                    &runtime.spawn_events,
10415                    &runtime.child_roster,
10416                    crash_exit_report(1),
10417                ))
10418        }));
10419        let panic = result.err().expect("recording must still unwind");
10420        assert_eq!(
10421            panic.downcast_ref::<String>().map(String::as_str),
10422            Some("spawn event sequence exhausted")
10423        );
10424        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10425    }
10426
10427    /// A stop the supervisor itself requests still stops a protocol-none
10428    /// module, even though the child answers the SIGTERM with exit 0.
10429    #[cfg(unix)]
10430    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10431    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10432        for disable in [false, true] {
10433            let label = if disable {
10434                "none-requested-disable"
10435            } else {
10436                "none-requested-stop"
10437            };
10438            let dir = subc_test_support::TestTempDir::new(label);
10439            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10440            let supervisor = Supervisor::new_for_test(
10441                Arc::new(Registry::default()),
10442                RestartPolicy::new(3, Duration::ZERO),
10443            );
10444            let module = supervisor.spawn(spec).unwrap();
10445            wait_for_file(&ready).await;
10446
10447            if disable {
10448                module.set_enabled(false).await.unwrap();
10449            } else {
10450                module.stop().await.unwrap();
10451            }
10452            assert!(
10453                marker.exists(),
10454                "{label}: the child must have left through its SIGTERM handler with exit 0"
10455            );
10456
10457            // Long enough for a zero-backoff respawn to have happened if the
10458            // exit had been treated as a crash.
10459            sleep(Duration::from_millis(500)).await;
10460            let status = module.status().unwrap();
10461            let expected = if disable {
10462                ModuleState::Disabled
10463            } else {
10464                ModuleState::Stopped
10465            };
10466            assert_eq!(status.state, expected, "{label}");
10467            assert_eq!(
10468                status.spawn_generation, 1,
10469                "{label}: respawned after a requested stop"
10470            );
10471            let history = module.terminal_history();
10472            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10473            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10474            assert_ne!(
10475                history.entries[0].disposition,
10476                TerminalDisposition::Restarting,
10477                "{label}"
10478            );
10479        }
10480    }
10481
10482    /// A subc-wire module that exits 0 on its own is still a stop: the
10483    /// protocol-none rule must not reach it.
10484    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10485    async fn subc_wire_clean_exit_is_still_a_stop() {
10486        let supervisor = Supervisor::new_for_test(
10487            Arc::new(Registry::default()),
10488            RestartPolicy::new(3, Duration::ZERO),
10489        );
10490        let module = supervisor
10491            .spawn(ModuleSpec {
10492                module_id: "wire-clean-exit".to_string(),
10493                program: fake_aft_stub_path(),
10494                args: Vec::new(),
10495                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10496                reserved: false,
10497                reserved_prefixes: Vec::new(),
10498                protocol: ModuleProtocol::Subc,
10499                overlap: Default::default(),
10500            })
10501            .unwrap();
10502
10503        let deadline = Instant::now() + Duration::from_secs(10);
10504        while module.terminal_history().entries.is_empty() {
10505            assert!(Instant::now() < deadline, "module never exited");
10506            sleep(Duration::from_millis(10)).await;
10507        }
10508        // Long enough for a zero-backoff respawn to have happened.
10509        sleep(Duration::from_millis(500)).await;
10510        let status = module.status().unwrap();
10511        assert_eq!(status.state, ModuleState::Stopped);
10512        assert_eq!(status.spawn_generation, 1);
10513        let history = module.terminal_history();
10514        assert_eq!(history.entries.len(), 1, "{history:?}");
10515        assert_eq!(history.entries[0].exit_code, Some(0));
10516        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10517    }
10518
10519    #[cfg(unix)]
10520    #[tokio::test]
10521    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10522        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10523        let record = dir.join("live-children.json");
10524        let supervisor = Supervisor::new_for_test(
10525            Arc::new(Registry::default()),
10526            RestartPolicy::new(0, Duration::ZERO),
10527        );
10528        let mut runtime = supervisor.runtime_config();
10529        runtime.child_roster.record_to(record.clone());
10530        let gate = Arc::new(super::ReloadExitRecordGate::default());
10531        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10532        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10533        let spec = ModuleSpec {
10534            module_id: "reload-exit-roster".into(),
10535            program: fake_aft_stub_path(),
10536            args: Vec::new(),
10537            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10538            reserved: false,
10539            reserved_prefixes: Vec::new(),
10540            protocol: ModuleProtocol::Subc,
10541            overlap: Default::default(),
10542        };
10543        let mut child = None;
10544        let reload = super::finish_reload_child(
10545            &spec,
10546            &runtime,
10547            &supervisor.registry,
10548            &supervisor.process_liveness,
10549            &snapshot,
10550            &mut child,
10551        );
10552        tokio::pin!(reload);
10553        tokio::select! {
10554            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10555            _ = gate.reached.notified() => {}
10556        }
10557        assert!(runtime
10558            .terminal_ring
10559            .lock()
10560            .unwrap()
10561            .snapshot()
10562            .entries
10563            .is_empty());
10564        assert_eq!(
10565            crate::live_children::read_record(&record).unwrap().len(),
10566            1,
10567            "shutdown must still wait for the reaped child until its terminal record exists"
10568        );
10569        runtime.child_roster.close();
10570        gate.resume.notify_one();
10571        assert!(reload.await.is_err());
10572        assert!(crate::live_children::read_record(&record)
10573            .unwrap()
10574            .is_empty());
10575        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10576        assert_eq!(history.entries.len(), 1);
10577        assert_eq!(
10578            history.entries[0].disposition,
10579            TerminalDisposition::DaemonShutdown
10580        );
10581    }
10582
10583    /// Each restart-producing arm has its own state transition. Keeping their
10584    /// lifetime count assertions adjacent prevents a later new arm from silently
10585    /// spending budget without recording the historical restart.
10586    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10587    async fn every_restart_increment_path_advances_lifetime_count() {
10588        let supervisor = Supervisor::new_for_test(
10589            Arc::new(Registry::default()),
10590            RestartPolicy::new(1, Duration::ZERO),
10591        );
10592        let runtime = supervisor.runtime_config();
10593        let spec = ModuleSpec {
10594            module_id: "lifetime-increment-path".to_string(),
10595            program: PathBuf::from("/unused/lifetime-increment-path"),
10596            args: Vec::new(),
10597            env: Vec::new(),
10598            reserved: false,
10599            reserved_prefixes: Vec::new(),
10600            protocol: ModuleProtocol::Subc,
10601            overlap: Default::default(),
10602        };
10603
10604        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10605        assert!(matches!(
10606            on_child_exit(
10607                &spec,
10608                runtime.restart_policy,
10609                &supervisor.registry,
10610                &crash_snapshot,
10611                &runtime.terminal_ring,
10612                &runtime.spawn_events,
10613                &runtime.child_roster,
10614                ExitReport {
10615                    kind: ExitKind::Crash,
10616                    code: Some(1),
10617                    signal: None,
10618                    at_ms: 1,
10619                },
10620            )
10621            .await,
10622            NextAction::Restart { schedule: _ }
10623        ));
10624        let (crash_restarts, crash_lifetime) = {
10625            let state = lock_snapshot(&crash_snapshot).unwrap();
10626            (state.crash_restarts.len(), state.lifetime_restarts)
10627        };
10628        assert_eq!(crash_restarts, 1);
10629        assert_eq!(crash_lifetime, 1);
10630
10631        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10632        let mut health_child = None;
10633        assert!(matches!(
10634            health_restart_child(
10635                &spec,
10636                &runtime,
10637                &supervisor.registry,
10638                &supervisor.process_liveness,
10639                &health_snapshot,
10640                &mut health_child,
10641                SupervisorHealthStatus::Failing,
10642                None,
10643                2,
10644            )
10645            .await,
10646            Ok(())
10647        ));
10648        assert!(health_child.is_none());
10649        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10650        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10651        let (health_restarts, health_lifetime) = {
10652            let state = lock_snapshot(&health_snapshot).unwrap();
10653            (state.crash_restarts.len(), state.lifetime_restarts)
10654        };
10655        assert_eq!(health_restarts, 1);
10656        assert_eq!(health_lifetime, 1);
10657
10658        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10659        let mut reload_child = None;
10660        assert!(matches!(
10661            handle_reload_spawn_failure(
10662                &spec,
10663                &runtime,
10664                &supervisor.process_liveness,
10665                &reload_snapshot,
10666                &mut reload_child,
10667                "forced reload spawn failure".to_string(),
10668            )
10669            .await,
10670            Err(SuperviseError::ReloadFailed { .. })
10671        ));
10672        let (reload_restarts, reload_lifetime) = {
10673            let state = lock_snapshot(&reload_snapshot).unwrap();
10674            (state.crash_restarts.len(), state.lifetime_restarts)
10675        };
10676        assert_eq!(reload_restarts, 1);
10677        assert_eq!(reload_lifetime, 1);
10678    }
10679
10680    #[tokio::test]
10681    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10682        let supervisor = Supervisor::new_for_test(
10683            Arc::new(Registry::default()),
10684            RestartPolicy::new(3, Duration::ZERO),
10685        );
10686        let runtime = supervisor.runtime_config();
10687        let spec = ModuleSpec {
10688            module_id: "deliberately-severed".to_string(),
10689            program: PathBuf::from("/unused/deliberately-severed"),
10690            args: Vec::new(),
10691            env: Vec::new(),
10692            reserved: false,
10693            reserved_prefixes: Vec::new(),
10694            protocol: ModuleProtocol::Subc,
10695            overlap: Default::default(),
10696        };
10697        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10698        let process = ProcessIdentity {
10699            pid: 41,
10700            start_time: 101,
10701        };
10702        record_deliberate_severance(&snapshot, process).unwrap();
10703        let exit_report = apply_deliberate_severance_marker(
10704            &snapshot,
10705            Some(process),
10706            ExitReport {
10707                kind: ExitKind::Crash,
10708                code: Some(1),
10709                signal: None,
10710                at_ms: 1,
10711            },
10712        );
10713        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10714
10715        assert!(matches!(
10716            on_child_exit(
10717                &spec,
10718                runtime.restart_policy,
10719                &supervisor.registry,
10720                &snapshot,
10721                &runtime.terminal_ring,
10722                &runtime.spawn_events,
10723                &runtime.child_roster,
10724                exit_report,
10725            )
10726            .await,
10727            NextAction::Restart { schedule: _ }
10728        ));
10729        let state = lock_snapshot(&snapshot).unwrap();
10730        assert_eq!(state.lifetime_restarts, 1);
10731        assert_eq!(state.crash_restarts.len(), 0);
10732    }
10733
10734    #[tokio::test]
10735    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10736        let supervisor = Supervisor::new_for_test(
10737            Arc::new(Registry::default()),
10738            RestartPolicy::new(3, Duration::ZERO),
10739        );
10740        let runtime = supervisor.runtime_config();
10741        let spec = ModuleSpec {
10742            module_id: "genuine-crash".to_string(),
10743            program: PathBuf::from("/unused/genuine-crash"),
10744            args: Vec::new(),
10745            env: Vec::new(),
10746            reserved: false,
10747            reserved_prefixes: Vec::new(),
10748            protocol: ModuleProtocol::Subc,
10749            overlap: Default::default(),
10750        };
10751        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10752
10753        assert!(matches!(
10754            on_child_exit(
10755                &spec,
10756                runtime.restart_policy,
10757                &supervisor.registry,
10758                &snapshot,
10759                &runtime.terminal_ring,
10760                &runtime.spawn_events,
10761                &runtime.child_roster,
10762                ExitReport {
10763                    kind: ExitKind::Crash,
10764                    code: Some(1),
10765                    signal: None,
10766                    at_ms: 1,
10767                },
10768            )
10769            .await,
10770            NextAction::Restart { schedule: _ }
10771        ));
10772        let state = lock_snapshot(&snapshot).unwrap();
10773        assert_eq!(state.lifetime_restarts, 1);
10774        assert_eq!(state.crash_restarts.len(), 1);
10775    }
10776
10777    fn crash_exit_report(at_ms: u64) -> ExitReport {
10778        ExitReport {
10779            kind: ExitKind::Crash,
10780            code: Some(1),
10781            signal: None,
10782            at_ms,
10783        }
10784    }
10785
10786    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10787        ModuleSpec {
10788            module_id: module_id.to_string(),
10789            program: PathBuf::from("/unused").join(module_id),
10790            args: Vec::new(),
10791            env: Vec::new(),
10792            reserved: false,
10793            reserved_prefixes: Vec::new(),
10794            protocol: ModuleProtocol::Subc,
10795            overlap: Default::default(),
10796        }
10797    }
10798
10799    /// A real crash loop still stops. Three crashes with nothing aging out spend
10800    /// a budget of two and the third respawn is refused, and both surfaces an
10801    /// operator has -- the log line and the retained terminal record -- name the
10802    /// window rather than only the cap, because `max_restarts=2` alone is what
10803    /// this budget used to mean.
10804    #[tokio::test]
10805    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10806        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10807        let supervisor = Supervisor::new_for_test(
10808            Arc::new(Registry::default()),
10809            RestartPolicy::new(2, Duration::ZERO),
10810        );
10811        let runtime = supervisor.runtime_config();
10812        let spec = windowed_crash_spec("crash-loop-in-window");
10813        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10814
10815        for attempt in 1..=2 {
10816            assert!(
10817                matches!(
10818                    on_child_exit(
10819                        &spec,
10820                        runtime.restart_policy,
10821                        &supervisor.registry,
10822                        &snapshot,
10823                        &runtime.terminal_ring,
10824                        &runtime.spawn_events,
10825                        &runtime.child_roster,
10826                        crash_exit_report(attempt),
10827                    )
10828                    .await,
10829                    NextAction::Restart { schedule: _ }
10830                ),
10831                "crash {attempt} is inside the budget and must respawn"
10832            );
10833        }
10834
10835        assert!(matches!(
10836            on_child_exit(
10837                &spec,
10838                runtime.restart_policy,
10839                &supervisor.registry,
10840                &snapshot,
10841                &runtime.terminal_ring,
10842                &runtime.spawn_events,
10843                &runtime.child_roster,
10844                crash_exit_report(3),
10845            )
10846            .await,
10847            NextAction::Stop { .. }
10848        ));
10849
10850        {
10851            let state = lock_snapshot(&snapshot).unwrap();
10852            assert_eq!(state.state, ModuleState::Failed);
10853            assert_eq!(state.crash_restarts.len(), 2);
10854            assert_eq!(state.lifetime_restarts, 2);
10855        }
10856
10857        let history = runtime
10858            .terminal_ring
10859            .lock()
10860            .expect("terminal ring is not poisoned")
10861            .snapshot();
10862        let last = history
10863            .entries
10864            .last()
10865            .expect("the refused crash is retained");
10866        assert_eq!(last.disposition, TerminalDisposition::Failed);
10867        assert_eq!(
10868            last.disposition_detail.as_deref(),
10869            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10870        );
10871
10872        let captured = crate::router::test_log::captured_logs(&logs);
10873        assert!(
10874            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10875            "the stop must be logged with its window: {captured}"
10876        );
10877    }
10878
10879    /// The rate, stated as a test: three crashes where the first has aged past
10880    /// the window are two crashes as far as the budget is concerned, so the
10881    /// third respawn is allowed and the ring holds only the two recent ones.
10882    ///
10883    /// This is the case a lifetime counter got wrong -- and the case the daemon
10884    /// now hits routinely, since a module exits non-zero every time its
10885    /// connection to the daemon drops.
10886    #[tokio::test]
10887    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10888        let supervisor = Supervisor::new_for_test(
10889            Arc::new(Registry::default()),
10890            RestartPolicy::new(2, Duration::ZERO),
10891        );
10892        let runtime = supervisor.runtime_config();
10893        let spec = windowed_crash_spec("crash-across-windows");
10894        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10895
10896        for attempt in 1..=2 {
10897            assert!(matches!(
10898                on_child_exit(
10899                    &spec,
10900                    runtime.restart_policy,
10901                    &supervisor.registry,
10902                    &snapshot,
10903                    &runtime.terminal_ring,
10904                    &runtime.spawn_events,
10905                    &runtime.child_roster,
10906                    crash_exit_report(attempt),
10907                )
10908                .await,
10909                NextAction::Restart { schedule: _ }
10910            ));
10911        }
10912
10913        // The oldest crash moves out of the window; nothing else about the
10914        // module changes.
10915        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10916            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10917        })
10918        .unwrap();
10919
10920        assert!(
10921            matches!(
10922                on_child_exit(
10923                    &spec,
10924                    runtime.restart_policy,
10925                    &supervisor.registry,
10926                    &snapshot,
10927                    &runtime.terminal_ring,
10928                    &runtime.spawn_events,
10929                    &runtime.child_roster,
10930                    crash_exit_report(3),
10931                )
10932                .await,
10933                NextAction::Restart { schedule: _ }
10934            ),
10935            "a crash older than the window must not hold a budget slot"
10936        );
10937
10938        let state = lock_snapshot(&snapshot).unwrap();
10939        assert_eq!(state.state, ModuleState::Restarting);
10940        assert_eq!(
10941            state.crash_restarts.len(),
10942            2,
10943            "the aged instant is dropped and the new one takes its place"
10944        );
10945        assert_eq!(
10946            state.lifetime_restarts, 3,
10947            "the ledger counts every restart, including the ones the window forgot"
10948        );
10949    }
10950
10951    /// An operator restart hands the budget back whole, and the ledger keeps
10952    /// counting. Those are different questions -- "how close is this module to
10953    /// being stopped" and "how many times has it been replaced" -- and the
10954    /// operator action answers only the first.
10955    #[tokio::test]
10956    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10957        let supervisor = Supervisor::new_for_test(
10958            Arc::new(Registry::default()),
10959            RestartPolicy::new(2, Duration::ZERO),
10960        );
10961        let runtime = supervisor.runtime_config();
10962        let spec = windowed_crash_spec("operator-cleared-budget");
10963        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10964
10965        for attempt in 1..=2 {
10966            assert!(matches!(
10967                on_child_exit(
10968                    &spec,
10969                    runtime.restart_policy,
10970                    &supervisor.registry,
10971                    &snapshot,
10972                    &runtime.terminal_ring,
10973                    &runtime.spawn_events,
10974                    &runtime.child_roster,
10975                    crash_exit_report(attempt),
10976                )
10977                .await,
10978                NextAction::Restart { schedule: _ }
10979            ));
10980        }
10981
10982        reset_restart_count(&snapshot, &spec.module_id).unwrap();
10983        {
10984            let state = lock_snapshot(&snapshot).unwrap();
10985            assert!(
10986                state.crash_restarts.is_empty(),
10987                "an operator restart returns the full budget"
10988            );
10989            assert_eq!(
10990                state.lifetime_restarts, 2,
10991                "clearing the budget must not unmake the crashes"
10992            );
10993        }
10994
10995        assert!(
10996            matches!(
10997                on_child_exit(
10998                    &spec,
10999                    runtime.restart_policy,
11000                    &supervisor.registry,
11001                    &snapshot,
11002                    &runtime.terminal_ring,
11003                    &runtime.spawn_events,
11004                    &runtime.child_roster,
11005                    crash_exit_report(3),
11006                )
11007                .await,
11008                NextAction::Restart { schedule: _ }
11009            ),
11010            "the cleared budget must be spendable again"
11011        );
11012        let state = lock_snapshot(&snapshot).unwrap();
11013        assert_eq!(state.crash_restarts.len(), 1);
11014        assert_eq!(state.lifetime_restarts, 3);
11015    }
11016
11017    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
11018    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
11019        let severed = ProcessIdentity {
11020            pid: 41,
11021            start_time: 101,
11022        };
11023        let successor = ProcessIdentity {
11024            pid: 41,
11025            start_time: 202,
11026        };
11027        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
11028        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
11029            state.pid = Some(successor.pid);
11030            state.process_start_time = Some(successor.start_time);
11031        })
11032        .unwrap();
11033        assert!(!module.record_deliberate_severance(severed).unwrap());
11034
11035        let exit_report = apply_deliberate_severance_marker(
11036            &module.inner.snapshot,
11037            Some(successor),
11038            ExitReport {
11039                kind: ExitKind::Crash,
11040                code: Some(1),
11041                signal: None,
11042                at_ms: 1,
11043            },
11044        );
11045
11046        assert_eq!(exit_report.kind, ExitKind::Crash);
11047    }
11048
11049    #[tokio::test]
11050    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
11051        let registry = Registry::default();
11052        let supervisor = Supervisor::new_for_test(
11053            Arc::new(Registry::default()),
11054            RestartPolicy::new(3, Duration::ZERO),
11055        );
11056        let runtime = supervisor.runtime_config();
11057        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11058        let spec = ModuleSpec {
11059            module_id: "drain-deliberate-severance".to_string(),
11060            program: fake_aft_stub_path(),
11061            args: Vec::new(),
11062            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11063            reserved: false,
11064            reserved_prefixes: Vec::new(),
11065            protocol: ModuleProtocol::Subc,
11066            overlap: Default::default(),
11067        };
11068        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11069        let process = ProcessIdentity {
11070            pid: 41,
11071            start_time: 101,
11072        };
11073        child.process_identity = Some(process);
11074        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
11075            state.pid = Some(process.pid);
11076            state.process_start_time = Some(process.start_time);
11077        })
11078        .unwrap();
11079        record_deliberate_severance(&snapshot, process).unwrap();
11080
11081        drain_child_to_state(
11082            &spec.module_id,
11083            spec.protocol,
11084            // The child exits on its own; no signal may change the exit this
11085            // test classifies.
11086            StopNotice::SentOverConnection,
11087            &registry,
11088            None,
11089            &snapshot,
11090            &runtime.terminal_ring,
11091            &runtime.spawn_events,
11092            child,
11093            Duration::from_secs(1),
11094            ModuleState::Stopped,
11095            Some(false),
11096        )
11097        .await
11098        .unwrap();
11099
11100        let state = lock_snapshot(&snapshot).unwrap();
11101        assert_eq!(
11102            state.last_exit.as_ref().map(|exit| exit.kind),
11103            Some(ExitKind::DeliberateSeverance)
11104        );
11105        assert_eq!(state.lifetime_restarts, 1);
11106        assert_eq!(state.crash_restarts.len(), 0);
11107        drop(state);
11108        let history = runtime.terminal_ring.lock().unwrap().snapshot();
11109        assert_eq!(
11110            history.entries[0].exit_kind,
11111            subc_control::TerminalExitKind::DeliberateSeverance
11112        );
11113    }
11114
11115    #[tokio::test]
11116    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
11117        let registry = Registry::default();
11118        let supervisor = Supervisor::new_for_test(
11119            Arc::new(Registry::default()),
11120            RestartPolicy::new(3, Duration::ZERO),
11121        );
11122        let runtime = supervisor.runtime_config();
11123        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11124        let spec = ModuleSpec {
11125            module_id: "ordinary-drain".to_string(),
11126            program: fake_aft_stub_path(),
11127            args: Vec::new(),
11128            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11129            reserved: false,
11130            reserved_prefixes: Vec::new(),
11131            protocol: ModuleProtocol::Subc,
11132            overlap: Default::default(),
11133        };
11134        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11135
11136        drain_child_to_state(
11137            &spec.module_id,
11138            spec.protocol,
11139            // The child exits on its own; no signal may change the exit this
11140            // test classifies.
11141            StopNotice::SentOverConnection,
11142            &registry,
11143            None,
11144            &snapshot,
11145            &runtime.terminal_ring,
11146            &runtime.spawn_events,
11147            child,
11148            Duration::from_secs(1),
11149            ModuleState::Stopped,
11150            Some(false),
11151        )
11152        .await
11153        .unwrap();
11154
11155        let state = lock_snapshot(&snapshot).unwrap();
11156        assert_eq!(
11157            state.last_exit.as_ref().map(|exit| exit.kind),
11158            Some(ExitKind::Crash)
11159        );
11160        assert_eq!(state.lifetime_restarts, 0);
11161        assert_eq!(state.crash_restarts.len(), 0);
11162    }
11163
11164    #[test]
11165    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
11166        // The server's generic fatal-routing branch only knows that the
11167        // connection failed; it does not know that the daemon deliberately
11168        // initiated a process-killing severance. Keep this seam explicit so a
11169        // future connection error path cannot silently reintroduce the stale
11170        // exemption that mislabels a later genuine crash.
11171        assert!(!include_str!("server.rs")
11172            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
11173    }
11174
11175    /// The `route.closed` `drained` value must be the quiescence wait's own
11176    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
11177    /// measurement at all and `false` is the one honest constant. This is the exact
11178    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
11179    /// on every return path, including the one that used to return early via `?`
11180    /// with `route.closing` already sent and no `route.closed` ever following.
11181    #[test]
11182    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
11183        assert!(drained_after_quiescence_wait(&Ok(true)));
11184        assert!(!drained_after_quiescence_wait(&Ok(false)));
11185        assert!(!drained_after_quiescence_wait(&Err(
11186            SuperviseError::StatePoisoned { module_id: None }
11187        )));
11188    }
11189
11190    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
11191    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
11192    /// already reaped out-of-band) still leaves a terminal record rather than none
11193    /// at all. Triggering the real `wait()` I/O error from an integration test would
11194    /// need a genuine already-reaped-child race, which is OS-specific and not
11195    /// something this suite attempts elsewhere; this test instead verifies the
11196    /// record produced for that arm end-to-end through the real `TerminalRing`, and
11197    /// the call site itself is verified by inspection to sit in that exact arm.
11198    #[test]
11199    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
11200        let ring = Arc::new(Mutex::new(TerminalRing::new(
11201            TerminalRingConfig::default(),
11202            0,
11203        )));
11204        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
11205
11206        let snapshot = ring.lock().unwrap().snapshot();
11207        assert_eq!(snapshot.entries.len(), 1);
11208        let entry = &snapshot.entries[0];
11209        assert_eq!(entry.exit_code, None);
11210        assert_eq!(entry.exit_signal, None);
11211        assert_eq!(entry.disposition, TerminalDisposition::Failed);
11212    }
11213
11214    #[test]
11215    fn wait_error_exit_path_preserves_spawn_event_density() {
11216        let feed = super::SpawnEventFeed::default();
11217        feed.configure_incarnation("wait-error-density".to_string());
11218        feed.emit_spawned("wait-error", 41, 1);
11219        let ring = Arc::new(Mutex::new(TerminalRing::new(
11220            TerminalRingConfig::default(),
11221            0,
11222        )));
11223
11224        record_wait_error_terminal("wait-error", &ring, &feed);
11225        feed.emit_spawned("after-wait-error", 42, 2);
11226
11227        let state = feed.0.lock().unwrap();
11228        let sequences = state
11229            .events
11230            .iter()
11231            .map(|event| event.cursor.seq)
11232            .collect::<Vec<_>>();
11233        assert_eq!(sequences, vec![1, 2, 3]);
11234        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11235        assert_eq!(state.events[1].exit_code, None);
11236        assert_eq!(state.events[1].exit_signal, None);
11237    }
11238
11239    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
11240    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
11241    /// not a clean exit it never actually observed.
11242    #[test]
11243    fn wait_error_exit_report_is_classified_as_a_crash() {
11244        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11245    }
11246}
11247
11248#[cfg(test)]
11249mod health_evidence_tests {
11250    use super::{HealthProbeError, HealthProbeEvidence};
11251    use std::collections::HashSet;
11252
11253    /// The evidential asymmetry, asserted rather than described.
11254    ///
11255    /// Exactly ONE observation is proof a module cannot serve, and the one that
11256    /// fires under CPU starvation is not it. Before the split, all fifteen
11257    /// construction sites collapsed into a single String, so a timeout carried the
11258    /// same weight as a dead lane -- which is how a healthy module was restarted
11259    /// three times in one day.
11260    #[test]
11261    fn only_a_dead_lane_is_proof_of_death() {
11262        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11263        // Three non-proof classes, each for a different reason: silence is
11264        // consistent with health, a bad answer proves the module ALIVE, and a
11265        // daemon-side fault never reached the module at all.
11266        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11267        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11268        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11269    }
11270
11271    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
11272    ///
11273    /// A shared label renders two different observations identically in the line an
11274    /// operator reads after an unexplained restart -- the exact confusion this
11275    /// change removes.
11276    #[test]
11277    fn every_evidence_class_has_a_distinct_label() {
11278        let labels = [
11279            HealthProbeError::lane_dead("").label(),
11280            HealthProbeError::no_answer("").label(),
11281            HealthProbeError::bad_answer("").label(),
11282            HealthProbeError::misconfigured("").label(),
11283        ];
11284        let unique: HashSet<_> = labels.iter().collect();
11285        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11286    }
11287
11288    /// The class is additional information, not a replacement.
11289    ///
11290    /// An operator needs both "this was silence" and the specific text saying how
11291    /// long we waited; a classification that swallowed the message would trade one
11292    /// missing distinction for another.
11293    #[test]
11294    fn classification_preserves_the_original_message() {
11295        let err = HealthProbeError::no_answer("module did not answer within 5s");
11296        assert_eq!(err.to_string(), "module did not answer within 5s");
11297        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11298    }
11299}
11300
11301#[cfg(test)]
11302mod health_tombstone_tests {
11303    use std::{path::PathBuf, sync::Arc, time::Duration};
11304
11305    use subc_protocol::{
11306        manifest::Concurrency,
11307        session::{HealthStatus, ModuleControlResponse},
11308    };
11309    use tokio::sync::mpsc;
11310
11311    use super::{
11312        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11313        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11314    };
11315    use crate::{
11316        control::ControlHandler,
11317        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11318        registry::{ConnectionId, Registry},
11319        router::FrameSink,
11320    };
11321
11322    struct ProbeHarness {
11323        spec: ModuleSpec,
11324        runtime: SupervisorRuntimeConfig,
11325        forwarding: Arc<ForwardingTable>,
11326        module_connection: ConnectionId,
11327        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11328        handler: ControlHandler,
11329        module: super::SupervisedModule,
11330    }
11331
11332    fn probe_harness() -> ProbeHarness {
11333        let registry = Arc::new(Registry::default());
11334        let forwarding = Arc::new(ForwardingTable::default());
11335        let supervisor_handle = super::SupervisorHandle::new();
11336        let health = HealthConfig {
11337            http: None,
11338            cadence: Duration::from_secs(30),
11339            deadline: Duration::from_secs(5),
11340            failure_threshold: 3,
11341            on_degraded: HealthAction::Report,
11342            on_failing: HealthAction::Report,
11343            critical: false,
11344        };
11345        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11346            .with_forwarding(Arc::clone(&forwarding))
11347            .with_handle(supervisor_handle.clone())
11348            .with_health_config(health);
11349        let spec = ModuleSpec {
11350            module_id: "late-health-module".to_string(),
11351            program: PathBuf::from("disabled-module"),
11352            args: Vec::new(),
11353            env: Vec::new(),
11354            reserved: false,
11355            reserved_prefixes: Vec::new(),
11356            protocol: ModuleProtocol::Subc,
11357            overlap: Default::default(),
11358        };
11359        let module = supervisor
11360            .supervise_configured(spec.clone(), false)
11361            .unwrap();
11362        let runtime = supervisor.runtime_config();
11363        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11364            .with_supervisor(supervisor_handle);
11365        let module_connection = ConnectionId::new(700);
11366        let (module_tx, module_rx) = mpsc::channel(8);
11367        forwarding
11368            .register_module_connection(
11369                module_connection,
11370                spec.module_id.clone(),
11371                subc_protocol::PROTOCOL_VERSION,
11372                Concurrency::ModuleManaged,
11373                FrameSink::new(module_tx),
11374            )
11375            .unwrap();
11376
11377        ProbeHarness {
11378            spec,
11379            runtime,
11380            forwarding,
11381            module_connection,
11382            module_rx,
11383            handler,
11384            module,
11385        }
11386    }
11387
11388    async fn finish_after(
11389        harness: &mut ProbeHarness,
11390        stall: Duration,
11391    ) -> ModuleControlRpcCompletion {
11392        assert!(stall > harness.runtime.health.deadline);
11393        let deadline = harness.runtime.health.deadline;
11394        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11395        let answer = async {
11396            let frame = harness.module_rx.recv().await.expect("health.check frame");
11397            tokio::time::advance(deadline).await;
11398            tokio::task::yield_now().await;
11399            tokio::time::advance(stall - deadline).await;
11400            harness
11401                .forwarding
11402                .complete_module_control_rpc(
11403                    harness.module_connection,
11404                    frame.header.corr,
11405                    Some("health.check"),
11406                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11407                        status: HealthStatus::Ok,
11408                        detail: None,
11409                        metrics: None,
11410                    }),
11411                )
11412                .unwrap()
11413        };
11414        let (probe_result, completion) = tokio::join!(probe, answer);
11415        let err = probe_result.expect_err("probe must miss its deadline");
11416        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11417        completion
11418    }
11419
11420    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11421        let deadline = harness.runtime.health.deadline;
11422        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11423        let exhaust_deadline = async {
11424            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11425            tokio::time::advance(deadline).await;
11426            tokio::task::yield_now().await;
11427        };
11428        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11429        let err = probe_result.expect_err("probe must miss its deadline");
11430        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11431    }
11432
11433    async fn run_probe_cycle(harness: &mut ProbeHarness, answer: bool) {
11434        let registry = Arc::clone(&harness.module.inner.registry);
11435        let snapshot = Arc::clone(&harness.module.inner.snapshot);
11436        let process_liveness = super::SupervisorProcessLiveness::default();
11437        let mut child = None;
11438        let cycle = super::run_health_probe_cycle(
11439            &harness.spec,
11440            &harness.runtime,
11441            &registry,
11442            &process_liveness,
11443            &snapshot,
11444            &mut child,
11445        );
11446        let peer = async {
11447            let frame = harness.module_rx.recv().await.expect("health.check frame");
11448            if answer {
11449                harness
11450                    .forwarding
11451                    .complete_module_control_rpc(
11452                        harness.module_connection,
11453                        frame.header.corr,
11454                        Some("health.check"),
11455                        ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11456                            status: HealthStatus::Ok,
11457                            detail: None,
11458                            metrics: Some(serde_json::json!({"ready": true})),
11459                        }),
11460                    )
11461                    .unwrap();
11462            } else {
11463                tokio::time::advance(harness.runtime.health.deadline).await;
11464                tokio::task::yield_now().await;
11465            }
11466        };
11467        tokio::join!(cycle, peer);
11468    }
11469
11470    #[tokio::test(start_paused = true)]
11471    async fn unanswered_probe_is_unknown_until_threshold_and_ok_report_recovers() {
11472        let mut harness = probe_harness();
11473        // Drive the probe cycle directly with an in-memory wire peer. Stop the
11474        // disabled module's monitor so only this test owns lifecycle transitions;
11475        // no OS process is launched, and a restart is observed at scheduling.
11476        let monitor = harness.module.inner.monitor.lock().unwrap().take().unwrap();
11477        monitor.abort();
11478        let _ = monitor.await;
11479        super::update_snapshot(&harness.module.inner.snapshot, None, |state| {
11480            state.enabled = true;
11481            state.state = super::ModuleState::Running;
11482            state.process_alive = true;
11483        })
11484        .unwrap();
11485
11486        run_probe_cycle(&mut harness, true).await;
11487        assert_eq!(
11488            harness.module.status().unwrap().health.status,
11489            super::SupervisorHealthStatus::Ok
11490        );
11491
11492        for failures in 1..harness.runtime.health.failure_threshold {
11493            run_probe_cycle(&mut harness, false).await;
11494            let status = harness.module.status().unwrap();
11495            assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11496            assert_eq!(status.health.consecutive_failures, failures);
11497            assert!(status.health.last_probe_ms.is_some());
11498            assert!(status.health.detail.unwrap().starts_with("[no-answer]"));
11499            assert!(status.health.metrics.is_none());
11500            assert_eq!(status.state, super::ModuleState::Running);
11501            assert!(status.process_alive);
11502            assert_eq!(status.restart_count, 0);
11503            assert_eq!(status.lifetime_restarts, 0);
11504            assert!(status.health.last_action.is_none());
11505            assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11506        }
11507
11508        run_probe_cycle(&mut harness, true).await;
11509        let recovered = harness.module.status().unwrap();
11510        assert_eq!(recovered.health.status, super::SupervisorHealthStatus::Ok);
11511        assert_eq!(recovered.health.consecutive_failures, 0);
11512        assert!(recovered.health.detail.is_none());
11513        assert_eq!(
11514            recovered.health.metrics,
11515            Some(serde_json::json!({"ready": true}))
11516        );
11517        assert_eq!(recovered.lifetime_restarts, 0);
11518
11519        for failures in 1..=harness.runtime.health.failure_threshold {
11520            run_probe_cycle(&mut harness, false).await;
11521            let status = harness.module.status().unwrap();
11522            assert_eq!(status.health.consecutive_failures, failures);
11523            if failures < harness.runtime.health.failure_threshold {
11524                assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11525                assert_eq!(status.state, super::ModuleState::Running);
11526                assert_eq!(status.lifetime_restarts, 0);
11527                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11528            } else {
11529                assert_eq!(
11530                    status.health.status,
11531                    super::SupervisorHealthStatus::Unresponsive
11532                );
11533                assert_eq!(status.state, super::ModuleState::Restarting);
11534                assert_eq!(status.restart_count, 1);
11535                assert_eq!(status.lifetime_restarts, 1);
11536                assert_eq!(status.health.last_action.as_deref(), Some("restart"));
11537                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_some());
11538            }
11539        }
11540    }
11541
11542    #[tokio::test(start_paused = true)]
11543    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
11544        let mut harness = probe_harness();
11545
11546        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
11547        let first_latency = match &first {
11548            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11549            other => panic!("late answer was not retained: {other:?}"),
11550        };
11551        assert!(harness.handler.observe_module_control_completion(first));
11552
11553        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
11554        let second_latency = match &second {
11555            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11556            other => panic!("late answer was not retained: {other:?}"),
11557        };
11558        assert!(harness.handler.observe_module_control_completion(second));
11559
11560        assert_eq!(first_latency, Duration::from_secs(8));
11561        assert_eq!(
11562            second_latency - first_latency,
11563            Duration::from_secs(3),
11564            "latency must grow linearly with the additional stall"
11565        );
11566        let health = harness.module.status().unwrap().health;
11567        assert_eq!(health.late_answer_count, 2);
11568        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
11569    }
11570
11571    /// A module that answers every probe late must never march to the kill
11572    /// threshold: the late answer proves it is alive, so it must clear the miss
11573    /// streak the timeout recorded. Without the reset, a CPU-starved module
11574    /// that serves every probe seconds past the deadline accumulates
11575    /// `consecutive_failures` to the threshold and is killed — the exact
11576    /// sequence from the 2026-08-14 aft disable, where the daemon logged
11577    /// "proves the module is alive" five times while counting five misses.
11578    #[tokio::test(start_paused = true)]
11579    async fn late_answer_clears_the_consecutive_failure_streak() {
11580        let mut harness = probe_harness();
11581
11582        // Timeout recorded first: the probe path saw no answer in time.
11583        time_out_without_answer(&mut harness).await;
11584        harness
11585            .module
11586            .record_health_probe_failure_for_test("[no-answer] test miss")
11587            .unwrap();
11588        assert_eq!(
11589            harness.module.status().unwrap().health.consecutive_failures,
11590            1,
11591            "precondition: the miss must be on the streak before the late answer"
11592        );
11593
11594        // The stalled reply then lands: proof of life.
11595        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
11596        assert!(matches!(
11597            late,
11598            ModuleControlRpcCompletion::LateHealthAnswer { .. }
11599        ));
11600        assert!(harness.handler.observe_module_control_completion(late));
11601
11602        let health = harness.module.status().unwrap().health;
11603        assert_eq!(
11604            health.consecutive_failures, 0,
11605            "a late answer is an answer: the streak must reset"
11606        );
11607        assert_eq!(health.late_answer_count, 1);
11608    }
11609
11610    #[tokio::test(start_paused = true)]
11611    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
11612        let mut harness = probe_harness();
11613
11614        for _ in 0..20 {
11615            time_out_without_answer(&mut harness).await;
11616            assert_eq!(
11617                harness.forwarding.health_probe_tombstone_count().unwrap(),
11618                1
11619            );
11620        }
11621    }
11622}
11623
11624#[cfg(test)]
11625mod child_env_tests {
11626    use super::{
11627        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
11628        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
11629        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
11630    };
11631    use std::{ffi::OsStr, path::PathBuf};
11632    use tokio::process::Command;
11633
11634    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
11635        ModuleSpec {
11636            module_id: "env-plan".to_string(),
11637            program: PathBuf::from("/nonexistent"),
11638            args: Vec::new(),
11639            env,
11640            reserved: false,
11641            reserved_prefixes: Vec::new(),
11642            protocol: ModuleProtocol::Subc,
11643            overlap: Default::default(),
11644        }
11645    }
11646
11647    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
11648    /// one still gets its own.
11649    ///
11650    /// This is the narrow goal `env_clear()` was reached for, and the reason the
11651    /// fix is `env_remove` rather than deleting the line: an operator's ambient
11652    /// filter silently becoming an unconfigured module's log level is a real
11653    /// defect, just a much smaller one than clearing the environment.
11654    ///
11655    /// Asserted on the command plan rather than a spawned child because proving
11656    /// the ABSENCE of an inherited variable needs the parent's environment
11657    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
11658    /// removal as `(key, None)`, which is exactly the distinction wanted: not
11659    /// "absent because nobody set it" but "explicitly unset for the child".
11660    #[test]
11661    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
11662        let mut command = Command::new("/nonexistent");
11663        apply_child_env(&mut command, &spec(Vec::new()));
11664        let removed = command
11665            .as_std()
11666            .get_envs()
11667            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
11668        assert!(
11669            removed,
11670            "ambient CK_LOG must be explicitly removed for an unconfigured module"
11671        );
11672
11673        let mut configured = Command::new("/nonexistent");
11674        apply_child_env(
11675            &mut configured,
11676            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
11677        );
11678        let effective = configured
11679            .as_std()
11680            .get_envs()
11681            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
11682            .last()
11683            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11684        assert_eq!(
11685            effective,
11686            Some(Some("debug".to_string())),
11687            "a module's configured CK_LOG must survive the ambient removal"
11688        );
11689    }
11690
11691    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
11692    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
11693    /// the same reason as the CK_LOG test above.
11694    ///
11695    /// The argument is the load-bearing half: a stock binary exits on an
11696    /// unknown flag before it listens, so with `--subc` appended the mode
11697    /// could not supervise the one process it exists for. Found by the first
11698    /// conformance run (nats-server: `flag provided but not defined: -subc`).
11699    #[test]
11700    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
11701        let connection_file = std::path::Path::new("/run/subc-connection.json");
11702        let handle = SupervisorHandle::new();
11703
11704        let mut none_spec = spec(Vec::new());
11705        none_spec.protocol = ModuleProtocol::None;
11706        let mut none = Command::new("/nonexistent");
11707        let none_handoff =
11708            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
11709                .expect("protocol-none spawn args apply");
11710        assert!(
11711            none_handoff.is_none(),
11712            "protocol:none spawn must not receive a nonce descriptor"
11713        );
11714        assert!(
11715            !none.as_std().get_envs().any(|(key, value)| key
11716                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
11717                && value.is_some()),
11718            "protocol:none spawn must not name a nonce descriptor"
11719        );
11720        let none_args: Vec<String> = none
11721            .as_std()
11722            .get_args()
11723            .map(|a| a.to_string_lossy().into_owned())
11724            .collect();
11725        assert!(
11726            !none_args.iter().any(|a| a == SUBC_ARG),
11727            "protocol:none argv must not carry --subc; got {none_args:?}"
11728        );
11729        let none_has_nonce = none
11730            .as_std()
11731            .get_envs()
11732            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
11733        assert!(
11734            !none_has_nonce,
11735            "protocol:none spawn must not receive a launch nonce"
11736        );
11737        let none_has_module_id = none
11738            .as_std()
11739            .get_envs()
11740            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
11741        assert!(
11742            none_has_module_id,
11743            "SUBC_MODULE_ID is inert and stays on every path"
11744        );
11745        assert!(
11746            handle.spawn_nonce(&none_spec.module_id).is_none(),
11747            "no nonce record for a process that will never present one"
11748        );
11749
11750        // Control: the subc-wire path is unchanged by the branch above.
11751        let wire_spec = spec(Vec::new());
11752        let mut wire = Command::new("/nonexistent");
11753        let wire_handoff =
11754            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
11755                .expect("subc-wire spawn args apply");
11756        let wire_fd_env = wire
11757            .as_std()
11758            .get_envs()
11759            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
11760            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11761        #[cfg(unix)]
11762        assert_eq!(
11763            wire_fd_env,
11764            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11765            "a subc-wire spawn names the pipe it will receive at descriptor 3"
11766        );
11767        #[cfg(not(unix))]
11768        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11769        let wire_args: Vec<String> = wire
11770            .as_std()
11771            .get_args()
11772            .map(|a| a.to_string_lossy().into_owned())
11773            .collect();
11774        assert_eq!(
11775            wire_args,
11776            vec![
11777                SUBC_ARG.to_string(),
11778                connection_file.to_string_lossy().into_owned()
11779            ],
11780            "a subc-wire spawn still carries --subc <path>"
11781        );
11782        assert_eq!(
11783            wire.as_std()
11784                .get_envs()
11785                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11786            !cfg!(unix),
11787            "only Windows supplies the environment nonce"
11788        );
11789        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11790    }
11791
11792    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
11793    /// spec tries to set it; only a swap candidate carries it.
11794    ///
11795    /// "Set it only on candidates" is not enough, because spawn applies the
11796    /// spec's env verbatim and the daemon's own environment is inherited: either
11797    /// could hand a plain restart the swap role, and a module reading it would
11798    /// warm on its long swap budget while callers wait. Asserted as an explicit
11799    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
11800    /// test above gives.
11801    #[test]
11802    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11803        let role = |command: &Command| {
11804            command
11805                .as_std()
11806                .get_envs()
11807                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11808                .last()
11809                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11810        };
11811        let forged = spec(vec![(
11812            SUBC_SPAWN_ROLE_ENV.to_string(),
11813            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11814        )]);
11815
11816        let mut plain = Command::new("/nonexistent");
11817        apply_child_env(&mut plain, &forged);
11818        apply_spawn_role(&mut plain, SpawnRole::Plain);
11819        assert_eq!(
11820            role(&plain),
11821            Some(None),
11822            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11823        );
11824
11825        let mut candidate = Command::new("/nonexistent");
11826        apply_child_env(&mut candidate, &spec(Vec::new()));
11827        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11828        assert_eq!(
11829            role(&candidate),
11830            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11831        );
11832    }
11833
11834    /// Daemon-private capture retention keys never reach the child.
11835    ///
11836    /// cortexkit-log exposes retention as a Rust struct with no environment
11837    /// names, so these entries are supervisor metadata. Passing them through
11838    /// would invent a public child-process contract by accident.
11839    #[test]
11840    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11841        let mut command = Command::new("/nonexistent");
11842        apply_child_env(
11843            &mut command,
11844            &spec(vec![
11845                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11846                ("KEPT".to_string(), "yes".to_string()),
11847            ]),
11848        );
11849        let keys: Vec<String> = command
11850            .as_std()
11851            .get_envs()
11852            .filter(|(_, value)| value.is_some())
11853            .map(|(key, _)| key.to_string_lossy().into_owned())
11854            .collect();
11855        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11856        assert!(
11857            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11858            "daemon-private capture key leaked to the child: {keys:?}"
11859        );
11860    }
11861}
11862
11863#[cfg(test)]
11864mod jitter_tests {
11865    use super::jittered_health_delay;
11866    use std::{collections::HashSet, time::Duration};
11867
11868    /// Module ids drawn from a real fleet, so the dispersal claim is about names
11869    /// that actually occur rather than invented ones.
11870    ///
11871    /// This is a SAMPLE, not a registry: the property under test is that distinct
11872    /// ids disperse, which holds for any set of distinct strings. Several entries
11873    /// are already historical (modules get renamed), and that costs nothing here --
11874    /// but it means a reader must not mistake this for the live module set, and a
11875    /// rename sweep will match it without there being anything to change.
11876    const FLEET: [&str; 14] = [
11877        "aft",
11878        "alfonso-core",
11879        "magic-context",
11880        "broca",
11881        "thalamus",
11882        "quota",
11883        "engram",
11884        "plexus",
11885        "cerebellum",
11886        "astrocyte",
11887        "synapse",
11888        "subc-mcp",
11889        "cortexkit-credentials",
11890        "subc-federation",
11891    ];
11892
11893    /// Probes must not converge after a fleet-wide restart.
11894    ///
11895    /// This is the property the jitter exists for: every module reconnects at
11896    /// once, and without dispersal all fourteen would then probe on the same
11897    /// tick forever. Nothing failed visibly when this went untested -- a
11898    /// convergent fleet still probes correctly, just in a burst, so the symptom
11899    /// is a periodic load spike that looks like whatever else is running.
11900    #[test]
11901    fn probe_delays_disperse_across_the_fleet() {
11902        let cadence = Duration::from_secs(30);
11903        let delays: HashSet<Duration> = FLEET
11904            .iter()
11905            .map(|id| jittered_health_delay(id, 0, cadence))
11906            .collect();
11907        assert_eq!(
11908            delays.len(),
11909            FLEET.len(),
11910            "every supervised module must land on its own probe offset"
11911        );
11912    }
11913
11914    /// The offset may only ever DELAY a probe, never bring it forward.
11915    ///
11916    /// A delay below the cadence would probe a module more often than
11917    /// configured, which is the opposite of what an operator asked for and
11918    /// would tighten the failure budget without anyone changing it.
11919    #[test]
11920    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11921        let cadence = Duration::from_secs(30);
11922        let span = cadence / 10;
11923        for id in FLEET {
11924            for probe_index in 0..8 {
11925                let delay = jittered_health_delay(id, probe_index, cadence);
11926                assert!(
11927                    delay >= cadence,
11928                    "{id}#{probe_index}: jitter must not shorten the cadence"
11929                );
11930                assert!(
11931                    delay < cadence + span,
11932                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11933                );
11934            }
11935        }
11936    }
11937
11938    /// A module keeps its offset across daemon restarts.
11939    ///
11940    /// The delay is derived rather than randomised precisely so a restart does
11941    /// not re-roll every module into a fresh chance of collision. A random
11942    /// source would satisfy the dispersal test above and quietly lose this.
11943    #[test]
11944    fn a_module_offset_is_stable_across_restarts() {
11945        let cadence = Duration::from_secs(30);
11946        for id in FLEET {
11947            assert_eq!(
11948                jittered_health_delay(id, 0, cadence),
11949                jittered_health_delay(id, 0, cadence),
11950                "{id}: the same module and probe index must produce the same offset"
11951            );
11952        }
11953    }
11954
11955    /// A zero cadence disables probing rather than producing a busy loop.
11956    #[test]
11957    fn zero_cadence_yields_zero_delay() {
11958        assert_eq!(
11959            jittered_health_delay("aft", 0, Duration::ZERO),
11960            Duration::ZERO
11961        );
11962    }
11963}
11964
11965#[cfg(all(test, target_os = "linux"))]
11966mod cgroup_placement_tests {
11967    use super::{
11968        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11969        SupervisedChild,
11970    };
11971    use crate::stderr_tail::{StderrRing, StderrTailConfig};
11972    use std::{
11973        fs, io,
11974        path::{Path, PathBuf},
11975        sync::{Arc, Mutex},
11976    };
11977    use subc_test_support::TestTempDir;
11978    use tokio::process::Command;
11979
11980    #[tokio::test]
11981    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11982        use super::*;
11983        let dir = TestTempDir::new("unique-spawn-cgroups");
11984        let root = PathBuf::from(format!(
11985            "/sys/fs/cgroup/subc-unique-test-{}-{}",
11986            std::process::id(),
11987            unix_ms_now()
11988        ));
11989        if let Err(error) = fs::create_dir(&root) {
11990            assert!(
11991                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11992                "required cgroup test cannot execute: {error}"
11993            );
11994            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11995            return;
11996        }
11997        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11998        let group_count = || {
11999            fs::read_dir(root.join("subc-modules"))
12000                .unwrap()
12001                .map(|entry| entry.unwrap().file_type().unwrap())
12002                .filter(|kind| kind.is_dir())
12003                .count()
12004        };
12005        let supervisor =
12006            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
12007                .with_cgroup_placement(Some(placement.clone()));
12008        let runtime = supervisor.runtime_config();
12009        let mut spec = ModuleSpec {
12010            module_id: "unique-spawn".into(),
12011            program: PathBuf::from("/bin/sleep"),
12012            args: vec!["60".into()],
12013            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
12014                .into_iter()
12015                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
12016                .collect(),
12017            reserved: false,
12018            reserved_prefixes: vec![],
12019            protocol: ModuleProtocol::None,
12020            overlap: Default::default(),
12021        };
12022        let spawn = |spec: &ModuleSpec| {
12023            spawn_child(
12024                spec,
12025                None,
12026                None,
12027                &runtime.stderr_ring,
12028                None,
12029                &runtime.child_roster,
12030                Some(&placement),
12031            )
12032            .unwrap()
12033        };
12034        let mut live = spawn(&spec);
12035        for _ in 0..3 {
12036            // A new process can enter the old slot while retirement is pending.
12037            let next = spawn(&spec);
12038            assert_ne!(live.module_id, next.module_id);
12039            live.start_kill().unwrap();
12040            live.wait().await.unwrap();
12041            live = next;
12042            assert!(
12043                live.child.try_wait().unwrap().is_none(),
12044                "retiring the old slot must not kill the replacement"
12045            );
12046            assert_eq!(
12047                group_count(),
12048                1,
12049                "only the live spawn's cgroup should remain"
12050            );
12051        }
12052        supervisor.begin_daemon_shutdown();
12053        let reap = tokio::spawn(async move {
12054            live.wait().await.unwrap();
12055        });
12056        supervisor
12057            .end_children_for_daemon_shutdown(false, std::future::pending())
12058            .await;
12059        reap.await.unwrap();
12060        assert_eq!(group_count(), 0);
12061        // A normal exit uses the same tree-cleanup path as a killed spawn.
12062        spec.program = PathBuf::from("/bin/true");
12063        spec.args.clear();
12064        let fresh_roster = ChildRoster::default();
12065        let mut short = spawn_child(
12066            &spec,
12067            None,
12068            None,
12069            &runtime.stderr_ring,
12070            None,
12071            &fresh_roster,
12072            Some(&placement),
12073        )
12074        .unwrap();
12075        short.wait().await.unwrap();
12076        assert_eq!(group_count(), 0);
12077        spec.module_id = "_".repeat(255);
12078        let mut long_id = spawn_child(
12079            &spec,
12080            None,
12081            None,
12082            &runtime.stderr_ring,
12083            None,
12084            &fresh_roster,
12085            Some(&placement),
12086        )
12087        .unwrap();
12088        long_id.wait().await.unwrap();
12089        assert_eq!(
12090            group_count(),
12091            0,
12092            "valid long module IDs must not exceed cgroup NAME_MAX"
12093        );
12094        fs::remove_dir(root.join("subc-modules")).unwrap();
12095        fs::remove_dir(root).unwrap();
12096        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
12097    }
12098
12099    #[test]
12100    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
12101        let path = Path::new("/definitely-missing-subc-cgroup");
12102        let mut command = Command::new("true");
12103        let error = apply_cgroup_placement(
12104            &mut command,
12105            &ModuleSpec {
12106                module_id: "broken-cgroup".to_string(),
12107                program: PathBuf::from("true"),
12108                args: Vec::new(),
12109                env: Vec::new(),
12110                reserved: false,
12111                reserved_prefixes: Vec::new(),
12112                protocol: ModuleProtocol::Subc,
12113                overlap: Default::default(),
12114            },
12115            path,
12116        )
12117        .expect_err("a parent cgroup open failure must reject the supervised spawn");
12118        let reason = error.to_string();
12119
12120        assert!(
12121            matches!(error, SuperviseError::Cgroup { .. }),
12122            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
12123        );
12124        assert!(
12125            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
12126            "parent cgroup open failure must name cgroup.procs: {reason}"
12127        );
12128    }
12129
12130    #[tokio::test]
12131    async fn reaping_a_child_removes_its_empty_module_cgroup() {
12132        let root = TestTempDir::new("supervisor-reap-cgroup");
12133        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12134        let placement = subc_cgroup::prepare_at(&root)
12135            .expect("prepare scratch cgroup root")
12136            .expect("scratch root has a cgroup.procs marker");
12137        let module_id = "reaped-module";
12138        let module = placement
12139            .module_path(module_id)
12140            .expect("create scratch module cgroup");
12141        let child = Command::new("true")
12142            .env("XDG_DATA_HOME", root.path())
12143            .env("XDG_RUNTIME_DIR", root.path())
12144            .env("XDG_CONFIG_HOME", root.path())
12145            .spawn()
12146            .expect("spawn short-lived child");
12147        let pid = child.id().expect("spawned child has pid");
12148        let mut child = SupervisedChild {
12149            child,
12150            protocol: ModuleProtocol::Subc,
12151            module_id: module_id.to_string(),
12152            cgroup_placement: Some(placement),
12153            stdout_pump: None,
12154            stderr_pump: None,
12155            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
12156            spawned_at_ms: 0,
12157            spawned_from: PathBuf::from("true"),
12158            spawned_file_identity: None,
12159            process_start_time: None,
12160            process_identity: None,
12161            pid,
12162            roster_guard: None,
12163            #[cfg(target_os = "macos")]
12164            privacy_exec: None,
12165            spawn_failure: None,
12166        };
12167
12168        child.wait().await.expect("reap short-lived child");
12169
12170        assert!(
12171            !module.exists(),
12172            "reaping the supervised child must remove its empty cgroup"
12173        );
12174    }
12175
12176    #[test]
12177    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
12178        let root = TestTempDir::new("supervisor-non-empty-cgroup");
12179        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12180        let placement = subc_cgroup::prepare_at(&root)
12181            .expect("prepare scratch cgroup root")
12182            .expect("scratch root has a cgroup.procs marker");
12183        let module = placement
12184            .module_path("surviving-module")
12185            .expect("create scratch module cgroup");
12186        fs::write(module.join("surviving-process"), b"still present")
12187            .expect("make scratch cgroup non-empty");
12188        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
12189
12190        remove_module_cgroup(&placement, "surviving-module");
12191
12192        let logs = crate::router::test_log::captured_logs(&logs);
12193        assert!(
12194            module.exists(),
12195            "failed removal must leave the cgroup intact"
12196        );
12197        assert!(
12198            logs.contains("could not remove module cgroup after process exit; continuing teardown")
12199                && logs.contains("surviving-module"),
12200            "best-effort removal must report the failure without returning it: {logs}"
12201        );
12202    }
12203
12204    #[test]
12205    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
12206        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
12207        let reason = SuperviseError::Spawn {
12208            program: PathBuf::from("/bin/true"),
12209            source: io::Error::from_raw_os_error(13),
12210            cgroup_path: Some(cgroup_path.clone()),
12211        }
12212        .to_string();
12213
12214        assert!(
12215            reason.contains(&cgroup_path.display().to_string()),
12216            "a pre_exec spawn failure must name the cgroup path: {reason}"
12217        );
12218    }
12219}
12220
12221#[cfg(test)]
12222mod spawn_subscriber_lag_tests {
12223    use super::*;
12224
12225    /// A subscriber whose connection stops draining is dropped once its frame
12226    /// channel fills. The client must learn that from a terminal Error frame
12227    /// after the frames already queued for it, not from a stream that simply
12228    /// goes quiet.
12229    #[tokio::test]
12230    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
12231        let feed = SpawnEventFeed::default();
12232        feed.configure_incarnation("lag-incarnation".to_string());
12233        // A one-slot connection queue that nobody reads until the emits are
12234        // done: the forwarder parks on it and the subscriber channel fills.
12235        let (tx, mut rx) = mpsc::channel(1);
12236        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
12237            .expect("subscribe");
12238        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
12239        for index in 0..emitted {
12240            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
12241            // Let the forwarder take what it can so the fill point is the
12242            // subscriber channel, not a scheduling accident.
12243            tokio::task::yield_now().await;
12244        }
12245        assert_eq!(
12246            feed.subscriber_count(),
12247            0,
12248            "the lagged subscriber must be removed"
12249        );
12250
12251        let mut data = Vec::new();
12252        let mut last = None;
12253        loop {
12254            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
12255                .await
12256                .expect("the forwarder must finish once the subscriber is dropped");
12257            let Some(outbound) = next else { break };
12258            let frame = outbound.frame;
12259            if frame.header.ty == FrameType::StreamData {
12260                assert!(last.is_none(), "no data may follow the terminal frame");
12261                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
12262                data.push(event.cursor.seq);
12263            } else {
12264                assert!(last.is_none(), "exactly one terminal frame");
12265                last = Some(frame);
12266            }
12267        }
12268        assert!(!data.is_empty(), "queued frames drain before the terminal");
12269        for pair in data.windows(2) {
12270            assert_eq!(
12271                pair[1],
12272                pair[0] + 1,
12273                "queued frames arrive dense and in order"
12274            );
12275        }
12276        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
12277        assert_eq!(terminal.header.ty, FrameType::Error);
12278        assert_eq!(terminal.header.corr, 7);
12279        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
12280        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
12281        let detail = body.detail.expect("lagged error carries detail");
12282        assert_eq!(
12283            detail["first_undelivered_cursor"]["seq"],
12284            data.last().unwrap() + 1,
12285            "the named cursor is the first event the subscriber did not receive"
12286        );
12287        assert_eq!(
12288            detail["first_undelivered_cursor"]["daemon_incarnation"],
12289            "lag-incarnation"
12290        );
12291    }
12292}
12293
12294#[cfg(test)]
12295mod terminal_history_read_concurrency_tests {
12296    use super::*;
12297    use crate::terminal_journal::read_pause;
12298    use std::sync::mpsc as std_mpsc;
12299    use subc_test_support::TestTempDir;
12300
12301    fn journaled_ring(
12302        journal: &Arc<crate::terminal_journal::TerminalJournal>,
12303    ) -> Arc<Mutex<TerminalRing>> {
12304        Arc::new(Mutex::new(
12305            TerminalRing::new(TerminalRingConfig::default(), 1)
12306                .with_journal(Some(Arc::clone(journal))),
12307        ))
12308    }
12309
12310    fn crash(at_ms: u64) -> ExitReport {
12311        ExitReport {
12312            kind: ExitKind::Crash,
12313            code: Some(1),
12314            signal: None,
12315            at_ms,
12316        }
12317    }
12318
12319    /// Record an exit on another thread and report whether it finished within
12320    /// `bound`. The recorder thread is left running if it did not.
12321    fn record_within(
12322        module_id: &'static str,
12323        ring: &Arc<Mutex<TerminalRing>>,
12324        at_ms: u64,
12325        bound: Duration,
12326    ) -> bool {
12327        let ring = Arc::clone(ring);
12328        let (done, done_rx) = std_mpsc::channel();
12329        std::thread::spawn(move || {
12330            record_terminal(
12331                module_id,
12332                &ring,
12333                &SpawnEventFeed::default(),
12334                &crash(at_ms),
12335                TerminalDisposition::Restarting,
12336            );
12337            let _ = done.send(());
12338        });
12339        done_rx.recv_timeout(bound).is_ok()
12340    }
12341
12342    /// A history read in progress must not hold the journal writer (which every
12343    /// module's exit recording needs) or the module's own ring. Exits recorded
12344    /// while the read is paused complete promptly; the paused read answers as of
12345    /// the moment it started, and the next read has each exit exactly once.
12346    #[test]
12347    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12348        let dir = TestTempDir::new("terminal-history-concurrent-read");
12349        let path = dir.join("terminals.jsonl");
12350        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12351            path.clone(),
12352            "daemon".into(),
12353        ));
12354        let reader_ring = journaled_ring(&journal);
12355        let other_ring = journaled_ring(&journal);
12356        assert!(record_within(
12357            "reader-module",
12358            &reader_ring,
12359            10,
12360            Duration::from_secs(5)
12361        ));
12362
12363        let (started, release) = read_pause::install(&path);
12364        let reading = {
12365            let ring = Arc::clone(&reader_ring);
12366            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12367        };
12368        started
12369            .recv_timeout(Duration::from_secs(5))
12370            .expect("the history read reached its pause");
12371
12372        let bound = Duration::from_secs(1);
12373        assert!(
12374            record_within("other-module", &other_ring, 20, bound),
12375            "another module's exit waited on a history read (journal writer held)"
12376        );
12377        assert!(
12378            record_within("reader-module", &reader_ring, 30, bound),
12379            "the read module's own exit waited on its history read (ring held)"
12380        );
12381
12382        drop(release);
12383        let paused = reading.join().unwrap();
12384        assert_eq!(
12385            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12386            vec![10],
12387            "an exit recorded after the read began lands in neither half of it"
12388        );
12389        assert_eq!(paused.journal_skipped_lines, 0);
12390        assert_eq!(paused.journal_read_errors, 0);
12391
12392        let after = durable_terminal_history_of(&reader_ring, "reader-module");
12393        assert_eq!(
12394            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12395            vec![10, 30],
12396            "the next read merges ring and journal with no duplicate"
12397        );
12398        assert_eq!(after.journal_skipped_lines, 0);
12399    }
12400}
12401
12402/// What a restart does with the exited process's stderr reader. These drive
12403/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12404/// holds, so a reader that has not been scheduled by the bound is a controlled
12405/// input rather than something only a loaded machine produces.
12406#[cfg(test)]
12407mod stderr_settle_tests {
12408    use std::{
12409        future::Future,
12410        io,
12411        pin::Pin,
12412        sync::{Arc, Mutex},
12413        task::{Context, Poll},
12414        time::Duration,
12415    };
12416
12417    use tokio::{
12418        io::{AsyncRead, ReadBuf},
12419        sync::oneshot,
12420        time::Instant,
12421    };
12422
12423    use super::{settle_stderr_pump, StderrPump};
12424    use crate::stderr_tail::{
12425        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12426    };
12427
12428    const BOUND: Duration = Duration::from_millis(250);
12429
12430    /// Yields `before`, then stays pending until the gate is released, then
12431    /// yields `after` and reaches EOF. The bytes after the gate were written
12432    /// by a process that has already exited; only the reader is behind.
12433    struct HeldReader {
12434        before: Option<Vec<u8>>,
12435        gate: Option<oneshot::Receiver<()>>,
12436        after: io::Cursor<Vec<u8>>,
12437    }
12438
12439    impl AsyncRead for HeldReader {
12440        fn poll_read(
12441            mut self: Pin<&mut Self>,
12442            cx: &mut Context<'_>,
12443            buf: &mut ReadBuf<'_>,
12444        ) -> Poll<io::Result<()>> {
12445            if let Some(bytes) = self.before.take() {
12446                buf.put_slice(&bytes);
12447                return Poll::Ready(Ok(()));
12448            }
12449            if let Some(gate) = self.gate.as_mut() {
12450                match Pin::new(gate).poll(cx) {
12451                    Poll::Pending => return Poll::Pending,
12452                    Poll::Ready(_) => self.gate = None,
12453                }
12454            }
12455            Pin::new(&mut self.after).poll_read(cx, buf)
12456        }
12457    }
12458
12459    struct DiscardSink;
12460
12461    impl OutputSink for DiscardSink {
12462        fn write_line(&mut self, _line: &[u8]) {}
12463    }
12464
12465    fn line(text: &str) -> TailEntry {
12466        TailEntry::Line {
12467            text: text.to_string(),
12468            truncated: false,
12469            at_ms: None,
12470        }
12471    }
12472
12473    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
12474        ring.lock().unwrap()
12475    }
12476
12477    /// Start a reader for a new process generation that delivers `before`
12478    /// immediately and `after` only once the returned sender fires (or is
12479    /// dropped).
12480    fn held_pump(
12481        ring: &Arc<Mutex<StderrRing>>,
12482        before: &str,
12483        after: &str,
12484    ) -> (StderrPump, oneshot::Sender<()>) {
12485        let generation = lock(ring).begin_process();
12486        let (release, gate) = oneshot::channel();
12487        let reader = HeldReader {
12488            before: Some(before.as_bytes().to_vec()),
12489            gate: Some(gate),
12490            after: io::Cursor::new(after.as_bytes().to_vec()),
12491        };
12492        let task = tokio::spawn(pump_stderr_to(
12493            reader,
12494            Arc::clone(ring),
12495            generation,
12496            DiscardSink,
12497        ));
12498        (StderrPump { task, generation }, release)
12499    }
12500
12501    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
12502        for _ in 0..1000 {
12503            if done(&lock(ring)) {
12504                return;
12505            }
12506            tokio::time::sleep(Duration::from_millis(1)).await;
12507        }
12508        panic!(
12509            "ring never reached the expected state: {:?}",
12510            lock(ring).snapshot(None, None)
12511        );
12512    }
12513
12514    #[tokio::test(start_paused = true)]
12515    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
12516        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12517        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
12518
12519        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
12520        let before_release = lock(&ring).snapshot(None, None);
12521        assert!(
12522            matches!(before_release.capture, CaptureState::Incomplete { .. }),
12523            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
12524        );
12525
12526        // The restart: the next process starts and writes before the old
12527        // reader catches up.
12528        let next = lock(&ring).begin_process();
12529        lock(&ring).push_line_from(next, "next process booting");
12530        release.send(()).unwrap();
12531        wait_until(&ring, |ring| {
12532            ring.snapshot(None, None).capture == CaptureState::Captured
12533        })
12534        .await;
12535
12536        assert_eq!(
12537            untimed(lock(&ring).snapshot(None, None).entries),
12538            vec![
12539                line("booting"),
12540                line("config error: missing storage"),
12541                TailEntry::ProcessStart,
12542                line("next process booting"),
12543            ],
12544            "the crash's last line must survive a slow reader and stay in the crashed process's section"
12545        );
12546    }
12547
12548    #[tokio::test(start_paused = true)]
12549    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
12550    ) {
12551        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12552        // `_held` is never fired: a descendant keeps the pipe open for the
12553        // whole test.
12554        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
12555
12556        let started = Instant::now();
12557        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
12558        assert_eq!(
12559            started.elapsed(),
12560            BOUND,
12561            "the restart must wait exactly the bound for a pipe that stays open, no longer"
12562        );
12563
12564        let next = lock(&ring).begin_process();
12565        lock(&ring).push_line_from(next, "next process booting");
12566        tokio::time::sleep(Duration::from_secs(60)).await;
12567
12568        let snapshot = lock(&ring).snapshot(None, None);
12569        match &snapshot.capture {
12570            CaptureState::Incomplete { reason } => assert!(
12571                reason.contains("had not reached EOF") && reason.contains("250ms"),
12572                "the reason must say what is missing and after how long: {reason}"
12573            ),
12574            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
12575        }
12576        assert_eq!(
12577            untimed(snapshot.entries),
12578            vec![
12579                line("parent exiting"),
12580                TailEntry::ProcessStart,
12581                line("next process booting"),
12582            ]
12583        );
12584    }
12585
12586    #[tokio::test(start_paused = true)]
12587    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
12588        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12589        let (pump, release) = held_pump(&ring, "one\n", "two\n");
12590        release.send(()).unwrap();
12591
12592        settle_stderr_pump("clean", &ring, pump, BOUND).await;
12593
12594        let snapshot = lock(&ring).snapshot(None, None);
12595        assert_eq!(snapshot.capture, CaptureState::Captured);
12596        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
12597    }
12598}
12599
12600/// Containment of a module's process tree (issue #109).
12601///
12602/// The behaviour these defend against is a module helper surviving its module:
12603/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
12604/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
12605/// compounds it.
12606///
12607/// They run against the SUPERVISOR rather than the job-object crate because the
12608/// claim is about teardown: a crate-level test proves a job can reap a tree, not
12609/// that the daemon's drain path reaches it.
12610///
12611/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
12612/// lane there is a separate containment path with its own tests.
12613#[cfg(all(test, windows))]
12614mod job_containment_tests {
12615    use super::*;
12616    use std::{
12617        path::{Path, PathBuf},
12618        sync::{Arc, Mutex},
12619        time::{Duration, Instant},
12620    };
12621    use subc_test_support::TestTempDir;
12622
12623    /// The stub, expected beside this test executable.
12624    ///
12625    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
12626    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
12627    /// failure then reads as a broken test rather than an unbuilt dependency.
12628    fn stub_path() -> PathBuf {
12629        let mut path = std::env::current_exe().expect("current_exe available in tests");
12630        path.pop();
12631        path.pop();
12632        path.push("fake-aft-stub.exe");
12633        assert!(
12634            path.exists(),
12635            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
12636             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
12637            path.display()
12638        );
12639        path
12640    }
12641
12642    /// Poll for the grandchild pid the stub records, and parse it.
12643    fn read_grandchild_pid(path: &Path) -> u32 {
12644        let deadline = Instant::now() + Duration::from_secs(10);
12645        loop {
12646            if let Ok(contents) = std::fs::read_to_string(path) {
12647                if let Ok(pid) = contents.trim().parse() {
12648                    return pid;
12649                }
12650            }
12651            assert!(
12652                Instant::now() < deadline,
12653                "the stub never recorded a grandchild pid at {}",
12654                path.display()
12655            );
12656            std::thread::sleep(Duration::from_millis(10));
12657        }
12658    }
12659
12660    /// Everything one fixture run needs, so the two tests below differ in exactly
12661    /// one place: whether the child is contained.
12662    struct Fixture {
12663        _dir: TestTempDir,
12664        module_id: String,
12665        grandchild: u32,
12666        child: Option<SupervisedChild>,
12667        registry: Arc<Registry>,
12668        snapshot: Arc<Mutex<SupervisorSnapshot>>,
12669        terminal_ring: Arc<Mutex<TerminalRing>>,
12670        spawn_events: SpawnEventFeed,
12671    }
12672
12673    fn fixture(label: &str, module_id: &str) -> Fixture {
12674        let dir = TestTempDir::new(label);
12675        let pid_file = dir.join("grandchild.pid");
12676        let supervisor = Supervisor::new_for_test(
12677            Arc::new(Registry::default()),
12678            RestartPolicy::new(3, Duration::ZERO),
12679        );
12680        let runtime = supervisor.runtime_config();
12681        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12682        let spec = ModuleSpec {
12683            module_id: module_id.to_string(),
12684            program: stub_path(),
12685            // Zero args deliberately: a `--subc` argument would make the stub dial
12686            // a daemon that is not there, and the failure would land in the same
12687            // stderr ring this fixture exists to keep quiet.
12688            args: Vec::new(),
12689            env: vec![
12690                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
12691                (
12692                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
12693                    pid_file.display().to_string(),
12694                ),
12695            ],
12696            reserved: false,
12697            reserved_prefixes: Vec::new(),
12698            protocol: ModuleProtocol::Subc,
12699            overlap: Default::default(),
12700        };
12701        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
12702            .expect("spawn the supervised fixture");
12703        let grandchild = read_grandchild_pid(&pid_file);
12704        Fixture {
12705            _dir: dir,
12706            module_id: module_id.to_string(),
12707            grandchild,
12708            child: Some(child),
12709            registry: Arc::new(Registry::default()),
12710            snapshot,
12711            terminal_ring: Arc::clone(&runtime.terminal_ring),
12712            spawn_events: SpawnEventFeed::default(),
12713        }
12714    }
12715
12716    impl Fixture {
12717        /// Drain through the supervisor's own teardown path.
12718        async fn drain(&mut self) {
12719            let child = self
12720                .child
12721                .take()
12722                .expect("the fixture child is still present");
12723            drain_child_to_state(
12724                &self.module_id,
12725                ModuleProtocol::Subc,
12726                // No forwarding table in this fixture, so nothing reaches the
12727                // child over a connection.
12728                StopNotice::NotSent,
12729                &self.registry,
12730                None,
12731                &self.snapshot,
12732                &self.terminal_ring,
12733                &self.spawn_events,
12734                child,
12735                Duration::from_millis(500),
12736                ModuleState::Stopped,
12737                Some(false),
12738            )
12739            .await
12740            .expect("drain the supervised fixture");
12741        }
12742    }
12743
12744    /// Teardown reaps the grandchild, not merely the direct child.
12745    ///
12746    /// This is the assertion the change exists for. Before containment the
12747    /// grandchild survived: it is a separate process, and `start_kill` is
12748    /// `TerminateProcess` scoped to one pid.
12749    #[tokio::test]
12750    async fn teardown_reaps_the_grandchild() {
12751        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
12752        let grandchild = fixture.grandchild;
12753
12754        assert!(
12755            subc_jobobject::process_exists(grandchild),
12756            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
12757        );
12758
12759        fixture.drain().await;
12760
12761        assert!(
12762            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12763            "grandchild {grandchild} outlived module teardown: the tree was not contained"
12764        );
12765    }
12766
12767    /// The mutation control: with containment withheld, the grandchild survives
12768    /// the same kill.
12769    ///
12770    /// This is the defect reproduction from #109 — a direct-child kill reaches
12771    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
12772    /// supervisor because `spawn_and_mark_running` now always contains on
12773    /// Windows, which is the point: there is no longer a path that spawns
12774    /// uncontained, so the control has to construct one.
12775    ///
12776    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
12777    /// grandchild ever dies here, that test is passing for a reason unrelated to
12778    /// the job object and the containment claim is unproven.
12779    #[test]
12780    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12781        let dir = TestTempDir::new("teardown-uncontained");
12782        let pid_file = dir.join("grandchild.pid");
12783        let mut child = std::process::Command::new(stub_path())
12784            .env("FAKE_AFT_NEVER_CONNECT", "1")
12785            .env(
12786                "FAKE_AFT_GRANDCHILD_PID_FILE",
12787                pid_file.display().to_string(),
12788            )
12789            .stdin(std::process::Stdio::null())
12790            .stdout(std::process::Stdio::null())
12791            .stderr(std::process::Stdio::null())
12792            .spawn()
12793            .expect("spawn the uncontained fixture");
12794        let grandchild = read_grandchild_pid(&pid_file);
12795
12796        // Exactly what the pre-fix teardown did: kill the direct child.
12797        child.kill().expect("kill the direct child");
12798        let _ = child.wait();
12799
12800        assert!(
12801            subc_jobobject::process_exists(grandchild),
12802            "grandchild {grandchild} died with the direct child, so this control no longer \
12803             distinguishes contained from uncontained teardown and the regression test is \
12804             passing vacuously"
12805        );
12806
12807        // The orphan this control demonstrates is the leak the fix prevents, so
12808        // the control must not leave one behind.
12809        kill_tree(grandchild);
12810    }
12811
12812    /// Crash durability: closing the containment handle reaps the tree with no
12813    /// teardown code running at all.
12814    ///
12815    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
12816    /// call anything — and it is why containment is a kernel property of the
12817    /// handle rather than a step in the drain. Discovered by getting the
12818    /// mutation control wrong: clearing `job` to "disable" containment instead
12819    /// killed the tree, which is the guarantee, not a mistake.
12820    #[tokio::test]
12821    async fn dropping_containment_reaps_the_grandchild() {
12822        let mut fixture = fixture("drop-containment", "tree-drop");
12823        let grandchild = fixture.grandchild;
12824
12825        assert!(subc_jobobject::process_exists(grandchild));
12826
12827        // No `drain` call, no kill: dropping the handle is the entire mechanism.
12828        fixture.child.as_mut().expect("child present").job = None;
12829
12830        assert!(
12831            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12832            "grandchild {grandchild} survived the containment handle closing, so a daemon \
12833             crash would leave the tree behind"
12834        );
12835    }
12836
12837    /// Kill a pid and its tree, then confirm it is gone.
12838    fn kill_tree(pid: u32) {
12839        let _ = std::process::Command::new("taskkill.exe")
12840            .args(["/PID", &pid.to_string(), "/T", "/F"])
12841            .stdin(std::process::Stdio::null())
12842            .stdout(std::process::Stdio::null())
12843            .stderr(std::process::Stdio::null())
12844            .status();
12845        assert!(
12846            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12847            "could not clean up grandchild {pid}"
12848        );
12849    }
12850}
12851
12852#[cfg(test)]
12853mod privacy_trampoline_configuration_tests {
12854    #[tokio::test]
12855    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12856    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
12857        #[cfg(target_os = "macos")]
12858        {
12859            let supervisor = super::Supervisor::new(
12860                std::sync::Arc::new(crate::Registry::default()),
12861                super::RestartPolicy::default(),
12862            );
12863            let error = supervisor.spawn(spec()).unwrap_err();
12864            assert!(
12865                error
12866                    .to_string()
12867                    .contains("no privacy trampoline configured"),
12868                "{error}"
12869            );
12870        }
12871    }
12872
12873    #[tokio::test]
12874    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12875    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
12876        #[cfg(target_os = "macos")]
12877        {
12878            let supervisor = super::Supervisor::new(
12879                std::sync::Arc::new(crate::Registry::default()),
12880                super::RestartPolicy::default(),
12881            )
12882            .with_privacy_trampoline(std::env::current_exe().unwrap());
12883            let error = supervisor.spawn(spec()).unwrap_err();
12884            assert!(
12885                error
12886                    .to_string()
12887                    .contains("binary does not implement the privacy trampoline protocol"),
12888                "{error}"
12889            );
12890        }
12891    }
12892
12893    #[cfg(target_os = "macos")]
12894    fn spec() -> super::ModuleSpec {
12895        super::ModuleSpec {
12896            module_id: "privacy-configuration".into(),
12897            program: "/bin/sleep".into(),
12898            args: vec!["30".into()],
12899            env: vec![],
12900            reserved: false,
12901            reserved_prefixes: vec![],
12902            protocol: subc_control::ModuleProtocol::None,
12903            overlap: super::ModuleOverlap::Exclusive,
12904        }
12905    }
12906}
12907
12908#[cfg(test)]
12909mod privacy_exec_boundary_tests {
12910    #[cfg(target_os = "macos")]
12911    use super::*;
12912    #[cfg(target_os = "macos")]
12913    use std::{
12914        io::{Read, Write},
12915        net::{TcpListener, TcpStream},
12916    };
12917
12918    /// Unit-test-only pause at the actual early image read, not at a later
12919    /// status read. Production supervisors never inspect this environment key.
12920    #[cfg(target_os = "macos")]
12921    pub(super) fn before_image_sample(spec: &ModuleSpec, pid: u32) {
12922        if let Some((_, path)) = spec
12923            .env
12924            .iter()
12925            .find(|(key, _)| key == "SUBC_TEST_PRIVACY_SAMPLE_BARRIER")
12926        {
12927            let mut barrier = TcpStream::connect(path).unwrap();
12928            barrier
12929                .set_read_timeout(Some(Duration::from_secs(30)))
12930                .unwrap();
12931            barrier.write_all(&pid.to_ne_bytes()).unwrap();
12932            let mut release = [0];
12933            barrier.read_exact(&mut release).unwrap();
12934            assert_eq!(&release, b"X");
12935        }
12936    }
12937
12938    #[cfg(target_os = "macos")]
12939    fn spec(program: &str, args: &[&str]) -> ModuleSpec {
12940        ModuleSpec {
12941            module_id: "privacy-boundary".into(),
12942            program: program.into(),
12943            args: args.iter().map(|arg| (*arg).into()).collect(),
12944            env: vec![],
12945            reserved: false,
12946            reserved_prefixes: vec![],
12947            protocol: ModuleProtocol::None,
12948            overlap: ModuleOverlap::Exclusive,
12949        }
12950    }
12951
12952    #[cfg(target_os = "macos")]
12953    async fn accept(listener: TcpListener) -> TcpStream {
12954        // Socket readiness, not elapsed time, establishes both pause points.
12955        let listener = tokio::net::TcpListener::from_std({
12956            listener.set_nonblocking(true).unwrap();
12957            listener
12958        })
12959        .unwrap();
12960        let (stream, _) = tokio::time::timeout(Duration::from_secs(30), listener.accept())
12961            .await
12962            .unwrap()
12963            .unwrap();
12964        let stream = stream.into_std().unwrap();
12965        stream.set_nonblocking(false).unwrap();
12966        stream
12967            .set_read_timeout(Some(Duration::from_secs(30)))
12968            .unwrap();
12969        stream
12970    }
12971
12972    #[tokio::test]
12973    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12974    async fn macos_roster_withholds_a_nonnull_trampoline_image_until_exec_confirmation() {
12975        #[cfg(target_os = "macos")]
12976        {
12977            let root = subc_test_support::TestTempDir::new("privacy-roster-barrier");
12978            // Loopback sockets also work when the replay adapter's TMPDIR is
12979            // longer than Darwin's Unix-domain socket path limit.
12980            let exec_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12981            let sample_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12982            let record = root.join("live-children.json");
12983            let supervisor =
12984                Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
12985                    .with_live_children_record(&record);
12986            let runtime = supervisor.runtime_config();
12987            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12988            let mut spec = spec("/bin/sleep", &["30"]);
12989            spec.env = vec![
12990                (
12991                    "SUBC_TEST_PRIVACY_EXEC_BARRIER".into(),
12992                    exec_listener.local_addr().unwrap().to_string(),
12993                ),
12994                (
12995                    "SUBC_TEST_PRIVACY_SAMPLE_BARRIER".into(),
12996                    sample_listener.local_addr().unwrap().to_string(),
12997                ),
12998            ];
12999            let spawn = tokio::task::spawn_blocking(move || {
13000                spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap()
13001            });
13002            let mut sample = accept(sample_listener).await;
13003            let mut pid = [0; 4];
13004            sample.read_exact(&mut pid).unwrap();
13005            let pid = u32::from_ne_bytes(pid);
13006            let mut exec = accept(exec_listener).await;
13007            let mut ready = [0];
13008            exec.read_exact(&mut ready).unwrap();
13009            assert_eq!(&ready, b"R");
13010            // The early read is guaranteed to see a real, non-null trampoline
13011            // image: the fixture has reached its barrier and cannot exec yet.
13012            let trampoline = subc_os::file_identity(&test_privacy_trampoline()).unwrap();
13013            assert_eq!(
13014                observe_spawned_image(pid).unwrap().executable,
13015                Some(trampoline)
13016            );
13017            sample.write_all(b"X").unwrap();
13018            let mut child = spawn.await.unwrap();
13019            let early = crate::live_children::read_record(&record).unwrap();
13020            assert_eq!(early.len(), 1);
13021            assert_eq!(early[0].pid, pid);
13022            assert_eq!(
13023                early[0].executable, None,
13024                "unconfirmed trampoline image entered the roster"
13025            );
13026            assert!(child.report_ready.get().is_none());
13027            // The barrier's duration is unrelated to the production five-second
13028            // exec budget. Start the test's confirmation budget upon release.
13029            child.privacy_exec.as_mut().unwrap().deadline =
13030                tokio::time::Instant::now() + Duration::from_secs(30);
13031            exec.write_all(b"X").unwrap();
13032            child.confirm_privacy_exec().await;
13033            assert_eq!(child.spawn_failure, None);
13034            assert!(child.report_ready.get().is_some());
13035            let confirmed = crate::live_children::read_record(&record).unwrap();
13036            let module = subc_os::file_identity(std::path::Path::new("/bin/sleep")).unwrap();
13037            assert_ne!(module, trampoline);
13038            assert_eq!(confirmed[0].executable, Some(module.into()));
13039            child.start_kill().unwrap();
13040            child.wait().await.unwrap();
13041            child.release_roster();
13042        }
13043    }
13044
13045    #[tokio::test]
13046    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13047    async fn macos_already_exited_121_in_try_wait_is_a_module_exit_not_a_trampoline_refusal() {
13048        #[cfg(target_os = "macos")]
13049        {
13050            let registry = Arc::new(Registry::default());
13051            let policy = RestartPolicy::new(0, Duration::ZERO);
13052            let supervisor = Supervisor::new_for_test(Arc::clone(&registry), policy);
13053            let runtime = supervisor.runtime_config();
13054            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13055            let spec = spec("/bin/sh", &["-c", "exit 121"]);
13056            // Drive spawn and confirmation separately instead of starting the
13057            // monitor. WNOWAIT observes a real exit without consuming its status,
13058            // so confirmation's first try_wait must take the already-exited arm.
13059            let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
13060            let pid = child.pid;
13061            tokio::task::spawn_blocking(move || {
13062                subc_os::fork_exec_test::wait_for_child_exit_without_reaping(pid)
13063            })
13064            .await
13065            .unwrap()
13066            .unwrap();
13067            child.privacy_exec.as_mut().unwrap().deadline =
13068                tokio::time::Instant::now() + Duration::from_secs(30);
13069            let status = child.wait().await.unwrap();
13070            assert_eq!(status.code(), Some(121));
13071            assert!(child.privacy_exec.is_none());
13072            assert!(
13073                child.report_ready.get().is_none(),
13074                "an exited module must not publish a live pid"
13075            );
13076            let report = classify_reaped_child_exit(&snapshot, &child, &status);
13077            on_child_exit(
13078                &spec,
13079                policy,
13080                &registry,
13081                &snapshot,
13082                &runtime.terminal_ring,
13083                &runtime.spawn_events,
13084                &runtime.child_roster,
13085                report,
13086            )
13087            .await;
13088            let state = lock_snapshot(&snapshot).unwrap();
13089            assert_eq!(state.state, ModuleState::Failed);
13090            assert_eq!(state.last_exit.as_ref().unwrap().code, Some(121));
13091            assert_eq!(state.reported_pid(), None);
13092            drop(state);
13093            let history = runtime.terminal_ring.lock().unwrap().snapshot();
13094            assert_eq!(history.entries.len(), 1);
13095            let terminal = &history.entries[0];
13096            assert_eq!(terminal.exit_code, Some(121));
13097            assert_eq!(terminal.exit_kind, subc_control::TerminalExitKind::Crash);
13098            assert_eq!(terminal.disposition, TerminalDisposition::Failed);
13099            assert_eq!(
13100                terminal.disposition_detail.as_deref(),
13101                Some(policy.budget_exhausted_detail().as_str()),
13102                "module exit 121 was classified as a trampoline refusal: {terminal:?}"
13103            );
13104            assert_eq!(child.spawn_failure, None);
13105            child.release_roster();
13106        }
13107    }
13108}
13109
13110/// The daemon's real spawn path hands a subc-wire child its launch nonce on
13111/// descriptor 3, without an environment copy. The shell records the nonce
13112/// and its environment after exec so these tests observe the real handover.
13113#[cfg(all(test, unix))]
13114mod launch_nonce_descriptor_tests {
13115    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
13116    use crate::stderr_tail::{StderrRing, StderrTailConfig};
13117    use std::{
13118        path::PathBuf,
13119        sync::{Arc, Mutex},
13120        time::{Duration, Instant},
13121    };
13122    use subc_test_support::TestTempDir;
13123
13124    async fn probe(role: super::SpawnRole) {
13125        let scratch = TestTempDir::new("launch-nonce-descriptor");
13126        let fd_copy = scratch.join("from-descriptor");
13127        let env_copy = scratch.join("environment");
13128        let script = format!(
13129            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
13130            fd = fd_copy.display(), env = env_copy.display(),
13131        );
13132        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
13133        let spec = ModuleSpec {
13134            module_id: "nonce-descriptor-probe".to_string(),
13135            program: PathBuf::from("/bin/sh"),
13136            args: vec!["-c".to_string(), script],
13137            env: vec![
13138                xdg("XDG_DATA_HOME"),
13139                xdg("XDG_RUNTIME_DIR"),
13140                xdg("XDG_CONFIG_HOME"),
13141                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
13142            ],
13143            reserved: true,
13144            reserved_prefixes: Vec::new(),
13145            protocol: ModuleProtocol::Subc,
13146            overlap: Default::default(),
13147        };
13148        let handle = SupervisorHandle::new();
13149        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13150        let roster = ChildRoster::default();
13151        #[cfg(target_os = "macos")]
13152        {
13153            let path = super::test_privacy_trampoline();
13154            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
13155        }
13156        let child = super::spawn_child_in_slot(
13157            &spec,
13158            None,
13159            Some(&handle),
13160            &ring,
13161            None,
13162            &roster,
13163            #[cfg(target_os = "linux")]
13164            None,
13165            role,
13166            matches!(role, super::SpawnRole::SwapCandidate),
13167        )
13168        .expect("spawn probe");
13169        let deadline = Instant::now() + Duration::from_secs(10);
13170        while !(fd_copy.exists() && env_copy.exists()) {
13171            assert!(Instant::now() < deadline, "probe never wrote its copies");
13172            tokio::time::sleep(Duration::from_millis(20)).await;
13173        }
13174        let nonce = std::fs::read_to_string(fd_copy).unwrap();
13175        assert!(!nonce.is_empty());
13176        let environment = std::fs::read_to_string(env_copy).unwrap();
13177        assert!(environment
13178            .lines()
13179            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
13180        let copy = environment
13181            .lines()
13182            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
13183        assert_eq!(
13184            copy, None,
13185            "Unix children must never receive the environment nonce"
13186        );
13187        if matches!(role, super::SpawnRole::Plain) {
13188            assert_eq!(
13189                handle.spawn_nonce(&spec.module_id).as_deref(),
13190                Some(nonce.as_str())
13191            );
13192        }
13193        drop(child);
13194    }
13195
13196    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13197    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
13198        probe(super::SpawnRole::Plain).await;
13199    }
13200
13201    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13202    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
13203        probe(super::SpawnRole::SwapCandidate).await;
13204    }
13205}
13206
13207#[cfg(all(test, target_os = "linux"))]
13208mod cgroup_containment_tests {
13209    use super::*;
13210    use subc_test_support::TestTempDir;
13211
13212    fn running(pid: u32) -> bool {
13213        // An orphan can remain a zombie until the container init reaps it.
13214        std::fs::read_to_string(format!("/proc/{pid}/stat"))
13215            .ok()
13216            .and_then(|stat| {
13217                stat.rsplit_once(") ")
13218                    .map(|(_, rest)| rest.starts_with('Z'))
13219            })
13220            .is_some_and(|zombie| !zombie)
13221    }
13222
13223    #[tokio::test]
13224    async fn linux_teardown_reaps_the_grandchild() {
13225        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
13226    }
13227
13228    #[tokio::test]
13229    async fn linux_shutdown_straggler_reaps_the_grandchild() {
13230        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
13231    }
13232
13233    async fn teardown_tree(test_name: &str, shutdown: bool) {
13234        let dir = TestTempDir::new(test_name);
13235        let root = PathBuf::from(format!(
13236            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
13237            std::process::id(),
13238            unix_ms_now()
13239        ));
13240        if let Err(error) = std::fs::create_dir(&root) {
13241            assert!(
13242                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13243                "required cgroup test cannot execute: {error}"
13244            );
13245            eprintln!(
13246                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
13247                root.display()
13248            );
13249            return;
13250        }
13251        let placement = subc_cgroup::prepare_at(&root)
13252            .expect("prepare isolated kernel cgroup")
13253            .expect("isolated cgroup is delegated");
13254        let module_id = "tree-teardown";
13255        let module = placement
13256            .module_path(module_id)
13257            .expect("create isolated module cgroup");
13258        if !module.join("cgroup.kill").exists() {
13259            std::fs::remove_dir(&module).unwrap();
13260            std::fs::remove_dir(root.join("subc-modules")).unwrap();
13261            std::fs::remove_dir(&root).unwrap();
13262            assert!(
13263                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13264                "required cgroup.kill interface unavailable"
13265            );
13266            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
13267            return;
13268        }
13269        let supervisor = Supervisor::new_for_test(
13270            Arc::new(Registry::default()),
13271            RestartPolicy::new(3, Duration::ZERO),
13272        )
13273        .with_cgroup_placement(Some(placement));
13274        let mut runtime = supervisor.runtime_config();
13275        runtime.child_roster = runtime
13276            .child_roster
13277            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
13278        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13279        let pid_file = dir.join("grandchild.pid");
13280        let spec = ModuleSpec {
13281            module_id: module_id.to_string(),
13282            program: PathBuf::from("/bin/sh"),
13283            args: vec![
13284                "-c".into(),
13285                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
13286                "fixture".into(),
13287                pid_file.display().to_string(),
13288            ],
13289            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
13290                .into_iter()
13291                .map(|key| (key.to_string(), dir.display().to_string()))
13292                .collect(),
13293            reserved: false,
13294            reserved_prefixes: Vec::new(),
13295            protocol: ModuleProtocol::None,
13296            overlap: Default::default(),
13297        };
13298        let child =
13299            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
13300        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
13301        let grandchild: u32 = loop {
13302            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
13303                if let Ok(pid) = contents.trim().parse() {
13304                    break pid;
13305                }
13306            }
13307            assert!(
13308                tokio::time::Instant::now() < deadline,
13309                "grandchild pid was not recorded"
13310            );
13311            tokio::time::sleep(Duration::from_millis(10)).await;
13312        };
13313        assert!(
13314            running(grandchild),
13315            "grandchild must be alive before teardown"
13316        );
13317        if shutdown {
13318            let mut child = child;
13319            crate::child_roster::end_children_for_daemon_shutdown(
13320                &runtime.child_roster,
13321                false,
13322                std::future::pending(),
13323            )
13324            .await;
13325            child.wait().await.expect("reap shutdown straggler");
13326        } else {
13327            drain_child_to_state(
13328                module_id,
13329                ModuleProtocol::None,
13330                StopNotice::NotSent,
13331                &Registry::default(),
13332                None,
13333                &snapshot,
13334                &runtime.terminal_ring,
13335                &SpawnEventFeed::default(),
13336                child,
13337                Duration::from_millis(100),
13338                ModuleState::Stopped,
13339                Some(false),
13340            )
13341            .await
13342            .expect("real supervisor teardown");
13343        }
13344        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
13345        while running(grandchild) && tokio::time::Instant::now() < deadline {
13346            tokio::time::sleep(Duration::from_millis(10)).await;
13347        }
13348        let survived = running(grandchild);
13349        // Kill a surviving grandchild so a failed test does not leave it behind.
13350        if survived {
13351            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
13352            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
13353            tokio::time::sleep(Duration::from_millis(100)).await;
13354        }
13355        if module.exists() {
13356            std::fs::remove_dir(&module).expect("remove empty module cgroup");
13357        }
13358        std::fs::remove_dir(root.join("subc-modules")).unwrap();
13359        std::fs::remove_dir(&root).unwrap();
13360        assert!(
13361            !survived,
13362            "grandchild {grandchild} outlived module teardown"
13363        );
13364        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
13365    }
13366}